diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 83ce1a8..4021e6b 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -5,14 +5,14 @@ }, "metadata": { "description": "DevSquad - Engineering Manager that coordinates AI coding agents", - "version": "0.10.0" + "version": "0.11.0" }, "plugins": [ { "name": "devsquad", "source": "./plugin", "description": "Engineering Manager that coordinates AI coding agents through enforced delegation", - "version": "0.10.0" + "version": "0.11.0" } ] -} \ No newline at end of file +} diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..643ee9f --- /dev/null +++ b/.env.example @@ -0,0 +1,4 @@ +# Copy to .env, set permissions to 600, and fill the key locally. +# Never paste credentials into chat or commit them. +# The Jev probe reads this only with --env-file .env; runtime routing stays off. +TYPESAFE_API_KEY= diff --git a/.github/workflows/offline.yml b/.github/workflows/offline.yml new file mode 100644 index 0000000..2d23111 --- /dev/null +++ b/.github/workflows/offline.yml @@ -0,0 +1,68 @@ +name: Offline compatibility + +on: + push: + pull_request: + workflow_dispatch: + +permissions: + contents: read + +jobs: + legacy-bash: + name: Legacy Bash 3.2 (macOS) + runs-on: macos-latest + steps: + - uses: actions/checkout@v4 + - name: Verify system Bash floor + run: test "$(/bin/bash -c 'printf "%s.%s" "${BASH_VERSINFO[0]}" "${BASH_VERSINFO[1]}"')" = "3.2" + - name: Run legacy and packaging contracts + run: /bin/bash test/run.sh + + core: + name: Core Python ${{ matrix.python }} (macOS) + runs-on: macos-latest + strategy: + fail-fast: false + matrix: + python: ["3.11", "3.14"] + steps: + - uses: actions/checkout@v4 + with: + fetch-depth: 0 + - uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python }} + cache: pip + - name: Install build-only test tooling + run: python -m pip install "setuptools>=68" wheel + - name: Verify generated command and schema reference + run: python scripts/generate-core-reference.py --check + - name: Run dependency-free core, package and native-protocol fixtures + env: + PYTHONDONTWRITEBYTECODE: "1" + PYTHONPATH: plugin/core/src + PYTHONWARNINGS: error::ResourceWarning + run: python scripts/run-core-tests.py --failfast + + optional-mcp: + name: Optional MCP 2.2.0 (provider-offline) + runs-on: macos-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.11" + cache: pip + - name: Install exact optional transport lock + run: python -m pip install -r plugin/core/requirements-mcp.lock + - name: Run MCP SDK and stdio compatibility fixtures + env: + PYTHONDONTWRITEBYTECODE: "1" + PYTHONPATH: plugin/core/src + PYTHONWARNINGS: error::ResourceWarning + run: python test/core/test_mcp.py -q + +# Test bodies make no provider or application network calls. Dependency setup +# resolves pinned public packages; live provider/app smoke runs remain explicit, +# bounded receipts outside CI. diff --git a/.gitignore b/.gitignore index cd007e9..dbdade6 100644 --- a/.gitignore +++ b/.gitignore @@ -4,8 +4,18 @@ .claude/ .agent/ +# Local credentials; commit only the blank example. +.env +.env.* +!.env.example + # Dependencies node_modules/ +__pycache__/ +*.py[cod] +.pytest_cache/ +*.egg-info/ +plugin/core/build/ # Logs *.log diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..76a3c53 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,31 @@ +# Working on DevSquad + +Read [CONTRIBUTING.md](CONTRIBUTING.md) before changing code. Run +`bash test/run.sh` before every commit; core changes also need the relevant +offline Python tests. Preserve the existing Bash 3.2 and optional-jq contracts. + +## Continuing the engineering-team build + +This build is already underway. After a usage limit, interrupted session or +handoff, read these files before starting work: + +1. [RESUME.md](docs/plans/engineering-team/RESUME.md): latest checkpoint, + verified results, open findings and exact next action. +2. [backlog.json](docs/plans/engineering-team/backlog.json): milestone status + and evidence. +3. [SOL-HANDOFF.md](docs/plans/engineering-team/SOL-HANDOFF.md): the full + authorized delivery scope and acceptance criteria. + +Compare the recovery note with `git status` and recent commits; retain work +newer than the note. Continue on `codex/engineering-team` unless the user +directs otherwise. Do not restart the architecture exercise or discard +implementation to return to `main`. + +Checkpoint coherent partial work and update RESUME.md before long probes or +handoffs. Record failures and incomplete gates truthfully. Finish a pause with +a clean tree; never use a stash as the recovery mechanism. Keep raw provider +diagnostics and credentials outside tracked evidence. + +Usage limits do not authorize buying credits, redeeming reset credits, using +a paid API fallback or changing global AI settings. The saved artifacts must +allow the next session to resume without relying on conversation memory. diff --git a/CHANGELOG.md b/CHANGELOG.md index 3e682fe..51d7da0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,39 @@ All notable changes to DevSquad are documented here. > **Note:** Project was renumbered from 2.x to 0.x semver in Feb 2026 to reflect pre-stable status. Entries below have been renumbered accordingly. +## [0.11.0] — 2026-10-02 + +### Added +- Standalone dependency-free Python core (`squad` 0.1.0) with durable saved + runs, isolated candidates, bounded Claude implementation, independent Codex + review, candidate-bound checks and fenced host acceptance. +- Guided setup, readiness, review/fix, status/result, finish, cancel and safe + resume commands; optional pinned MCP transport for the same local ledger. +- Catalog-backed profiles, shared subscription-pool capacity fences, explicit + outcome/trial reports and guarded qualification/promotion/rollback. +- Reversible content-addressed installation and guarded schema-17 upgrade. + The wheel packages every runtime schema, including the experimental Council + contract. Old schema-16 clients cannot mutate the upgraded ledger. + +### Fixed +- Interrupted guided completion now saves exact intent and recovers only + matching authority, retaining prior expiry rejection history. +- Native probes fail closed without process identity. Bounded natural exit + preserves version/auth exit status before owned cleanup; controlled Python + fixtures cannot invalidate a pristine candidate with undeclared bytecode. +- Bash 3.2 compatibility, optional-jq behavior and legacy wrapper error + contracts remain supported. + +### Limitations +- Council is experimental and native-unavailable; automatic invocation is OFF. + Its fixture comparison is inconclusive, not a quality or savings claim. +- Jev routing remains OFF. Provider operations depend on installed harness + capabilities and subscription authentication; paid API fallback is not used. +- Claude CLI handoff and Claude-to-Codex delivery have live receipts. Claude + desktop Code-tab proof is separate and unverified. Antigravity means the + `agy` CLI, not its IDE; Grok/Antigravity MCP status does not prove automatic + implementation/review roles. See the runtime guide for supported boundaries. + ## [0.10.0] — 2026-07-07 ### Maintainability & scalability review diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index aac70ea..b46285a 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -6,6 +6,9 @@ git clone https://github.com/joshidikshant/devsquad.git cd devsquad bash test/run.sh # no network, no real CLIs required — should be all green +PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src \ + python3 scripts/run-core-tests.py +python3 scripts/generate-core-reference.py --check ``` The test suite (`test/run.sh`) is the contract. It runs offline against fake @@ -54,3 +57,10 @@ Bump the version in `plugin/.claude-plugin/plugin.json` and both entries in `.claude-plugin/marketplace.json`, add a `CHANGELOG.md` entry, then after pushing run `claude plugin update devsquad@devsquad-marketplace`. Never point hook commands at a versioned cache dir — that freezes hooks at install time. + +Before a release, also run a fresh standalone install into temporary +`DEVSQUAD_INSTALL_ROOT` and `DEVSQUAD_BIN_DIR` locations. The tracked installer +tests cover Claude-free installation, idempotence, actual payload drift, +plugin contents and an active run surviving a release switch. The operator +commands and supported/deferred surface boundaries live in +[docs/RUNTIME-GUIDE.md](docs/RUNTIME-GUIDE.md). diff --git a/README.md b/README.md index 68169e8..c2b42b4 100644 --- a/README.md +++ b/README.md @@ -2,11 +2,13 @@ # DevSquad -### Your AI coding agent ignores your rules. Hooks don't. +### A local engineering team with saved work and verifiable results -**DevSquad turns Claude Code into an engineering manager that _physically intercepts_ tool calls and routes the grunt work to Gemini, Codex, and Grok — then runs a live A/B test on whether that even helps.** +**DevSquad coordinates bounded implementation, independent review, tests and +host acceptance across your installed AI harnesses. Runs and evidence survive +closing a client. The standalone runtime does not require Claude's plugin.** -[![tests](https://img.shields.io/badge/tests-177%20passing-brightgreen)](test/) +[![tests](https://github.com/joshidikshant/devsquad/actions/workflows/offline.yml/badge.svg)](https://github.com/joshidikshant/devsquad/actions/workflows/offline.yml) [![bash](https://img.shields.io/badge/bash-3.2%2B-blue)](CONTRIBUTING.md) [![jq](https://img.shields.io/badge/jq-optional-blue)](CONTRIBUTING.md) [![license](https://img.shields.io/badge/license-MIT-black)](LICENSE) @@ -22,6 +24,48 @@ --- +## Standalone runtime quickstart + +Requires a Unix-like host, Python 3.11+ and existing provider CLI subscription +logins. Installation is local and does not download Python dependencies: + +```bash +git clone https://github.com/joshidikshant/devsquad.git +cd devsquad +./install.sh --core-only +export PATH="$HOME/.local/bin:$PATH" +squad doctor +squad review --base main --dry-run +``` + +In the project you want to work on, commit the inputs, inspect the dry run, +then use `squad review --base main --wait` or +`squad fix "the bounded issue" --write-path src --wait`. The saved run stops +at a host handoff with checks and the exact finish command; assess the evidence +before accepting. `squad status`, `squad result RUN_ID`, `squad cancel RUN_ID` +and `squad resume RUN_ID` operate on the same saved work. + +For Codex/Claude/Antigravity CLI/Grok MCP setup, safe updates, supported +operations and troubleshooting, read the [runtime guide](docs/RUNTIME-GUIDE.md). +The optional MCP transport needs an explicitly prepared pinned wheelhouse; +it is not silently downloaded. Existing paid accounts do not imply that every +harness supports every worker role. No paid API fallback is enabled. + +Council is native-unavailable and automatic use is OFF. Jev routing is OFF; +no production quality, cost savings or Plus-window savings are claimed. +Claude CLI handoff and implementation → independent Codex review → tests have +live receipts; Claude desktop Code-tab proof remains unverified. Antigravity +means `agy`, not the IDE. + +Continuing development after an interruption? Read the +[recovery checkpoint](docs/plans/engineering-team/RESUME.md) and +[implementation plan](docs/plans/engineering-team/START-HERE.md). + +## Legacy Claude plugin + +The sections below describe the separately supported Claude hook plugin, +not automatic worker-role coverage in the standalone runtime. + ## The 30-second version You told Claude to delegate the boring stuff. It nodded. Then it read 40 files itself, blew through its context window, and you paid for every token. diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 4448f30..853e1ac 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -2,6 +2,14 @@ One page for future maintainers (including future Claude sessions). +> **September 2026:** The diagram below describes the legacy Claude plugin; +> provider names, deployment details and measurements are historical. The +> [current assessment](audits/2026-09-06-engineering-team-assessment.md) records +> verified gaps. The proposed shared runner is specified in +> [ADR-002](adr/ADR-002-surface-independent-engineering-team.md), with a +> [coding-agent execution packet](plans/engineering-team/START-HERE.md). +> That new runtime is planned, not implemented. + ## End-to-end flow ```mermaid @@ -118,16 +126,26 @@ Enforced by `test/test_wrapper_contract.sh` (offline, fake CLI binaries): ## Deployment modes (the drift trap) -- **User mode**: `install.sh` registers global hooks pointing at the - marketplace clone (`~/.claude/plugins/marketplaces/devsquad-marketplace/plugin`). - Refresh with `claude plugin marketplace update devsquad-marketplace` + - `claude plugin update devsquad@devsquad-marketplace` after each release. -- **Dev mode** (this machine): global hooks point at the source checkout, so - hooks run at HEAD. Agents/commands/skills STILL load from the installed - plugin — after pushing, update the plugin or subagents run stale code. -- Never let hook commands reference a versioned cache dir - (`plugins/cache/...//`): that froze production at 0.3.0 for - five months while fixes accumulated unreleased. +- **Standalone mode**: `scripts/install-core.sh` creates an immutable release + below `~/.devsquad/releases/`, atomically selects it through + `~/.devsquad/current` and keeps `~/.local/bin/squad` stable. It needs no + Claude installation. `--status --json` compares source, plugin and actual + installed payload digests. +- **Claude plugin mode**: `install.sh` installs the standalone runtime first + and, when Claude is available, installs or updates the marketplace plugin. + `plugin/hooks/hooks.json` is the hook registration source. The installer no + longer writes a duplicate set into global Claude settings. +- **Dev mode**: run legacy hooks from this source checkout only when explicitly + testing changes at HEAD. Agents/commands/skills may still come from an + installed plugin, so doctor and install-status output must be checked before + attributing behavior to the source tree. +- Never point a stable launcher or manual hook at a versioned Claude cache + directory (`plugins/cache/...//`). Content-addressed standalone + releases are retained for active runs; the selector, not a running process, + moves during updates. + +See [the runtime guide](RUNTIME-GUIDE.md) for install, operation, recovery and +the honestly blocked live-surface matrix. ## State (per project, `.devsquad/`, self-gitignored) diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md new file mode 100644 index 0000000..f959557 --- /dev/null +++ b/docs/RUNTIME-GUIDE.md @@ -0,0 +1,356 @@ +# DevSquad runtime guide + +This guide covers the surface-independent Python runtime. The older Claude +plugin remains available, but it is not required for the standalone command. + +## Install and update + +Prerequisites are macOS or another Unix-like host and Python 3.11 or newer. +The base runtime has no third-party Python dependencies and installation does +not contact a provider or package index. + +From a DevSquad checkout: + +```bash +./install.sh --core-only +export PATH="$HOME/.local/bin:$PATH" +squad --version +./scripts/install-core.sh --status --json +``` + +`scripts/install-core.sh` creates a Python virtual environment and an exact +copy of `plugin/core` under an immutable, content-addressed directory in +`~/.devsquad/releases/`. `~/.devsquad/current` is changed atomically and +`~/.local/bin/squad` remains stable. A second identical install reports +`"changed":false`. The status report compares the source, plugin and actual +installed payload digests; it does not trust the release manifest alone. + +`./install.sh` also installs or updates the legacy Claude plugin when the +`claude` command exists. Use `--with-claude` to require that path. The plugin's +`hooks/hooks.json` is the only hook registration written by the installer; +the composite installer does not add another copy to global settings. + +The optional local MCP bridge is isolated from the dependency-free base. It +uses the exact `mcp==2.2.0` lock and never downloads implicitly. Prepare a +wheelhouse containing every package in `plugin/core/requirements-mcp.lock`, +then run: + +```bash +./scripts/install-core.sh --with-mcp --mcp-wheelhouse /absolute/path/to/wheels +squad setup --dry-run --json +squad setup --json +squad doctor --json +``` + +Setup registers only one stable `squad` launcher. It refuses duplicate, +inherited, malformed or ambiguous MCP registrations instead of guessing which +one to replace. An upgrade retains every previous release, so a process that +started before the selector changed can finish against its frozen package. + +A schema-changing update defers while the ledger contains active or +recoverable runs. Finish, resume or cancel them with the previous release, +then retry the same installer command. Activation checks the ledger under +its lock and never advances the schema before selecting the new release. +The first new-release ledger operation performs the guarded migration; +already-open old clients cannot write after that migration commits. An +interruption before selector replacement leaves the old schema usable; an +interruption after replacement leaves migration safely retryable. Old +releases and saved receipts remain present. Custom runtime directories get +the same migration guard on first access; use `DEVSQUAD_RUNTIME_DIR` for the +installer's explicitly scoped ledger check. + +A long-running MCP server keeps the package it started with. After a schema +upgrade, reconnect only the DevSquad MCP connection in each already-open host +to load the selected release. An old server may return `SCHEMA_UNSUPPORTED`; +that is the old-client fence, not lost work. Do not remove the ledger or change +unrelated servers. New CLI/MCP processes already use the stable launcher. + +Antigravity's non-interactive print mode also enforces project permissions. +For unattended read-only status checks, add this exact grant to the DevSquad +project's Permissions list in Antigravity: + +```text +mcp(devsquad/squad_status) +``` + +This is narrower than a server wildcard and does not authorize terminal or +file access. Interactive use may instead approve the requested MCP operation +when prompted. Do not use the blanket permission-bypass option for setup or a +smoke test. + +## Operate a run + +The generated [command and schema reference](generated/core-reference.md) +lists every command form, packaged schema digest and a strict task-shape +example. Normal review and fix entry resolves committed refs, discovers the +installed Codex catalog without a generation, freezes bounded subscription +profiles and embeds the validated routing snapshot. It does not require task, +profile or policy JSON: + +```bash +squad doctor +squad review --base main --dry-run +squad review --base main --wait +squad fix "the bounded issue to resolve" \ + --write-path src --wait +squad status +squad finish RUN_ID --accept --reason "Reviewed the saved candidate and evidence." +squad result RUN_ID +``` + +Use `--dry-run` first to inspect exact commit IDs, role/profile selections, +scope and checks without creating a run or invoking a model. `squad fix` +defaults to repository-wide write scope when `--write-path` is omitted; narrow +it whenever the issue permits. `--wait` automatically advances the saved +candidate from implementation into independent review, then returns at the +host-lead handoff with the review summary, check results, evidence report and +finish command. That pause exits 2; it is saved work awaiting your assessment. +Use `--reject` or `--revise` instead of `--accept` when appropriate, with a +reason. Finish binds every artifact from the exact current packet and applies +the same claim, independent-review and required-check gates as the low-level +API. It refuses terminal replays and app-owned host claims; use the saved claim +for a handoff already owned by an app. If interrupted after acquiring its own +claim, finish saves the exact decision atomically with that claim. Status shows +the exact retry command. Only that same packet, disposition and reason can +recover the live claim or its expired fence; an owner name alone never permits +recovery. If the decision was already submitted, use `squad resume RUN_ID` to +finish the saved continuation. A claim that expires just before submission is +still rejected and audited. An exact guided retry under its own fresh valid +fence can recover that expiry: the unique submission row is an operational +projection, while an append-only recovery event preserves the complete original +rejected row and its digest. Final event exports retain both rejection and +recovery history. App claims and other rejection reasons cannot use this path. +The normal lead is the terminal host; +an explicitly configured headless run follows its existing lead through a +temporary handoff when observed with `--wait`. Omitting `--wait` returns the run +ID immediately. + +Normal commands show readable output by default. Add `--json` for the versioned +automation envelope. Omitted IDs on status, result and finish resolve only when +the current canonical Git project has exactly one saved run. With no run, the +command explains how to start; with multiple runs it lists IDs and states and +requires a choice. It never selects another project's run or guesses the latest +run. Use `--project-dir PATH` when observing from outside the project. + +Checks come from regular tracked files at the selected target commit. For +DevSquad this includes the Bash suite, Python core runner and generated +reference check. Delivery requires all three to pass; review-only checks are +reported even when they fail. Explicit `--check` arguments add bounded approved +commands and exact duplicates run once. A directory called `tests` is not +enough to infer a Python suite without actual tracked Python test files. + +```mermaid +flowchart LR + A[review or fix] --> B[Exact candidate review and checks] + B --> C[Saved handoff and evidence] + C --> D[finish with accept, reject or revise] + D --> E[Saved receipt or bounded revision] +``` + +The lower-level automation path remains available. A hand-written task must +name an existing Git repository and committed refs. It may use either committed +routing files or an exact embedded registry/policy pair. + +New public runs automatically record one final learning outcome, including +failed attempts and repairs. `squad report --project "$PWD" --json` includes +these without manual imports. Later feedback remains an explicit late +correction via `squad outcome add RUN --file correction.json --json`; it never +overwrites the original final outcome. Historical missing outcomes stay +missing rather than being manufactured during an upgrade. + +For an explicitly reviewed, predeclared comparison, start each bounded arm: + +```bash +squad trial --experiment experiment.json --case CASE --arm control \ + --task-file control-task.json --idempotency-key trial-CASE-control --wait --json +squad trial --experiment experiment.json --case CASE --arm candidate \ + --task-file candidate-task.json --idempotency-key trial-CASE-candidate --wait --json +``` + +This advanced automation command requires the v2 declaration (case/split, +input and concrete execution hashes, gates, budgets) before either arm. Use +branch reviews for reviewer comparisons or issue delivery for implementer +comparisons. All arms share the declared reservation/wall budget; failed and +fallback slots count. Automatic experimentation and promotion remain off. +The same `policy evaluate`, profile qualification and reviewed binding-change +commands operate on the resulting saved-run evidence; missing arms cannot +create a completed pair or authorize promotion. + +```bash +squad start --task-file task.json --idempotency-key issue-123 --json +squad status RUN_ID --json +squad events RUN_ID --after 0 --limit 100 --json +squad result RUN_ID --json +``` + +Keep the returned run ID. Reusing an idempotency key with an identical request +returns the original run; reusing it with different content conflicts. Status +is the authority for the current state and next action. Result artifacts and +their SHA-256 values are authoritative; a chat summary is not. + +For terminal use, `squad finish` is the supported guided disposition command. +Automation and app hosts can still use the low-level claim/complete protocol. +When a host-lead workflow pauses, JSON status returns `claim_handoff` and the +current run version. Claim and complete the saved packet without editing the claim: + +```bash +squad handoff claim RUN_ID --expected-version VERSION --owner local-operator --json > claim-response.json +python3 -c 'import json,sys; json.dump(json.load(sys.stdin)["data"]["claim"],sys.stdout)' < claim-response.json > claim.json +squad handoff complete RUN_ID --claim-file claim.json --decision-file decision.json --json +``` + +The initial claim command returns the claim; `--claim-file` on that command is +only for renewing an existing claim. Construct `decision.json` from the exact +packet and evidence references returned with the claim, following the strict +[handoff contract](plans/engineering-team/CONTRACTS.md). + +Every surface uses these operations directly or through the thin local MCP +bridge described in [MCP-LOCAL-ACCESS.md](plans/engineering-team/MCP-LOCAL-ACCESS.md). +Closing an app does not cancel the detached run. + +## Manual Council (partial source implementation) + +Council reuses the saved runner with two independent read-only proposers, a +distinct critic and the existing sole lead. Automatic triggering is always off; +there is one round and no internal retry. Prepare a scoped question without task +JSON: + +```bash +squad council "Which retry policy avoids duplicate side effects?" \ + --read-path src/retry.py --lead host --max-invocations 3 --dry-run +``` + +Inspect the exact commits, read scope, rubric, checks and three catalog model +identities. Repeat `--model MODEL_ID` exactly three times to pin `proposer_a`, +`proposer_b` and critic; catalog availability is not tested quality qualification. +`--criterion ID=DESCRIPTION`, `--evidence ARTIFACT_ID:SHA256` and `--check COMMAND` +freeze explicit rubric, saved evidence and bounded checks. Preparation performs +no generation. Remove `--dry-run` only after doctor and per-run capability gates +are satisfied. + +Native Council currently fails closed before any role generation: +`native_ready:false` means a genuine nongenerating backend response under the +exact default-deny macOS sandbox has not been attested. Bootstrap or cached +catalog success is not that proof. Unsupported operating systems have no unsafe +fallback. Normal review/fix remain separate. Do not broaden filesystem/MCP +access, bypass provider permissions or use a paid API to work around this gate. + +Once the gated workflow is available, a host-led run pauses with shuffled A/B +proposals, criterion assessments, mandatory checks and retained dissent. Status +does not choose a proposal or infer a disposition. Finish explicitly: + +```bash +squad status RUN_ID +squad finish RUN_ID --accept --choose synthesis \ + --reason "Retain the idempotency and expiry safeguards." \ + --supported-claim "Never retry a non-idempotent request blindly." \ + --discarded-alternative "Blind retry without a stable key." \ + --validation "Exercise duplicate requests and key-retention expiry." +squad result RUN_ID +``` + +Choose `A`, `B` or `synthesis`; repeat supported/discarded flags as needed. Every +critic objection remains in the decision. Reject with explicit assessment when +appropriate; votes cannot override failed checks. Extra deliberation requires a +new capped run, not `--revise`. `council-finish` is an explicit-choice alias; +omitted IDs follow the same unique-current-project rule as normal finish. +An interrupted guided finish exposes an exact retry command, including every +choice input. Only the latest matching canonical guided claim can recover an +expired fence; an app-owned claim, even named `terminal-operator`, cannot be +adopted. A submitted decision continues with `squad resume RUN_ID`. + +The default headless lead uses four capped worker invocations, chooses only from +validated saved evidence and continues its own handoff with `--wait` or resume; +a host cannot claim it. Missing participants, invalid identities, quota, +cancellation and check failures remain failures, never fabricated consensus. +The retained predeclared matched/held-out public fixture comparison is +inconclusive and proves mechanics only: native quality, escaped defects, rework +and native allowance remain unknown. It authorizes no automatic use or model +promotion. + +## Recover or cancel + +Never delete the runtime database, an active release or a run-owned worktree +to recover a job. Inspect first: + +```bash +squad status RUN_ID +squad events RUN_ID --after 0 --limit 100 --json +``` + +If status supplies a recovery object, save that exact object as +`recovery.json`, reconcile the process identity it describes, and then run: + +```bash +squad resume RUN_ID --recovery-file recovery.json --json +``` + +If status does not request recovery, do not invent a recovery decision. +Cancellation is durable and idempotent: + +```bash +squad cancel RUN_ID +squad status RUN_ID +``` + +For installation drift, run `scripts/install-core.sh --status --json` from the +intended checkout. Reinstalling identical content is safe. A mismatched or +corrupt content-addressed release is rejected rather than repaired in place; +install a new source digest and retain the old directory for run evidence. + +## Supported and deferred boundaries + +The source candidate contract is Python 3.11+, public JSON contract version 1, +SQLite schema 17 and optional MCP SDK 2.2.0 exactly. The accepted installed R5 +boundary remains schema 16 until an explicitly accepted Council update; R6 +source also uses epoch 16, with its own acceptance tracked separately. This +guide does not claim that source Council is installed or live accepted. Epoch 17 +fences old clients that cannot understand Council handoff authority, and its +installer defers on active or recoverable schema-16 work. Native Codex fixtures and +recorded live proofs cover bundled `codex-cli 0.153.4` and +`codex-cli 0.155.0-alpha.9.2`; the latter passed a fresh native initialize and +complete model-catalog probe. The resolver prefers that verified bundled +binary over an older unverified PATH binary. The Claude worker adapter is +version-scoped to CLI 2.1.220. Antigravity 1.2.13 has a live read-only MCP +receipt; later installed CLI/MCP receipts cover Grok 1.0.46. Check the current +doctor report rather than treating an older receipt as proof for a new version. +Version changes are capability drift and require a fresh conformance probe; +brand names are not a compatibility promise. + +Implemented surfaces and evidence: + +| Surface | Current evidence | +|---|---| +| Terminal | Standalone install and real saved-run cancellation; fresh installed normal review/fix/finish flow verified with offline provider binaries | +| Codex App/CLI | Matching MCP registration and a fresh installed-runtime `squad_status` receipt on 0.155.0-alpha.9.2 | +| Claude Code local Code tab | Matching registration and real Claude MCP handoff; local Code-tab UI proof remains open | +| Antigravity CLI (`agy`) | Matching registration and a live Gemini `squad_status` receipt with one project-scoped grant; IDE is outside the clarified request | +| Grok Build | Matching registration and real Grok 1.0.46 MCP status operation | + +The historical installed surface evidence source is +[`M7-installed-runtime-2026-09-29.json`](plans/engineering-team/evidence/M7-installed-runtime-2026-09-29.json). +The installed normal-entry evidence is +[`M7-normal-entry-2026-09-29.json`](plans/engineering-team/evidence/M7-normal-entry-2026-09-29.json). +Later verified runtime proofs, including the accepted two-model delivery, +actual Claude handoff, Grok MCP and Gemini CLI/MCP recheck, are recorded in +[`R8-installed-workflows-2026-10-01.json`](plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json). +These are operation-scoped receipts; the Claude local Code-tab UI proof and +remaining R6/R8 acceptance gates remain open. Antigravity acceptance uses the +CLI, not IDE trust or UI. Doctor separates installed binaries, supported +adapter versions, non-generating authentication checks, registrations and +operation verification. Unknown verification stays unknown. A CLI that is +installed but unsupported or missing subscription authentication does not make +review/fix ready. Complete login through the named provider's normal flow. + +Portable task files, handoff packets, event ledgers and hashed artifacts are +the cross-host interface. A reviewed upstream Codex integration demonstrates +optional native Claude-to-Codex transcript import, but DevSquad does not yet +expose or test that import path. It is deferred and must never be substituted +for the portable handoff contract or described as general chat-history +transfer. + +The one authorized Jev pilot is recorded separately and runtime classification +remains off; normal routing uses the deterministic zero-call path. Laya is not +installed unless its declared trigger fires. C1 Council is implemented partially +in source with offline process/isolation/epoch proofs; native backend attestation, +live quality and final installed acceptance remain open. diff --git a/docs/adr/ADR-002-surface-independent-engineering-team.md b/docs/adr/ADR-002-surface-independent-engineering-team.md new file mode 100644 index 0000000..74b15cc --- /dev/null +++ b/docs/adr/ADR-002-surface-independent-engineering-team.md @@ -0,0 +1,173 @@ +# ADR-002: One engineering team, accessible from every local surface + +- **Date:** 2026-09-06 +- **Status:** Implementation design prepared from Dikshant's brief. The architecture and backlog are ready for a coding agent; the runtime described here is **not implemented**. +- **Scope:** One user, one machine, multiple repositories, multiple local AI applications and subscription CLIs. +- **Basis:** [Repository and GitHub assessment](../audits/2026-09-06-engineering-team-assessment.md), source at `c7f5930`, and the user's subsequent requirements for model/effort selection, capacity, learning, documentation and surface independence. +- **Execution entry:** [START-HERE](../plans/engineering-team/START-HERE.md). + +## Decision in one picture + +```mermaid +flowchart TB + T[Terminal CLI] --> C[squad commands] + A[Codex App] --> M[Thin local MCP bridge] + B[Claude Code App / CLI] --> M + G[Antigravity IDE / CLI] --> M + X[Grok Build] --> M + M --> C + C --> R["Shared runner
Persisted runs and bounded workflows"] + R <--> S["SQLite events and state
Artifacts and handoffs"] + R --> P["Choose role + harness + model + effort
Capabilities first, capacity second"] + P --> W["Native CLI workers
Claude · Codex · Grok · Antigravity"] + W --> V[Independent review and checks] + V --> S + S --> L[Evidence → experiments → versioned policy] + L --> P +``` + +**The product is an engineering team with shared memory and verifiable work.** A provider's distinctive information access is one capability among coding, diagnosis, design, testing, review, tool use and synthesis. Roles stay stable; assignments change with evidence. + +Optimize accepted outcomes first, then reduce rework, latency and scarce allowance. Do not maximize the number of models involved, equalize provider use, or assume that more reasoning effort improves every task. + +## What remains and what changes + +[ADR-001](ADR-001-contract-and-ledger-core.md) remains the historical July decision. This design retains its packaged core inside `plugin/`, reusable adapters, bounded invocation, explicit capabilities, static initial routing and evidence requirements. For this proposed build it replaces these parts: + +| July design | September implementation decision | +|---|---| +| Bash-only coordinator | Python 3.11+ standard-library coordinator; preserve Bash 3.2 adapter compatibility and use verified native protocols where available | +| JSONL as primary ledger | Transactional SQLite event ledger and projections; JSONL is an export | +| Claude session always synthesizes | One explicitly selected lead: current host or a headless worker | +| Provider/role aliases and implicit self fallback | Exact execution profiles; qualified fallback or an explicit blocked state | +| Three-provider demonstration | Useful branch review first; two-harness delivery workflow next | +| D1 determines the platform's future | D1 evaluates hook enforcement only; engineering outcomes evaluate the team | +| Timing/volume threshold as learning prerequisite | Begin manual evidence-based improvements immediately; defer automatic learned routing | + +This does not retroactively mark new decisions as accepted in July. Old product descriptions and historical measurements are not evidence that the new runtime exists. + +## Runtime and packaging + +Add a small Python package inside `plugin/core/`. Its standard library provides JSON validation logic, SQLite, subprocess control and the CLI. Use the official Python MCP SDK as an **optional**, pinned dependency for the MCP bridge; ordinary CLI operations must work without it. Resolve and test the precise SDK version during implementation. + +```text +plugin/core/ + bin/squad stable command entry + pyproject.toml Python floor and optional MCP dependency + src/devsquad/ + cli.py, contracts.py one service API for CLI and MCP + store.py, supervisor.py transactions, ownership, process lifecycle + workflows.py, router.py two fixed workflows, profile selection + adapters.py, capacity.py provider bridge and account pools + learning.py, reports.py observations, evaluations, derived docs + mcp_server.py short tool calls; no separate business logic + schemas/ versioned public JSON contracts + adapters/ manifests and Bash bridge to existing wrappers + policies/ starter policy and reusable role prompts + integrations/ minimal instructions/config templates per host +test/core/ offline unit, process and integration tests +``` + +Use one local database, `~/.devsquad/runtime/state.sqlite3`, on local storage with WAL and schema migrations. Repository identity comes from the canonical Git common directory; worktrees share an identity. Store runs below `~/.devsquad/runtime/projects//runs//`. Runtime data stays outside Git. Versioned project policy lives in a new `devsquad/` directory, avoiding the existing ignored `.devsquad/` legacy state. + +The initial runtime needs no web server, queue service or permanent daemon. `squad start` persists a request, starts one detached supervisor per run and immediately returns its ID. Supervisors own native CLI subprocess groups. Any local surface can inspect, cancel or resume the same run. Closing an app or its MCP connection does not cancel a run. + +A standalone install exposes `~/.local/bin/squad` through a stable launcher to an immutable release directory under `~/.devsquad/releases/`. Plugin and standalone packages use the same core. Running jobs pin their release path and digest. An explicit development install may target the source checkout, with a dirty-source fingerprint. `doctor` reports installation drift. No runtime path should require an active Claude session or a versioned Claude cache path. + +## Surfaces are clients, not separate orchestrators + +| Surface | Integration | Boundary | +|---|---|---| +| Terminal | `squad` directly | Can use a supplied task and a headless lead | +| Codex desktop / CLI / IDE | Local stdio MCP | Local Codex surfaces share MCP configuration on the same host [1] | +| Claude Code desktop, Code tab / CLI | Local stdio MCP | Use shared user/project MCP configuration; local Code sessions have the relevant CLI integration [2] | +| Antigravity IDE / CLI | Local stdio MCP | Verify installed version against documented settings; configure one DevSquad server [3] | +| Grok Build | Local stdio MCP | Detect inherited Claude/project registrations before adding another [4] | + +V1 supports these **local** surfaces on the same machine. Cloud sessions need a later authenticated remote transport; a local stdio server does not make the Mac remotely reachable. Private chat history, hidden reasoning and host-specific tools do not transfer. Task specifications, files, results, decisions and handoff packets do. + +MCP exposes ordinary short `start/status/events/result/cancel/resume/handoff` operations. Do not depend on every host supporting MCP's optional long-running-task extension. Host-specific instructions explain when to use these operations; they must not contain their own router or provider matrix. + +## One lead, bounded workers + +The host supplies a structured task with acceptance criteria. In `lead.mode=host`, the current app is the lead and the runner executes a fixed workflow. When a decision is needed, the run becomes `awaiting_host` with a saved packet. Another surface can claim that handoff using a fenced ownership token. + +In `lead.mode=headless`, a selected CLI profile performs the same synthesis/disposition step. The core does not start an additional planner. V1 accepts structured tasks and two fixed workflow templates; natural-language planning and arbitrary workflow DAGs can come later. + +```mermaid +flowchart LR + I[Task + acceptance criteria] --> E[Implement in isolated worktree] + E --> R["Different-model review
Bound to patch hash"] + R --> Q[Checks on that revision] + Q --> D{Lead disposition} + D -->|bounded repair| E + D -->|criteria met| O[Verified result + patch + receipt] + D -->|unresolved| B[Blocked / failed with evidence] +``` + +Start with the smaller `branch-review` template: snapshot → review → checks → disposition → receipt. Add `issue-delivery` after this is useful. Repository changes occur in an isolated worktree with one writer. Reviewers have verified read-only permissions and a frozen revision. Checks use trusted argv arrays. Patch changes invalidate prior review/test acceptance. V1 returns work for integration; it does not merge, push, deploy or publish automatically. + +Preserve the existing shell wrapper API and four error prefixes. Extract shared argument-building/classification helpers for CLI adapters; prefer a verified native app-server adapter for new Codex jobs. Python owns the protocol child or CLI subprocess, timeouts, cancellation, draining and reaping; the legacy wrapper keeps its separately repaired bounded invocation path. Do not nest competing watchdogs or two job coordinators. The [native adapter amendment](../plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md) records the source-backed extension. + +## Select configurations, measure outcomes + +An execution profile is `(harness, harness version, model family, exact model, effort, tools, permissions, account pool)`. Antigravity may expose several model families: the harness name alone does not identify the model. Every attempt records requested settings, observed settings and verification confidence. Native X, Google Search or YouTube access must be verified for the **selected harness/model/tool combination**; model branding or training history is insufficient. + +Routing starts with a small versioned preference list per role/task class. Filter for capability, permission, quality eligibility and verified model/effort support. Then consider every applicable quota window, cooldown, concurrency, deadline and latency. If no eligible profile is available, block with a reason. Never silently relax required capabilities or switch to paid API usage. + +Selection is automatic by default. A user may pin a validated profile for any role; unpinned roles remain automatic. The selected profile defines the permitted toolbox, and the worker selects actual tool calls within it. Overrides have explicit fallback behavior. Stable role aliases resolve to qualified concrete profiles and are frozen per run. Discovery/evaluation can propose replacements; a reviewed update policy may authorize guarded automatic binding promotions, while policy changes remain reviewed. See the [selection amendment](../plans/engineering-team/SELECTION-AND-COUNCIL.md) and [model lifecycle](../plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md). + +**September 26 amendment:** Evaluate optional typed decision helpers under +[DECISION-CLASSIFIERS.md](../plans/engineering-team/DECISION-CLASSIFIERS.md). +Default-off shadow experiments may test semantic task/profile, skill and +context suggestions. A separately reviewed policy may consume frozen hints +after held-out validation; exact eligibility, pins, permissions, billing, +capacity and lead authority stay unchanged. This narrowly extends the original +deferral of learned routing; it authorizes neither autonomous training/policy +changes nor a required hosted service. The current no-model route remains the +baseline and fallback, and M1–M5 acceptance gates are unchanged. + +Account pools span applications and repositories where the underlying allowance is shared. Provider observations have sources and expiry times; unknown allowance is unknown. DevSquad's concurrency reservations do not reserve quota with a provider. Spend estimates, token counts, characters and subscription allowance are distinct measurements. + +## Learning and documentation are part of completion + +```mermaid +flowchart LR + A["Every attempt
Settings, outputs, failures"] --> B["Task verdict
Checks, review, lead repairs"] + B --> C["Comparable evidence
Later user corrections included"] + C --> D["One hypothesis
Small budgeted experiment"] + D --> E{Quality and capacity evidence} + E -->|supported| F[Versioned policy change] + E -->|inconclusive| G[Keep current policy] + F --> H[Monitor drift / regressions] + H --> C +``` + +Use two loops: bounded corrections inside the current task, and deliberate policy improvement across tasks. Initial profile preferences are hypotheses, not a provider leaderboard. Retain failed attempts, fallbacks and lead rework; a repaired final success must not become an unqualified success for the original worker. Hold out evaluation tasks and avoid tuning and testing on the same answer. + +Runtime records generate a receipt at every terminal state and a handoff packet when waiting. Curated `devsquad/learning/` records track `observed → hypothesis → tested → adopted/rejected → revalidate`. Policy changes link evaluations and rollback versions. Update these artifacts as work completes; a later scheduled summary can be added when requested, but no background automation is installed by this plan. + +## Delivery gates + +| Gate | User-visible result | +|---|---| +| M1–M2 | Reliable adapter bridge and recoverable local run; fake process tests prove the lifecycle | +| M3 | A real branch review from terminal produces an inspectable receipt | +| M4 | Start in one local app, inspect/finish in another, using the same run ID | +| M5 | Bounded issue → implementation → different-model review → checks | +| M6 | Account-pool-aware scheduling and evidence-backed profile comparisons | +| M7 | Fresh installation, documentation and real smoke receipts for all requested surfaces | + +The detailed [contracts](../plans/engineering-team/CONTRACTS.md) and [work packages](../plans/engineering-team/IMPLEMENTATION.md) are normative for the build. Their gates replace claims based only on dry runs or syntax checks. + +**Optional C1 after M6:** Adapt LLM Council's independent-proposal, critique and synthesis pattern for difficult decisions. Use a bounded council within this runner, with evidence-based judgement and saved dissent. It does not delay M7 or replace routine implementation/review/checks. The [source study and extension gate](../plans/engineering-team/SELECTION-AND-COUNCIL.md) explain the protocol and capacity tradeoff; automatic council triggering requires evaluation evidence. + +## Source notes + +Product integration documentation checked on 2026-09-06; recheck at installation because paths and host versions change. + +1. [Codex MCP configuration](https://learn.chatgpt.com/docs/extend/mcp?surface=cli). +2. [Claude Code desktop](https://code.claude.com/docs/en/desktop). +3. [Antigravity MCP](https://antigravity.google/docs/mcp) and [CLI MCP](https://antigravity.google/docs/cli/mcp/). +4. [Grok Build MCP servers](https://docs.x.ai/build/features/mcp-servers). +5. [Official MCP Python SDK](https://github.com/modelcontextprotocol/python-sdk) and [optional MCP Tasks](https://modelcontextprotocol.io/extensions/tasks/overview). diff --git a/docs/audits/2026-09-06-branch-consolidation.md b/docs/audits/2026-09-06-branch-consolidation.md new file mode 100644 index 0000000..be78538 --- /dev/null +++ b/docs/audits/2026-09-06-branch-consolidation.md @@ -0,0 +1,62 @@ +# Branch consolidation — September 6, 2026 + +The repository now has two working branches, with matching names locally and on GitHub. Runtime implementation is still pending; branch cleanup does not complete any implementation milestone. + +```mermaid +flowchart LR + M[main: published runtime] --> E[codex/engineering-team: architecture + Sol build] + E --> V[Verified milestone commits] + V --> R[Review completed implementation] + R --> P[Merge to main when authorized] + A[February legacy history] --> T[Retained tags + local recovery bundle] +``` + +## Canonical branches + +| Branch | Purpose | Starting evidence | +|---|---|---| +| `main` | Published runtime baseline; tracks `origin/main` | `fa68f68621dcad166ab24cf4b3db379ecbe819ac` | +| `codex/engineering-team` | Complete architecture, execution prompt and subsequent Sol implementation | Contains all five design commits through `ff1fa602d546c9b19c50ae32fc00d7066a7aaf10`, followed by this cleanup record | + +Use [SOL-HANDOFF.md](../plans/engineering-team/SOL-HANDOFF.md) as the full build assignment. Start on the build branch or a worktree based on it. Before implementation, require a clean tree and `git merge-base --is-ancestor ff1fa60 HEAD` to succeed. GitHub's default branch remains `main`, so a default-branch checkout alone does not contain the plan. + +## What was sorted + +| Previous branch | Finding | Disposition | +|---|---|---| +| `codex/surface-independent-team-plan` | Latest complete plan and handoff, five commits ahead of `main` | Renamed to `codex/engineering-team`, published with upstream tracking | +| `codex/engineering-team-assessment` | `c7f5930`, already contained in the plan branch | Redundant local branch removed; commit retained in build history | +| `holdout-protocol` | `044dd75`, fully merged; `main` is 20 commits ahead and zero behind | Redundant local and GitHub branches removed; commit retained in `main` | +| `backup-bug-fixes` | `f6d9c4c`, 85 commits in unrelated February history, no merge base with current `main` | Branch removed after verified complete backup; existing local tag `v0.1.1-bug-fixes` retains the exact tip | + +The old bug-fix lineage was reviewed by Sol before implementation. Its meaningful fixes were ported by `b70fc22`, an ancestor of current `main`, and subsequently evolved in the current plugin. Do not cherry-pick the old production snapshot: its duplicated packaging and older wrappers would regress the present structure. Historical audit documents remain accessible through the tag. + +Existing tags `v0.1.1-bug-fixes`, `v1.0` and `v1.1` were retained unchanged. Only `v1.0` was already on GitHub; the two local historical tags and recovery bundle remain local. Cleanup does not create a release or publish the separate legacy history. + +## Recovery + +A complete, verified Git bundle was created before any branch deletion: + +```text +~/.codex/backups/devsquad/2026-09-06-before-branch-cleanup-ff1fa60.bundle +``` + +The bundle preserves all original branch refs, tags and reachable history. It is outside the repository and is not a cloud backup. Restore an old branch only when actually needed: + +```bash +git branch backup-bug-fixes v0.1.1-bug-fixes +git branch holdout-protocol 044dd75 +git branch codex/engineering-team-assessment c7f5930 +``` + +If local tags are unavailable, recover the legacy branch from the bundle: + +```bash +git fetch "$HOME/.codex/backups/devsquad/2026-09-06-before-branch-cleanup-ff1fa60.bundle" refs/heads/backup-bug-fixes:refs/heads/backup-bug-fixes +``` + +## Ongoing branch discipline + +Keep milestone checkpoints on `codex/engineering-team`. Use separate worktrees only for concurrent work, then integrate their verified commits and remove their temporary branch refs/worktrees. Preserve unique work before removal. Keep `main` as the reviewed baseline and merge the completed build only when authorized. Fetch/prune before branch cleanup; check ancestry rather than interpreting a branch name or date as proof that it is obsolete. + +The cleanup audit found one working tree, no stashes, no open or historical GitHub pull requests, and no remote changes after fetching. Validation includes ancestry checks, complete bundle verification, the existing offline test suite, clean working-tree checks and exact local/GitHub branch-tip comparison. These checks establish repository hygiene; they do not establish that the planned engineering-team runtime works. diff --git a/docs/audits/2026-09-06-engineering-team-assessment.md b/docs/audits/2026-09-06-engineering-team-assessment.md new file mode 100644 index 0000000..b769781 --- /dev/null +++ b/docs/audits/2026-09-06-engineering-team-assessment.md @@ -0,0 +1,183 @@ +# DevSquad: engineering team assessment and path to usability + +Date: 2026-09-06 + +Baseline: main at fa68f68621dcad166ab24cf4b3db379ecbe819ac, plugin 0.10.0 + +Status: Assessment and proposed delivery plan. This document does not supersede ADR-001 or claim the proposed features are implemented. + +## Verdict + +DevSquad is worth continuing as a personal AI engineering team coordinator. It has useful provider integration code and regression tests, but it is not yet a dependable team that owns engineering outcomes or makes effective use of subscription capacity. + +The intended product has two connected objectives: better engineering through complementary capabilities and independent judgment; and more accepted work from the subscriptions already available. Information access is one specialization alongside architecture, implementation, debugging, design, testing, and review. + +The current product grew around preserving Claude's context. Its hooks, agent prompts, measurement, and capacity messages still express that earlier objective. Those choices explain much of the present mismatch. The next iteration should make a complete, verified engineering task the unit of work and measurement. + +Keep the adapter experience, error taxonomy, compatibility tests, and the accepted shared-core direction. Deliver a small working team before extending platform scope. Five delivery gates below define that path; each requires observable behavior rather than completion checkboxes. + +## Scope and evidence + +Reviewed the local source, complete Git history, relevant local planning and verification records, GitHub branches/releases/workflows, installed plugin metadata and files, scoped DevSquad configuration/telemetry, and current official provider documentation. Earlier in this review, all 177 assertions in nine offline test files passed under Bash 3.2. Those tests establish selected implementation contracts, not live provider compatibility or output superiority. + +GitHub main matches the baseline. The holdout-protocol branch is fully merged. Installed 0.10.0 plugin source matches the current plugin source. GitHub currently has no Actions workflows or published releases; the legacy v1.0 tag references a manifest numbered 0.1.0. Runtime code has not changed since July 7; later commits concern presentation, documentation, and motion assets. + +The local probes inspected CLI help, versions, model listings, and model resolution. They did not execute paid model tasks or benchmark engineering quality. Grok reported expired authentication. The cached model list dates to July 6 and differs from the current Antigravity model list. Usage evidence is scoped to the directories inspected; its age does not prove that no AI work occurred elsewhere. + +Historical planning records under .planning are local and gitignored. They were read as historical evidence, not treated as current instructions or proof that their claims remain true. Findings from those records are summarized here so the plan does not depend on readers possessing them. + +## How DevSquad arrived here + +| Period | What shipped or was recorded | What it teaches | +| --- | --- | --- | +| February 13 | Claude-oriented delegation plugin, marketplace restructuring, hook fixes, capacity reporting, acceptance tracking, and estimated token savings | Installation and execution wiring were part of the product problem from the beginning. Counting suggestions is different from completing delegated work. | +| February 18–19 | Git-health helpers, Gemini-to-Codex skill generation, and a real sequential shell workflow runner with gates/checkpoints | These are substantive utilities, but the unit of delivery became scripts and templates rather than completed engineering outcomes. | +| February 19 verification | A completion summary described the workflow as working end-to-end after syntax, JSON, grep, and dry-run checks. A separate verification record still called for real execution; a milestone audit identified cross-component runtime breaks | Structural verification was promoted into a stronger claim than the evidence supported. Future milestones need a recorded real run and its accepted artifact. | +| February 20–25 | Version renumbering, packaging corrections, and repeated validator/audit fixes through 0.3.0 | A source checkout working correctly does not establish that an installed plugin works in another project. | +| April 16 | 0.4.0 repaired new-project hook registration, paths, and noninteractive wrapper behavior | Fresh installation and ordinary project use must be acceptance tests, not post-release discoveries. | +| July 5–7 | Holdout experiment, Antigravity transition, Grok integration, common adapter, model catalog, expanded tests, and accepted ADR-001 | This is the strongest reusable engineering foundation. It also documents deployment drift, duplicated hooks, sparse observed delegation, and the need for a shared core. | +| July 7 onward | 0.10.0 maintenance release followed by README, growth, and motion work; August's main change is README-only | The accepted core/manifest/dispatch plan did not reach implementation. The next milestone should be smaller and centered on use. | + +Supporting commits include [initial implementation](https://github.com/joshidikshant/devsquad/commit/ac797bb), [February milestone](https://github.com/joshidikshant/devsquad/commit/a480e5a), [April fixes](https://github.com/joshidikshant/devsquad/commit/b98558e), [shared adapter](https://github.com/joshidikshant/devsquad/commit/5e6b08b), [ADR acceptance](https://github.com/joshidikshant/devsquad/commit/3a96bef), and [0.10.0](https://github.com/joshidikshant/devsquad/commit/5afc2f0). + +The July 5 routing record reports four completed delegation usage records plus test artifacts across its examined history. ADR-001 uses a different scope and describes approximately one meaningful completion and zero accepted suggestions out of 80. These are dated historical observations, not a new September adoption count. They support demanding better completion evidence; they do not establish that users or models reject the team concept. See [routing history](../../ROUTING-CHANGELOG.md) and [ADR-001](../adr/ADR-001-contract-and-ledger-core.md). + +My interpretation: DevSquad has accumulated substantial integration knowledge, but the feedback loop from ordinary use to accepted outcomes is weaker than the planning and review loop. More architectural discussion alone will not close that gap. + +## Readiness against the engineering team objective + +| Requirement | Current state | Practical consequence | +| --- | --- | --- | +| Reliable provider execution | Three thin wrappers and a common adapter; incident tests exist, but process cleanup and semantic success need work | Useful foundation, not yet a reliable unattended execution boundary | +| Interchangeable team leadership | Claude-specific hooks and eight Claude Sonnet relay agents; no neutral job command or Claude worker adapter | Codex/Astra cannot use the same team interface symmetrically | +| Roles matched to capabilities | Coarse keyword routes, hardcoded hook choices, mixed model families behind Gemini role names | Actual tools and model diversity do not reliably determine assignment | +| Complete implementation ownership | Agents are framed around short drafts; workflow runner executes shell strings | A coding agent is underused when a task needs sustained implementation and validation | +| Shared work and handoffs | Project state and text responses; no durable task/result contract or dependency-aware handoff | Recovery and integration depend on the lead reconstructing context | +| Independent review and QA | No explicit reviewer lane or acceptance gate in the core | An exit code or plausible explanation can be mistaken for a successful outcome | +| Capacity allocation | Manual usage cache and advisory messages; fixed cooldowns; Grok omitted from capacity schema | Capacity reporting does not yet allocate jobs or maximize usable subscription work | +| Safe parallel engineering | Some session-scoped counters, but shared mutable state and whole-tree checkpoint behavior remain | Parallel writers can collide or lose accounting; isolation must precede concurrency | +| Quality and value measurement | Character counts, suggestion outcomes, and Claude-focused holdout reconciliation | There is no evidence yet that the proposed team improves accepted engineering output | + +## Immediate blockers + +1. **Model identity is unreliable.** With the inspected local catalog/configuration, Gemini researcher/developer frontier pins resolve to Claude Opus 4.6 through Antigravity. The algorithm ranks cross-family numeric version strings. Preserve model family intent and distinguish harness, model provider, requested model, observed model, available tools, and account pool. See [model-catalog.sh](../../plugin/lib/model-catalog.sh), especially resolve_model_tier. +2. **Hook routing bypasses configured routing.** Reading and WebSearch are assigned to Gemini and tests to Codex directly. Unify this policy with workflow/manual dispatch. The lead can specify a role and required capabilities explicitly, avoiding brittle natural-language classification without adding another routing model. See [pre-tool-use.sh](../../plugin/hooks/scripts/pre-tool-use.sh) and [routing.sh](../../plugin/lib/routing.sh). +3. **Directory context is incomplete.** The file whitelist omits TSX, JSX, and other common source types; it would omit all nine TSX files in this project's motion source directory. String-split file arguments also mishandle spaces. Use an explicit file manifest, honor ignore rules, and report omissions and size limits. See [gemini-wrapper.sh](../../plugin/lib/gemini-wrapper.sh). +4. **The portable watchdog can add its full timeout to successful captured calls.** This Mac has no timeout/gtimeout binary. An immediate fake CLI response took 2.025 seconds with a two-second bound because the watchdog's sleep retained the capture pipe. Fix process/descriptor cleanup and verify elapsed-time behavior, timeout termination, and cancellation. See [adapter.sh](../../plugin/lib/adapter.sh). +5. **Process success is not task success.** Output may be empty, a tool may be denied, tests may never execute, or the requested artifact may not exist despite exit zero. Normalize execution status separately from acceptance, preserve provider events where supported, and require verifiable artifacts/checks. +6. **State and Git checkpoints need ownership.** JSON array rewrites can lose concurrent updates. Project-wide pending suggestions and workflow state mix sessions. Checkpoints stage all files and suppress commit failure. Use isolated run state, serialized event writes, scoped worktrees, and truthful checkpoint outcomes. See [usage.sh](../../plugin/lib/usage.sh), [enforcement.sh](../../plugin/lib/enforcement.sh), and [lib-workflow.sh](../../plugin/skills/workflow-orchestration/scripts/lib-workflow.sh). +7. **Installed-version behavior needs a health check.** Grok needs reauthentication. Local Antigravity/Grok versions must be checked against the features used, rather than inheriting assumptions from current documentation. Local duplicate DevSquad hook registration is presently absent; the installer/onboarding registration paths still need an idempotence test so this historical defect is not reintroduced. + +These blockers justify targeted fixes, not a wholesale rewrite. The exact affected behavior should have a regression test before being labeled repaired. + +## What “usable” must mean + +The first release should let the user state a bounded engineering objective from Codex or Claude, then receive a tested patch and an independent review without manually copying prompts between products. A provider becoming limited or unavailable must preserve the work, select a suitable alternative when available, and explain the resulting state. A restart must resume or truthfully report the interrupted task. + +Minimum observable experience: + +1. Inspect the available team once: runtime versions, authentication state, verified capabilities, and capacity freshness. +2. Give the lead an issue, a repository, and acceptance criteria in ordinary language. +3. See a compact plan with accountable roles; only independent work runs concurrently. +4. Receive progress when a result, failure, handoff, or decision matters. +5. Receive the patch, reviewer findings/dispositions, checks executed, unresolved limitations, and a compact capacity receipt. +6. Continue that task from the same saved artifacts if execution stops or leadership changes. + +The user should not have to choose every model, prepare workflow JSON for each task, re-explain the repository after every handoff, or repeatedly enter usage percentages. Natural-language interaction belongs to the active lead; the core should execute explicit structured jobs beneath it. + +The first workflow should be one real issue to implementation, independent review, correction, and test verification. Use two model families initially; add Grok or Gemini specializations when the issue benefits from them. A mandatory four-provider chain is not a requirement. A passing dry-run, generated skill scaffold, or impressive research report is not this milestone. + +## Minimum architecture + +Keep the runtime in the installed plugin package. Expose the existing wrappers through a small executable interface; thin Claude and Codex integrations call the same core. Existing native harnesses continue to own their agent loops, credentials, tool execution, and provider sessions. DevSquad owns job assignment, state, policy, artifacts, acceptance, and capacity accounting. + +Proposed interfaces, not current commands: squad doctor; squad invoke; squad run; squad status; squad resume; squad report. A thin Codex skill can use the CLI first. MCP is an optional later transport when actual host integration needs it, not a prerequisite. + +Four small contracts are sufficient to start: + +| Contract | Essential fields | +| --- | --- | +| Task | task/run ID, objective, repository and base revision, role, required capabilities, input artifacts, acceptance criteria, allowed operations, dependencies, deadline/budget constraints | +| Adapter capability | harness and installed version, model provider/family, requested/observed model, tool capability, verification status/date, auth route, account pool, output format, permission profile | +| Result/handoff | task/attempt ID, execution status, provider session ID if available, model/tool evidence, worktree and patch/artifact references, checks and results, reviewer findings, remaining work, source evidence when relevant, duration and measured usage | +| Capacity observation | account pool, applicable limit windows, remaining/used value when available, reset time when known, observed time, source, freshness, confidence, unavailable reason | + +Separate execution states from acceptance states. An attempt can exit successfully while its task remains unverified or needs revision. Distinguish failed, interrupted, waiting for quota, blocked capability, and awaiting a user decision. Do not turn every failure into a generic text response. + +Use one worktree and one writer at a time per implementation task, with a read-only reviewer over a recorded revision. Additional independent tasks can get their own worktrees. A handoff includes the base revision, current diff, relevant decisions, completed checks, outstanding failures, and next action; it does not pretend that hidden model context transfers between providers. + +Store per-run artifacts and events. A JSONL file is not automatically a concurrency solution: choose a serialized writer or explicit locking, stable event IDs, and crash-safe updates. Derived dashboards or reports can be rebuilt from those records. + +## Subscription capacity is a scheduling input + +Current capacity reporting asks users for quartile ranges and stores their midpoints. The schema omits Grok; recommendation logic does not use Codex's weekly value to choose its displayed zone and does not connect to job routing. Missing values can appear as zero usage. This is a reporting aid, not an allocator. See [capacity command](../../plugin/commands/capacity.md) and [usage.sh](../../plugin/lib/usage.sh). + +The next scheduler should: + +- Establish capability and minimum quality eligibility first, then consider fresh availability, relevant limit windows, resets, latency, and the role preference order. +- Group executors by their actual account/limit pool. Codex app and CLI are not two budgets merely because they are two interfaces. Model identity also does not establish that two runtimes share a pool. +- Represent unknown capacity as unknown. Distinguish subscription limits, model context occupancy, per-call tokens, temporary cooldowns, and paid API spending. +- Prefer supported automatic observations. Accept a timestamped manual observation when necessary, with honest precision; do not require user input before every dispatch. +- Track account-level capacity across projects, while keeping task state per run. Coordinate concurrent launches and use conservative concurrency caps when remaining capacity cannot be measured precisely. +- Preserve capacity for the lead and final review when the user needs them. Fill independent work with other capable providers where useful; do not spend quotas simply to achieve utilization. +- On rate limit, checkpoint before a compatible handoff; on authentication failure, mark the adapter unavailable until repaired. If a required unique capability has no alternative, report that limitation rather than silently substituting an unsupported answer. +- Keep subscription execution and separately billed API paths explicit. Adding a video API connector should not silently change the payment route of ordinary work. + +The objective is accepted engineering work per available subscription window, under a quality floor. Neither raw token savings nor equal use of every provider captures that objective. + +## Five delivery gates + +These are acceptance gates, not five large releases or calendar promises. Gates 1–3 can ship together as one narrow vertical slice using two already-working harnesses: repair the execution/context defects that affect that task, add only its required invocation/results boundary, and finish the task. A read-only branch-review command can provide utility even earlier. Full registry migration, all-provider support, and a new Claude worker adapter must not block that first loop; add the Claude worker when the chosen team needs it. Complete the wider provider matrix, installation hardening, and CI before the daily-use release. Gate 3 provides an early usable engineering loop; Gate 4 delivers the combined team-plus-capacity proposition. Gate 5 determines whether it is ready to become the default daily workflow. + +| Gate | Deliverable | Required acceptance evidence | +| --- | --- | --- | +| 1. Trustworthy execution | Repair execution/context/model issues exercised by the first two harnesses; report their health and use explicit role permissions | Existing suite passes; targeted regressions cover the selected path's elapsed time, cancellation, source manifest, model identity, and denied/empty output; supported-version smoke runs are recorded | +| 2. Shared jobs and results | Package a neutral executable around the needed adapters, with minimal manifests, structured task/results, and durable per-run artifacts | The same bounded job is callable from Claude and Codex; actual/unknown model identity is reported honestly; failure retains artifacts; the packaged interface works outside the DevSquad checkout | +| 3. Complete engineering loop | One issue to plan, isolated implementation, different-model review, bounded revision, and executed checks | A real patch satisfies its predefined criteria; review findings are resolved or explicitly dispositioned; failed prerequisites block dependents; no manual prompt ferrying is required | +| 4. Capacity and recovery | Pool-aware eligibility, fresh/unknown capacity, bounded fallback, checkpointed handoffs, resume, and controlled concurrency | Simulated rate limit and authentication failure are distinguished; a handoff preserves existing edits and checks; a shared pool is not double-counted; required capabilities survive fallback; interrupted tasks resume without duplicate application | +| 5. Daily-use validation and release | Representative task trials, doctor/CI coverage for all advertised adapters, coherent install/version/release procedure, and user-facing evidence | Clean-install smoke verifies one hook registration; advertised provider paths are checked; pilot records quality, completion, rework, intervention, time, and capacity evidence; runtime/package versions match; limitations are published with a tagged release | + +Use isolated fake providers for failure-injection tests. Live smoke calls establish what the real installed providers can execute; they should not deliberately exhaust quotas or invalidate working credentials. Repeated real use is necessary to establish reliability beyond either kind of test. + +The shortest useful proof is a modest bug fix or feature in an existing repository. Avoid using DevSquad's own skill generator as the only proof: that narrows evaluation to scaffolding, mixes generated content with plugin internals, and fails to exercise ordinary engineering ownership. + +## ADR-001: preserve the foundation, revise the product assumptions + +Preserve its Bash-compatible runtime, executable JSON boundary, in-plugin packaging, thin adapters, deterministic policy, ledger, and evidence before learned routing. Existing callers should remain compatible while the new path is tested. + +Propose a short follow-up ADR covering these changes rather than editing the accepted record to imply past agreement: + +1. Both Claude and Codex may lead; either may serve as an external worker/reviewer. +2. The first product workflow is engineering delivery and independent verification. GrowthSquad and a compulsory three-vendor demo move later. +3. Task acceptance and subscription utility become first-class outcomes. D1 remains an experiment about Claude delegation behavior, not the survival criterion for the whole team product. +4. Fallback preserves capability requirements. Unconditional terminal self-answering is not acceptable when the task requires evidence or operations the lead cannot perform. +5. Role-specific permissions replace blanket approval-bypass defaults; scoped acceptance policies should respect the user's already-authorized work. +6. Model choice uses identity, actual capabilities, and evidence rather than cross-family version sorting or a fixed historical latency floor. + +Do not pre-approve a language rewrite, licensing change, separate repository, marketplace, hosted service, universal provider registry, learned model router, or parallel multi-writer system. Add each only when observed use creates a specific need. Role templates can start small; a generator is not required for the first team loop. + +## How to test the combination thesis + +Start with roughly 10–15 representative tasks across bug fixes, small features, refactors, test improvements, and an occasional task requiring external evidence. This is a usability pilot, not a statistically conclusive benchmark. + +Compare DevSquad with the strongest single-agent workflow the user already uses. Keep task scope, starting revision, acceptance criteria, and available evidence comparable. Review resulting patches/reports without provider labels where practical. Judge criteria established before seeing the outputs; account for repeated-task learning when interpreting results. + +Record accepted completion, defects/omissions, human correction and intervention, time to a verified result, work preserved after interruption, and available quota observations. Capture provider usage when exposed, otherwise mark it unknown. Do not infer consumed subscription percentage directly from character counts. + +Test three separate sources of benefit: complementary tool access, independent design/review judgment, and continuity when one provider is constrained. A failed token-saving experiment does not refute all three; conversely, using more models does not prove any of them. + +Release decisions should name task classes where the team helps, where the direct single-agent path is preferable, and where the evidence is insufficient. Keep that fast direct path as part of the product. If review repeatedly adds no useful findings, reduce that lane for the relevant task class. If a provider contributes unique evidence or preserves progress under limits, record that value explicitly. + +## Provider facts relevant to the implementation + +Current official documentation supports native headless execution for all four families of harness. Exact flags, event dialects, permissions, and telemetry vary by installed version; the adapter must verify that contract rather than assume interchangeability. [Claude headless](https://code.claude.com/docs/en/headless), [Codex non-interactive mode](https://learn.chatgpt.com/docs/non-interactive-mode), [Grok Build headless](https://docs.x.ai/build/cli/headless-scripting), [Antigravity headless](https://antigravity.google/docs/cli/headless/). + +Grok Build documents X search. Gemini video understanding through its API supports YouTube URLs, but that does not establish the same path in the installed Antigravity CLI. Record separate web, X, transcript, and video capabilities. [Grok changelog](https://x.ai/build/changelog), [Gemini video input](https://ai.google.dev/gemini-api/docs/video-understanding). + +The repository's blanket statement that Gemini CLI was decommissioned is too broad: Google's June transition affected individual free/AI Pro/Ultra access, while enterprise and paid API access remain supported. Correct the account-specific wording when updating setup documentation. [Google transition announcement](https://developers.googleblog.com/en/an-important-update-transitioning-gemini-cli-to-antigravity-cli/). + +## Recommended immediate next implementation + +Begin with Gate 1 against an isolated fixture repository, then the shared job/result interface and one issue-to-reviewed-patch workflow. Seed the new tracked delivery record with the five gates above and their evidence links. Keep February's local planning records as history, not the active completion dashboard. Make installed-package verification and one real accepted run part of every milestone that claims end-to-end usability. + +The next proof should be a user task delivered by the team with less coordination burden, preserved work during a provider handoff, and a result that passes independent checks. That is the path from the existing delegation plugin to a usable engineering team. diff --git a/docs/generated/core-reference.md b/docs/generated/core-reference.md new file mode 100644 index 0000000..a53d714 --- /dev/null +++ b/docs/generated/core-reference.md @@ -0,0 +1,148 @@ + +# DevSquad core command and schema reference + +Regenerate with `python3 scripts/generate-core-reference.py`; verify with +`python3 scripts/generate-core-reference.py --check`. + +## Command forms + +- `squad cancel [-h] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad capacity [-h] {observe} ...` +- `squad capacity observe [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad classify [-h] [--cwd CWD] [--model MODEL] [--effort EFFORT] [--permission {read_only,workspace_write}] [--timeout TIMEOUT] [--transport {cli_exec,native_protocol}] [--catalog-file CATALOG_FILE] --returncode RETURNCODE --stdout-file STDOUT_FILE --stderr-file STDERR_FILE {codex,antigravity,grok}` +- `squad council [-h] [--project-dir PROJECT_DIR] [--base BASE] [--target TARGET] [--read-path READ_PATH] [--evidence EVIDENCE] [--criterion CRITERION] [--check CHECK] [--model MODEL] [--effort EFFORT] [--lead {headless,host}] [--max-invocations MAX_INVOCATIONS] [--idempotency-key IDEMPOTENCY_KEY] [--dry-run] [--wait] [--json] [--runtime-dir RUNTIME_DIR] question` +- `squad council-finish [-h] (--accept | --reject) --reason REASON --choose {A,B,synthesis} [--supported-claim SUPPORTED_CLAIM] [--discarded-alternative DISCARDED_ALTERNATIVE] --validation VALIDATION [--json] [--project-dir PROJECT_DIR] [--runtime-dir RUNTIME_DIR] [run]` +- `squad doctor [-h] [--json] [--project-dir PROJECT_DIR] [--squad-executable SQUAD_EXECUTABLE]` +- `squad events [-h] [--after AFTER] [--limit LIMIT] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad finish [-h] (--accept | --reject | --revise) --reason REASON [--choose {A,B,synthesis}] [--supported-claim SUPPORTED_CLAIM] [--discarded-alternative DISCARDED_ALTERNATIVE] [--validation VALIDATION] [--project-dir PROJECT_DIR] [--json] [--runtime-dir RUNTIME_DIR] [run]` +- `squad fix [-h] [--base BASE] [--target TARGET] [--project-dir PROJECT_DIR] [--write-path WRITE_PATH] [--check CHECK] [--check-timeout CHECK_TIMEOUT] [--review-model REVIEW_MODEL] [--review-effort REVIEW_EFFORT] [--review-mode {standard,adversarial}] [--review-focus REVIEW_FOCUS] [--implementer-model IMPLEMENTER_MODEL] [--implementer-effort IMPLEMENTER_EFFORT] [--idempotency-key IDEMPOTENCY_KEY] [--dry-run] [--wait] [--json] [--runtime-dir RUNTIME_DIR] issue` +- `squad handoff [-h] {claim,complete} ...` +- `squad handoff claim [-h] --expected-version EXPECTED_VERSION --owner OWNER [--claim-file CLAIM_FILE] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad handoff complete [-h] --claim-file CLAIM_FILE --decision-file DECISION_FILE [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad learn [-h] {propose} ...` +- `squad learn propose [-h] --project PROJECT [--json] [--runtime-dir RUNTIME_DIR]` +- `squad mcp [-h] {serve} ...` +- `squad mcp serve [-h] [--runtime-dir RUNTIME_DIR] [--surface SURFACE] [--session-ref SESSION_REF]` +- `squad outcome [-h] {add} ...` +- `squad outcome add [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad policy [-h] {evaluate} ...` +- `squad policy evaluate [-h] --experiment EXPERIMENT [--revision-id REVISION_ID] [--previous-evaluation-sha256 PREVIOUS_EVALUATION_SHA256] [--json] [--runtime-dir RUNTIME_DIR]` +- `squad prepare [-h] [--cwd CWD] [--model MODEL] [--effort EFFORT] [--permission {read_only,workspace_write}] [--timeout TIMEOUT] [--transport {cli_exec,native_protocol}] [--catalog-file CATALOG_FILE] --prompt PROMPT {codex,antigravity,grok}` +- `squad profile [-h] {template-add,binding-bootstrap,qualification-add,binding-change,binding-fallback,binding-show} ...` +- `squad profile binding-bootstrap [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad profile binding-change [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad profile binding-fallback [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad profile binding-show [-h] [--json] [--runtime-dir RUNTIME_DIR] alias` +- `squad profile qualification-add [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad profile template-add [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad report [-h] --project PROJECT [--json] [--runtime-dir RUNTIME_DIR]` +- `squad result [-h] [--json] [--runtime-dir RUNTIME_DIR] [--project-dir PROJECT_DIR] [run]` +- `squad resume [-h] [--json] [--runtime-dir RUNTIME_DIR] [--recovery-file RECOVERY_FILE] run` +- `squad review [-h] [--base BASE] [--target TARGET] [--project-dir PROJECT_DIR] [--model MODEL] [--effort EFFORT] [--mode {standard,adversarial}] [--focus FOCUS] [--check CHECK] [--check-timeout CHECK_TIMEOUT] [--idempotency-key IDEMPOTENCY_KEY] [--dry-run] [--wait] [--json] [--runtime-dir RUNTIME_DIR]` +- `squad setup [-h] [--host {codex,claude-code,antigravity,grok}] [--dry-run] [--json] [--project-dir PROJECT_DIR] [--squad-executable SQUAD_EXECUTABLE]` +- `squad start [-h] --task-file TASK_FILE --idempotency-key IDEMPOTENCY_KEY [--supersedes-run SUPERSEDES_RUN] [--wait] [--json] [--runtime-dir RUNTIME_DIR]` +- `squad status [-h] [--json] [--runtime-dir RUNTIME_DIR] [--project-dir PROJECT_DIR] [run]` +- `squad trial [-h] --experiment EXPERIMENT --case CASE --arm {control,candidate} --task-file TASK_FILE --idempotency-key IDEMPOTENCY_KEY [--wait] [--json] [--runtime-dir RUNTIME_DIR]` + +## Common command examples + +```bash +squad --version +squad doctor --project-dir "$PWD" --json +squad start --task-file task.json --idempotency-key issue-123 --json +squad status RUN_ID --json +squad events RUN_ID --after 0 --limit 100 --json +squad result RUN_ID --json +squad cancel RUN_ID --json +squad resume RUN_ID --recovery-file recovery.json --json +squad setup --host codex --dry-run --json +``` + +`RUN_ID`, task paths and recovery evidence are operator-supplied values; the +runtime never infers them from chat history. + +## Packaged JSON schemas + +| File | Identifier | Required top-level fields | SHA-256 | +|---|---|---|---| +| `adapter.schema.json` | `https://devsquad.local/schemas/adapter-v1.json` | `schema_version`, `name`, `transport`, `binary_candidates`, `capabilities`, `permission_profiles` | `fc6b811b59bd2920d3dc1588d5bf447515a25d87219d07d76e1764efd60ad9c0` | +| `check-result.schema.json` | `https://devsquad.local/schemas/check-result-v2.json` | `schema_version`, `candidate_sha256`, `target_oid`, `id`, `argv`, `cwd`, `required_to_pass`, `status`, `returncode`, `error_code`, `duration_ms`, `stdout`, `stderr` | `5145755d8e60503151f8938f1753c657724f1a2de3a442500c14eda158fe04a1` | +| `council.schema.json` | `https://devsquad.local/schemas/council-v1.json` | `schema_version`, `enabled`, `automatic`, `reason`, `min_valid_proposals`, `required_critics`, `max_invocations`, `seed`, `evidence`, `rubric` | `8d72642179ce163a707168ad5e63e3e0cff794ce32afd422f232bfa5af13dff5` | +| `execution-identity.schema.json` | `https://devsquad.local/schemas/execution-identity-v1.json` | `harness`, `harness_version`, `model_provider`, `model_family`, `model`, `effort`, `tools`, `permissions`, `account_pool`, `verification` | `99386fb81c3567bdd3c880b37d363d5ca61cccb3ff35add5eb01a1ad916dfa90` | +| `launch-spec.schema.json` | `https://devsquad.local/schemas/launch-spec-v1.json` | `schema_version`, `adapter`, `transport`, `argv`, `cwd`, `stdin_path`, `timeout_seconds`, `requested`, `environment` | `12e2c3ffbfee6b3a9e93e41d9dd7d6a2c82a9d62e981d9378307b12f6ff12f5c` | +| `normalized-result.schema.json` | `https://devsquad.local/schemas/normalized-result-v1.json` | `schema_version`, `execution_status`, `error_code`, `output`, `artifact_status`, `acceptance_status`, `requested`, `observed`, `native_ids`, `events` | `8fb7eec0d3cfc952b9fa839b5fdbe950d5cfe06ed69bbbd3bcf835aad9df753d` | +| `policy.schema.json` | `https://devsquad.local/schemas/policy-v1.json` | `schema_version`, `id`, `version`, `roles`, `task_classes`, `require_different_model_for_review`, `account_pools`, `experiment_budget` | `b706f2fe91b0f22fe4e5d1c6d96ee48a8d5562040c8810c0f100ee9612b725a6` | +| `profile.schema.json` | `https://devsquad.local/schemas/profile-v1.json` | `id`, `harness`, `model_family`, `model_id`, `effort`, `required_tools`, `permission_policy`, `account_pool_id`, `billing_mode`, `quality_status`, `evidence_refs` | `fc51b49cd7dd6a134eedf2c2c941301392aef29a2163ac2d307b4fb78509ec96` | +| `profiles.schema.json` | `https://devsquad.local/schemas/profiles-v1.json` | `schema_version`, `profiles`, `bindings` | `c14823c89d1b81ee93b74f5bbc2701ae975825db795b5f0937c28fc057d84cf1` | +| `review-result.schema.json` | `https://devsquad.local/schemas/review-result-v1.json` | `schema_version`, `candidate_sha256`, `base_oid`, `target_oid`, `review_mode`, `verdict`, `summary`, `findings` | `96b852543f92dca21dafd6c0bc954f8f56d3dbfb0981d601cc05b9c1c9ba6e6b` | +| `task.schema.json` | `https://devsquad.local/schemas/task-v1.json` | `schema_version`, `project`, `workflow`, `goal`, `task_class`, `acceptance`, `checks`, `scope`, `lead`, `routing`, `budget`, `origin` | `37537e7f9c21ec07cb9b6a1076a69807278ec682103527059ab0415087ba3da3` | + +## Task-shape example + +This generated copy demonstrates the strict v1 task shape. Replace the +fixture repository, refs, routing files and checks before starting a run. + +```json +{ + "acceptance": [ + { + "description": "Findings identify the resolved candidate and supporting file locations.", + "evidence_kind": "review", + "id": "review-exact-diff" + }, + { + "description": "Include the fixture check result even when it reports a failure.", + "evidence_kind": "check", + "id": "report-check-outcome" + } + ], + "budget": { + "max_fallbacks_per_step": 1, + "max_revisions": 0, + "max_worker_invocations": 3, + "wall_seconds": 600 + }, + "checks": [ + { + "argv": [ + "python3", + "-m", + "unittest", + "discover", + "-s", + "tests" + ], + "cwd": ".", + "id": "fixture-tests", + "required_to_pass": false, + "timeout_seconds": 30 + } + ], + "goal": "Review the fixture change and report supported correctness findings.", + "lead": { + "mode": "host" + }, + "origin": { + "surface": "cli" + }, + "project": { + "base_ref": "fixture-base", + "repo_path": "/absolute/path/to/fixture-repository", + "target_ref": "fixture-candidate" + }, + "routing": { + "policy_file": "devsquad/policy.json", + "profiles_file": "devsquad/profiles.json" + }, + "schema_version": 1, + "scope": { + "read_paths": [ + "src", + "tests" + ], + "write_paths": [] + }, + "task_class": "fixture-review-small", + "workflow": "branch-review" +} +``` diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md new file mode 100644 index 0000000..ef2fa39 --- /dev/null +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -0,0 +1,414 @@ +# Engineering-team contracts v1 + +**Design contract, not an existing CLI reference.** These contracts implement [ADR-002](../../adr/ADR-002-surface-independent-engineering-team.md). M1 creates schemas; later milestones add the operations below without changing their meanings. Examples are deliberately fictional fixtures. + +## 1. Public operations + +All machine responses use `{schema_version: 1, ok: boolean, data: object|null, error: object|null}`. Exactly one of `data` and `error` is non-null. Errors contain `code`, `message`, `retryable` and structured `details`. JSON output contains no progress prose; stderr is for diagnostics. MCP invokes the same application functions as the CLI, without shelling out to string commands. + +| CLI | MCP tool | Effect | +|---|---|---| +| `squad doctor --json` | `squad_doctor` | Versions, capabilities, permissions, registration/install drift; metadata only | +| `squad start --task-file FILE --idempotency-key KEY [--supersedes-run RUN] --json` | `squad_start` | Validate, snapshot and persist; return run ID promptly | +| `squad status RUN --json` | `squad_status` | State, version, active steps, blockers, usage observations and next action | +| `squad events RUN --after CURSOR --limit N --json` | `squad_events` | Bounded event page and next cursor; no provider reasoning stream | +| `squad result RUN --json` | `squad_result` | Receipt and artifact references; `ready:false` before terminal state | +| `squad cancel RUN --json` | `squad_cancel` | Persist cancellation intent; return before termination finishes | +| `squad resume RUN [--recovery-file FILE] --json` | `squad_resume` | Reconcile ownership and effects; resume only if safe | +| `squad handoff claim RUN --expected-version N --owner OWNER [--claim-file FILE] --json` | `squad_handoff_claim` | Obtain or renew fenced host claim and input packet | +| `squad handoff complete RUN --claim-file FILE --decision-file FILE --json` | `squad_handoff_complete` | Submit disposition with current claim and evidence | + +MCP accepts parsed task/recovery/decision/claim objects instead of CLI file arguments, and optional `supersedes_run_id` on start. Bind request origin to the calling integration where possible; a user-supplied host label is provenance, not an authentication boundary. Local server operates as the same OS user. No network listener in v1. + +The MCP server entry is `squad mcp serve` over stdio. Its stdout is exclusively MCP transport traffic; logs go to stderr. Missing optional MCP dependencies produce actionable install guidance without preventing ordinary CLI use. + +`start --wait` is a CLI convenience using the same saved run. Ctrl-C exits observation without cancelling; print the run ID and explicit cancel command. Non-waiting accepted operations exit 0; invalid input exits 64; ownership/version conflict exits 75; runtime/internal failure exits 1. `--wait` exits 0 for `succeeded`, 2 for `blocked`/`awaiting_host`, 3 for `failed`, 4 for `cancelled`. `doctor` exits 1 when required readiness checks fail. A successfully retrieved failed run still gives `status` exit 0. + +Later M6 CLI additions: `capacity observe --file FILE`, `outcome add RUN --file FILE`, `report --project PATH`, `learn propose --project PATH`, `policy evaluate --experiment FILE`. These use the same envelope. Read-only reporting never modifies the active policy. + +Core operations are idempotent where specified, not “exactly once” execution of external effects. Reusing `(project_id, idempotency_key)` with an identical canonical request returns the original run; a different request hash returns `CONFLICT`. Retransmitted cancellations do not spawn another cleanup process. A repeated successful handoff completion returns its existing result if the submission hash matches; conflicting late completion is rejected and retained as an audit event. + +Hash the canonical **submitted** request, including an optional predecessor ID, before resolving moving refs or loading mutable policy contents. After structural validation and project identification, claim the key transactionally in a non-runnable `queued` record with `phase: preparing`. Only the owning preflight completes the immutable snapshots and enables execution. A replay returns that record without resolving refs again; concurrent preflights cannot overwrite it. A crash during preflight is recoverable under the same claim discipline. Syntax-invalid requests create no run; post-claim validation failures produce a failed receipt. `start` may return a preparing run while bounded preflight proceeds; it never waits for a provider call. + +## 2. Task and policy inputs + +The Task schema rejects unknown fields and checks finite bounds. It contains: + +| Field | Required meaning | +|---|---| +| `schema_version` | Integer `1` | +| `project.repo_path` | Existing absolute Git repository path, registered using canonical common directory | +| `project.base_ref`, `project.target_ref` | Resolve to immutable commit OIDs before launch; review compares base→target, delivery starts at target | +| `workflow` | `branch-review` or `issue-delivery`; no arbitrary shell/DAG step types | +| `goal`, `task_class` | Bounded objective and comparison class, e.g. `bugfix-python-small` | +| `acceptance[]` | Stable criterion `id`, concrete `description`, `evidence_kind` (`review`, `check`, `artifact`, `host`) | +| `checks[]` | `id`, `argv` string array, repository-relative `cwd`, `timeout_seconds`, `required_to_pass` | +| `review` | Optional `{mode, focus}`; mode defaults to `standard`, or `adversarial`; focus is allowed only for adversarial review | +| `scope` | Repository-relative `read_paths`, `write_paths`; non-empty writes only for delivery | +| `lead` | `mode: host` or `headless`; headless requires candidate profiles in policy | +| `routing` | `profiles_file`, `policy_file`; absolute or repository-relative trusted config files; optional per-role `overrides` | +| `budget` | `wall_seconds` of active run execution, `max_worker_invocations`, `max_revisions`, `max_fallbacks_per_step`; all finite non-negative integers, wall/invocations positive | +| `origin` | `surface` label, optional `session_ref`; no authorization or remote invocation implied | + +Snapshot the task, resolved refs, config files and their hashes before enqueue. Do not resolve a moving branch again halfway through a run. V1 uses committed inputs only: if dirty files intersect declared scope, reject with an actionable `INPUT_INVALID` rather than silently omitting them. Explicit dirty-worktree snapshot support is deferred. Validate paths against traversal and symlink escape. Task checks and policies are execution authority: repo content and provider output cannot add commands or widen permissions. + +`max_worker_invocations` counts launched adapter attempts, including retries, fallback attempts and headless lead invocations. It does not count native internal model requests: one CLI worker may run several model/tool turns. Observed native requests/tokens/quota and host work have separate nullable measurements. Pre-launch validation or capacity rejection does not consume a launch, though its elapsed work still counts toward execution time. The September selection/Council amendment renames the earlier design field `max_provider_calls` before schema implementation to make this unit explicit. Enforce native turn/tool limits only where the adapter verifies support; a worker-launch cap alone is not a token, quota or spending cap. + +The CLI must print the resolved scope/check plan in validation output; an app lead should supply it from the user's actual task. No magic inference from a README's embedded instructions. Checks may legitimately have side effects; run them in the isolated workspace with the declared process policy. “Read-only reviewer” does not mean executing arbitrary repository scripts is safe to treat as read-only. + +`Policy` contains an `id`, `version`, workflow role candidate lists, per-task-class minimum quality status, `require_different_model_for_review`, optional `prefer_different_harness_for_review`, account-pool policy and experiment budget. V1 role names are `implementer`, `reviewer`, `lead`; checks are deterministic process steps, not model calls. Add a `researcher` role only when a workflow needs specific research artifacts. Don't make every task pay for research or a council. + +Role candidate entries are tagged references `{kind: profile|alias, id}`. A profile reference resolves to an immutable concrete configuration; an alias such as `review.deep` resolves through the versioned qualified-binding registry. Preflight snapshots the resolved profile and fallback set, plus template/binding revisions. An alias may gain a qualified replacement for new runs without changing workflow files. Explicit `routing.overrides.profile_id` remains a concrete pin and never floats. Templates and binding changes follow the [model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md). + +`Profile` contains: + +```text +id; harness; model_family; model_id; effort {value, transport}; +required_tools[]; permission_policy; account_pool_id; billing_mode; +quality_status; evidence_refs[] +``` + +The path named by `routing.profiles_file` contains one strict profile registry, +not a bare array: `{schema_version:1, profiles:[Profile...], bindings:{...}}`. +Profile IDs are unique. Each binding is +`alias -> {profile_id, version}` and must target a profile in the same registry. +The registry's exact byte hash, every selected concrete profile/hash and every +resolved alias/version are frozen in preflight. Editing either file after that +cannot move an active run. Runtime-qualified binding promotion in M6 uses the +same versioned binding shape and affects new runs only. + +`effort.transport` is `native`, `model_variant`, or `provider_default`. Values are native to that harness/model; never translate “high” into a numeric equivalent across vendors. Unsupported explicit effort fails validation. Provider default may be allowed, but its effective value remains unknown unless reported. `quality_status` is `unvalidated`, `trial`, `proven` or `suspended`, scoped to task class by policy evidence. Trial profiles are eligible only in explicitly permitted classes. Exact family and model IDs are mandatory for cross-model independence claims; inability to verify identity blocks that claim. + +Install-time discovery reports supported values and evidence (`documented`, `probed`, `unavailable`, `unknown`) with `checked_at`, CLI version and toolset hash. Selecting a known catalog entry verifies that it exists, not that the invocation used it: attempts retain separate `requested` and `observed` fields. Manually verified mappings may establish identity for a versioned harness; silent model fallback must never be labelled confirmed. + +Claude implementation evidence v2 keeps the frozen requested profile separate +from `observed_identity`. The tested native result must have a success envelope, +bounded session ID and a concrete reported writer identity. Legacy native JSON +requires a single `modelUsage` entry; its key is the reported identity. Native +stream evidence v2 instead requires all top-level assistant messages to report +one concrete model under the final result's session, with that model also +present in terminal usage. Auxiliary usage entries remain visible and cannot +substitute for the writer; `canonicalModel` is pricing metadata only. +Tested family aliases may resolve to a reported concrete member of that family, +without hardcoded current revisions. Multiple usage-only entries, missing or +contradictory stream model/session reports, nested delegated messages and a +writer absent from terminal usage fail closed. Effective effort and backing revision remain null when +the result does not report them; `verification_scope: reported_model` does not +claim that these unknown settings were verified. Native session, typed usage, +alias resolution and the normalized result evidence are retained and validated +again on coordinator import, before candidate finalization. +The tested Claude 2.1.220 transport label `firstParty` maps to Anthropic only +for that verified harness version; the original label remains in evidence. +Unknown or contradictory provider labels still block identity verification. +Native error streams can report synthetic assistant messages; their strict +final error terminal is classified without claiming writer identity. A native +not-logged-in error reports AUTH_ERROR, ahead of a concurrent rate banner. + +New delivery review imports and acceptance require verified different reported +model IDs, regardless of requested aliases, family labels or harness names. +Unknown or mixed native/fixture identities cannot establish independence; +explicit all-fixture runs test orchestration only. Legacy native implementation +v1 receipts remain readable and exact terminal decisions replayable, but cannot +authorize new acceptance. Rejected native results retain unverified, bounded +model/session/usage diagnostics, native-output hash/size and durable stream +artifact references across failure, fallback and cancellation. Provider prose +and arbitrary result fields are not copied into those diagnostic projections. + +### Automatic selection and manual overrides + +Default selection is automatic among policy-eligible profiles. The host supplies task requirements; the deterministic router chooses the model/effort/tool profile without another planning-model call. Within the selected toolbox, the worker chooses individual tool calls. Discovery can enumerate supported configurations; initial quality preferences and account-pool mappings still require evidence and operator setup. + +`routing.overrides` defaults to `{}`. Each key is a model role supported by the chosen workflow, with value `{profile_id, fallback}`; `fallback` defaults to `none` and optionally allows `policy`. A pin constrains the initial profile; `none` also prohibits substitution/escalation to another profile. A pinned profile must pass the ordinary identity, capability, quality, permission and billing filters. Invalid pins fail validation; temporarily unavailable pins block. `policy` permits the normal qualified fallback list within the task budget. Roles without pins remain automatic. Record overrides and their origin in the routing receipt and separate them from automatic decisions in evaluations. + +Each `policy.account_pools` entry has +`allowed_billing_modes[]`, positive `max_concurrency`, and optional +`unknown_capacity_policy` (`allow_bounded` by default or `block`). Runtime +availability is a separate snapshot with status `available`, `exhausted` or +`unknown`, a non-negative local `in_flight` count and an optional sourced +timestamp. Unknown capacity never becomes an invented allowance: +`allow_bounded` permits at most one unresolved in-flight job in that pool, +while `block` permits none. Declaring `paid_api` in a profile is insufficient; +the pool policy must explicitly allow that billing mode. + +The [selection and Council amendment](SELECTION-AND-COUNCIL.md) gives examples and ownership boundaries. Evidence gathering/proposals are automatic. Initially promotions are reviewed; the [model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md) permits tested model-binding promotion under an explicitly enabled, previously reviewed `guarded_auto` policy. Changing permissions, billing authority or the promotion policy itself remains reviewed. Optional C1 is outside the initial two-workflow schema until that gated extension ships. + +## 3. Adapter execution boundary + +Retain `invoke_codex`, `invoke_gemini`, `invoke_grok` and their legacy stdout/exit/error contract. Add a Claude headless adapter. Adapters declare `transport: cli_exec | native_protocol`, while exposing the same normalized lifecycle/results. The CLI transport uses a bridge around shared wrapper configuration and classification helpers: + +1. `prepare(request.json)` returns `LaunchSpec`: fixed manifest-selected executable, argv array, working directory, optional stdin artifact, allowlisted environment overrides, requested model/effort, parser version, permission evidence. The bridge may source bundled wrapper functions; it must not source a request-selected file. +2. Python launches that argv directly in a new process session/group, with no `shell=True` or `eval`, owns lifecycle and captures bounded stdout/stderr to files. +3. `classify(exit_code, stdout_file, stderr_file, timed_out)` returns the legacy-compatible error class or execution completion plus native model/usage/session metadata. Extract shared logic rather than implementing two independent classifiers. A Bash argv builder can send NUL-separated arguments to a bundled Python serializer; never interpolate JSON strings into shell code. + +The new bridge does not call `_adapter_invoke`'s watchdog or write legacy JSON usage arrays. Python writes new-run telemetry once. Existing sourced callers retain their old bookkeeping. Per-run model/effort/permission overrides must not edit shared configuration. Native CLI timeouts may act as an earlier provider limit, but Python remains the sole supervisor and cleanup owner. + +For Codex, prefer a version-verified `native_protocol` adapter backed by a supervisor-owned stdio app-server. The adapter prepares typed thread/review/turn requests, maps native events into core events and retains native IDs. Requests come from validated task fields and adapter code, not arbitrary caller-supplied protocol methods. Use native interruption before process cleanup; require terminal evidence. An acknowledged start/interrupt is not task completion. Native errors normalize into the same legacy-compatible taxonomy plus structured core details. A broken transport cannot trigger a second writer without reconciliation. See the [native adapter amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md) for capability/version gates and retained Bash compatibility. + +The four adapter error codes stay `RATE_LIMITED | AUTH_ERROR | TIMEOUT | CLI_ERROR`, with auth checked before rate. Additional **core** errors include `INPUT_INVALID`, `PROFILE_UNSUPPORTED`, `CAPABILITY_UNAVAILABLE`, `CONFLICT`, `RECOVERY_REQUIRED`, `BUDGET_EXHAUSTED`, `POLICY_DENIED`, `INTERNAL_ERROR`. Do not expand the legacy prefix enum to represent orchestration states. + +Execution completion is separate from deliverable validity and acceptance. Empty output, malformed required JSON, tool/permission denial or an authentication banner can invalidate a nominal exit-0 invocation. Save raw native output and the parser verdict. Prompt compliance alone does not prove a reviewer was read-only: require a verified native restriction or OS process policy; otherwise that role is unavailable. Do not use the wrappers' current blanket approval flags in new worker profiles. + +Use Git's tracked-file inventory and explicit task scope for context. Preserve filenames with spaces, TSX/JSX and other tracked extensions. Bound bytes and document exclusions; never silently truncate required evidence. Exclude runtime state, secrets and ignored files by default; an explicitly required ignored input needs an intentional input artifact. Large context should use native scoped filesystem access when supported rather than concatenating every file. + +Workers get `DEVSQUAD_WORKER=1`, run/attempt IDs and a delegation-depth guard. DevSquad's worker-facing MCP tools reject new team starts and workflow mutations, and legacy hooks honor the guard. Native authentication remains available through the provider's normal mechanism, but credentials and environment contents are not logged. Do not assume prompt text alone stops recursive delegation. +Detached launch preserves PATH, the frozen package PYTHONPATH, HOME and USER +for the normal saved-login mechanism. It does not inherit API credentials, +provider overrides or arbitrary host environment variables. + +## 4. Durable state, concurrency and recovery + +SQLite schema has `projects`, `runs`, `steps`, `attempts`, `events`, `artifacts`, `claims`, `pool_observations`, `pool_reservations`, `outcomes`, `profile_templates`, `profile_bindings`, `qualification_runs` and `schema_migrations`. Entity IDs are opaque UUIDs; event cursors, versions, fencing tokens and migration versions are integers. UTC timestamps accompany durations measured with a monotonic clock. Mutations increment a run `version`; the append-only event and affected projections commit together in one short transaction. Binding updates use their own compare-and-swap version and event. File artifacts are atomically finalized and hashed before a transaction references them; crash-created unreferenced files are recoverable garbage, not valid results. + +```mermaid +stateDiagram-v2 + [*] --> queued + queued --> running + running --> awaiting_host + awaiting_host --> running + running --> blocked + blocked --> running: resume after reconciliation + running --> succeeded + running --> failed + queued --> cancelled + running --> cancelling + awaiting_host --> cancelled + blocked --> cancelling + cancelling --> cancelled: all children reaped or absence confirmed +``` + +Terminal states are `succeeded`, `failed`, `cancelled`; they are immutable. Retrying a terminal run uses `start --supersedes-run RUN` with a new idempotency key and a supplied task. Validate the predecessor is terminal and belongs to the same project; persist the link and resnapshot the new request. Terminal `resume` returns `CONFLICT` with this next action; it never creates a run. `blocked` has a typed reason and next action. A crash-recovery scan reconciles stuck `running` jobs. Event cursors are monotonically increasing database IDs, filtered per run (gaps are normal), with bounded pages. Preflight may transition `queued` directly to `failed`; cancellation of a queued preflight must fence its late snapshot completion. + +Enforce one active writer attempt per worktree and one supervisor claim per run transactionally. Store PID, process-group ID, process-start identity, heartbeat, attempt token and package digest. Lease expiry alone never licenses a replacement writer. On lost supervisor, inspect the child identity, pending writes and last artifact boundary. A live child remains owned; observation may reattach without relaunch. Ambiguous identity returns `RECOVERY_REQUIRED`; do not kill a reused PID. + +Cancel: persist intent → signal the verified process group → drain output → bounded grace (default five seconds) → kill if needed → reap/confirm absence → record terminal state. Total run and per-step deadlines include retries. Keep `cancelling` with diagnostic evidence if termination cannot be confirmed; no competing writer may start. Native detached remote effects outside this process group are a separate adapter limitation: unsupported tools cannot be advertised as safely cancellable. + +`budget.wall_seconds` counts elapsed active execution, including preflight, checks, retries and cleanup; parallel steps consume wall time once per run. It pauses in idle `queued` (excluding active preflight), `awaiting_host` and `blocked` only after all worker processes are stopped and ownership reconciled. Host thinking and quota waiting can therefore continue overnight without consuming execution allowance; record them separately as waiting/total elapsed time. Exhaustion stops work, performs bounded cleanup and ends `failed` with `BUDGET_EXHAUSTED`; cleanup may exceed the budget only to enforce safe termination. A lost heartbeat never pauses the clock while a child may still be running. + +Resume distinguishes (a) live work, (b) a verified resumable native session, and (c) interrupted work needing a new attempt. Native session IDs are only used if the installed adapter has a tested resume capability. Default interrupted writers need a recovery disposition referencing attempt ID, current workspace/patch hash, known effects and chosen checkpoint. Validate this evidence before creating the next attempt. A retry must not blindly repeat external effects. V1 worker policies disable publish/deploy and other irreversible remote actions. + +Host handoff claims use a bounded lease and increasing fencing token. `claim` needs the expected run version; `complete` needs the current token, handoff ID, submission ID/hash and evidence refs. An expired host can no longer advance the run. Save its late submission for audit without applying it. `accept`, `revise`, `reject` are the allowed dispositions. Acceptance cannot override a failed mandatory check, stale revision, exhausted correction budget or missing independent review. + +Default host lease is ten minutes. A competing claim while a live claim exists returns `CONFLICT`; no implicit takeover. The current owner renews with `claim` plus its claim object and current expected version before expiry. Renewal extends the same fencing token; after expiry a successful new claim increments the token, even for the same owner label. Claim responses contain handoff ID, token, expiry and run version. Cancellation invalidates claims. This lease governs permission to submit a decision, not the lifetime of the saved handoff. + +## 5. Workflows and acceptance + +| Template | Ordered steps | Completion | +|---|---|---| +| `branch-review` | Resolve base/target → frozen review workspace → reviewer → configured checks → lead disposition | Valid review artifact, evidence for each criterion, recorded check outcomes, accepted disposition | +| `issue-delivery` | Snapshot target → implementer → frozen candidate → different-model reviewer → checks → lead disposition | Candidate meets all criteria, independent review complete, mandatory checks pass | + +For review-only tasks a check may have `required_to_pass:false`: a failing test then becomes a finding in a successfully delivered review. Delivery's required checks must pass. Mark these policies before execution. A valid critical finding is a useful reviewer contribution, not automatically a failed reviewer attempt. + +For `issue-delivery`, `revise` loops back to implementation. For `branch-review`, it requests a corrected review of the same frozen diff; it never authorizes source edits. Both consume one revision allowance and require a reason. With no allowance remaining, a revise request ends the run failed with `BUDGET_EXHAUSTED` after cleanup. Each role may attempt bounded fallbacks; permission/capability restrictions carry across them. Nonrecoverable step failure blocks dependent steps. No continuing into acceptance after missing implementation or invalid review. A headless lead's invalid disposition is a failed attempt, not permission to invent success; changing lead mode requires a new explicitly supplied task/run in v1. + +Bind artifacts, findings, tests, dispositions and criteria to a candidate tree/patch hash and resolved baseline. Freeze an implementation candidate into a local run-owned commit before review; reject changes outside scope and capture untracked permitted outputs intentionally. Run checks in a separate candidate worktree so test-generated files cannot alter the reviewed candidate. If a check must change source, that creates a new candidate needing new review. A subsequent code change invalidates affected evidence and triggers review/checks again. Run results contain a patch/commit reference and integration instructions; the coordinator does not alter the user's current checkout. + +Each check may declare `output_paths` (default `[]`): at most 32 canonical +repository-relative file/directory paths for untracked build artifacts. These +are frozen with the check plan; root/Git metadata/traversal paths are forbidden. +An output declaration never permits changing tracked candidate files. Ignored +files are not implicitly approved outputs. Approved outputs may be consumed by +later checks; undeclared new inputs, tracked content/mode/index changes, HEAD +changes, or mutation of the review worktree invalidate the check and skip the +remaining checks. This integrity gate also blocks report-only check acceptance. +Checks compare actual tracked bytes, independent of Git stat-cache and +assume-unchanged hints. Unsupported Git submodule inputs fail closed. + +Check evidence v2 records integrity status, bounded mutation path/digest +witnesses and complete state fingerprints alongside the original process exit +and output. Both handoff and terminal reports retain the failure, including +when the lead fails. Legacy v1 receipts remain readable and completed decisions +remain replayable; they cannot authorize a new acceptance. Start a new run to +refresh unverified legacy checks. Boundary verification applies to trusted +checks; it is not a sandbox against a malicious command that mutates and +restores source entirely within one check invocation. + +Every attempt receipt includes role, parent step, profile/policy/prompt/schema versions, requested/observed model/effort, tools/permissions, runtime version, input/output hashes, process verdict, artifact verdict, latency, usage by source and error. The run receipt also includes criteria results, all attempt IDs, fallbacks, revisions, lead repairs, final disposition and remaining limitations. Host work is recorded as externally observed with unknown usage where unmeasured; do not omit it or estimate it as zero. + +## 6. Capacity without false precision + +Model providers, execution harnesses and billing accounts are separate identities. `account_pool_id` is a user-configured opaque identifier; no credentials. A pool may contain multiple windows and model-family sublimits. Unknown mapping is explicitly unresolved; do not assume two apps give two independent allowances or that every model within an account shares one limit. + +Observation fields: `pool_id`, `window_id`, `applies_to`, `observed_at`, `expires_at`, `source`, `used`, `limit`, `unit`, `resets_at`, `confidence`. Unknown measurements are null. Sources distinguish native reported values, manual reports and estimates. Percent observations retain their native unit; do not convert characters or token estimates into subscription percentage. Validate bounds and clock skew. Ignore stale observations for hard capacity decisions, while showing their last-known values. + +Route order: + +1. Capability, permissions, model/effort support and task-class quality eligibility. +2. All applicable fresh quota windows, auth failure, known cooldown and local in-flight concurrency/reserve constraints. +3. Versioned candidate preference, deadline and measured latency; log exclusions and selection rationale. + +An explicit `unknown_capacity_policy` is `allow_bounded` or `block`; default `allow_bounded` permits one short trial/in-flight job per unresolved pool, with normal per-run budgets, and labels uncertainty. A fresh known exhausted window blocks all affected profiles. Reset timestamps permit a refresh/reconsideration; they do not prove fresh availability. Auth errors require observed repair; rate errors use provider retry/reset metadata or conservative recorded cooldown. Never loop until a provider happens to recover. + +Local reservations are concurrency/scheduling records, not provider quota guarantees. Other apps can consume the allowance during a run. An account-level usage delta is not assigned wholesale to one task. Record per-call usage only when the provider identifies it, and preserve uncertainty. Billing mode is `subscription` or `paid_api`; policy must explicitly permit paid API profiles before selection. A subscription's marginal cash price is not its opportunity cost. + +## 7. Learning, evaluation and ongoing docs + +Run completion writes local `receipt.json`, `receipt.md`, `events.jsonl` export and an artifact manifest, including failures/cancellation. Waiting writes `handoff.json` and `handoff.md`. Raw outputs stay local. Tracked distilled records use: + +```text +devsquad/ + profiles.json, policy.json versioned intended configuration + learning/observations/.md outcome + provenance + later correction + learning/experiments/.json question, comparison, budget, stop rule + learning/evaluations/.md cases, failures, uncertainty, verdict + learning/decisions/.md adopt/reject/no-change + rollback target + learning/policy-changelog.md links to evidence and policy diffs +``` + +`report` derives summaries from the database; `learn propose` writes draft distilled records for review. It does not change routing. Raw task content is not automatically committed or uploaded. Track redacted evidence with reproducible case IDs/hashes and an explicit `evidence_availability` value (`local`, `tracked_fixture`, `unavailable`); a local path alone is not portable proof. Policy changes require review with evaluation links; use an ordinary Git diff/review during M6. A separately enabled `guarded_auto` qualification gate may update a model binding within that policy, writing an evidence/rollback receipt for each promotion. Runtime binding updates do not edit the user's Git checkout. + +Use this loop: + +1. Define task class, criteria and intended comparison before execution. +2. Record every attempt, including failure, fallback, reviewer contribution and lead repair. Attach later corrections/escaped defects via `outcome add`; append revisions to verdicts, never erase the original. +3. Propose one change: model, effort, prompt/context strategy, tool access or workflow. Compare like tasks and keep the remaining settings controlled or explicitly record confounders. +4. Run a small predeclared evaluation with held-out cases, budget and stopping rule. Avoid routing only hard tasks to one model and then treating unadjusted averages as model quality. +5. Promote only with supported quality evidence, acceptable rework/latency/capacity tradeoff and a rollback target; otherwise preserve the policy and record no-change. +6. Revalidate affected profiles after model/harness/prompt/tool/policy drift or escaped defects. Do not discard unrelated evidence or promote a model on release/discovery alone. Any guarded automatic binding promotion must satisfy the versioned qualification/evaluation rules. + +Default experiment budget is disabled until explicitly configured, then at most 10% of eligible runs with a hard call/time cap. Most work uses the current proven policy. V1 uses human-governed static preferences, not exhaustive permutations, an automatic bandit or foundation-model fine-tuning. Report sample sizes and missingness; tiny samples justify hypotheses, not provider rankings. + +New public runs (schema 16) request an objective final-outcome projection at +admission. Terminal receipts remain immutable. A durable outbox and exact +outcome ID make projection replay-safe after a crash; status/result, project +reporting and experiment evaluation repair only their relevant pending jobs. +Legacy history is not retrospectively assigned or rewritten. Prelaunch +failures/cancellations have no worker contributions. Completed worker failures +remain failed; subsequent fallback/revision success is repair, not independently +successful original work. Reviewer findings and explicit lead revisions remain +visible. Successful criterion status cites the fenced lead's final acceptance, +not a fabricated check result; unevaluated criteria stay unknown. Subjective +later feedback is an explicit append-only late correction, not another final. + +The opt-in `trial --experiment FILE --case ID --arm control|candidate +--task-file FILE --idempotency-key KEY` command starts one arm through the +ordinary runner. It freezes the complete v2 declaration under the existing +preparation fence before any attempt. Reviewer trials require branch-review's +same frozen candidate; implementer trials require issue-delivery's same +baseline/task/check contract. The shared reservation transaction counts all +durable experiment attempt slots (including failed, fallback, revision and +in-flight slots) against the declared call cap; the wall deadline starts at +declaration freeze. Controller limits are 100 cases, 1,000 reservations and +3,600 seconds, never an entitlement to spend that much. Missing/unstarted arms +do not become completed pairs. This explicit manual operation does not enable +automatic experimentation, dispatch a background campaign or promote a binding. + +Measure acceptance and critical defects first; also show retries, lead rework, elapsed time, measured usage by pool, blocked time and unmeasured overhead. Final task success and original worker quality are distinct. Pair deterministic checks with review and human correction; a model judging itself is not sufficient evidence. + +### Independent experiment provenance (v2) + +Outcome IDs are globally unique across all cases, arms and splits of a new +evaluation. A profile-binding experiment v2 adds its tested `role` and both +concrete `control_profile_sha256` / `candidate_profile_sha256` values, plus +per-arm `control_execution_sha256` / `candidate_execution_sha256` fingerprints +of the profile and frozen native adapter (including binary/version/provider). +Each +case declares two distinct fingerprints: `case_sha256` identifies the exact +task/candidate corpus item independently of runtime settings, while +`input_sha256` binds the controlled pair context. Corpus identities cannot be +relabelled across evaluation/held-out splits by changing package or policy. + +The corpus projection retains frozen source identity, goal, workflow/task +class, acceptance, checks (including argv/cwd/output declarations), scope and +review mode/focus. Implementer comparisons use the original baseline, never +their produced candidates; reviewer comparisons include the exact candidate. +Pair context additionally retains runtime package, policy, lead mode, budgets, +supporting-role profiles/adapters and all fallback configuration. Only the +explicitly declared tested execution binding and incidental machine paths/origin/runtime capacity +observations are excluded. Absolute path-looking check arguments are not +blanket-removed. Assignment/spec hashes are excluded to avoid a hash cycle. + +An assignment binds the spec hash, project Git common directory, case, split, +arm, tested role/profile/execution fingerprints, both input fingerprints and outcome ID. +Schema 14 records the full spec and original preparation snapshot under the +run's preparation fence, atomically before any attempt. One run cannot be +rebound; one experiment arm cannot acquire a second run. This is an ordering +guarantee, not a comparison of wall-clock timestamps. Plain runs are unchanged. + +Evaluation consumers must derive provenance from those saved records and all +actual relevant attempts. A supplied assignment dictionary or experimental +label is not authority. V2 normalized chains bind final/correction hashes and +distinct run/attempt IDs; an absent partner cannot hide invalid provenance. +Their evidence digest changes when outcomes or correction chains change. +Historical rows remain immutable; current eligibility for a new qualification, +promotion, rollback or fallback must be checked separately from historical +read/replay. Implementation progress and remaining reader/eligibility/public +controller work are tracked in [RESUME.md](RESUME.md), not implied by this +contract or by pure normalized-chain unit fixtures. + +### Current eligibility and explicit evaluation revisions + +Historical evidence and present authority are distinct. One transaction-bound +eligibility result identifies the experiment, spec SHA256, pinned evaluation +SHA256, saved/current evidence SHA256 and concrete ineligibility reasons. +Legacy v1 evaluations remain readable but cannot authorize a new qualification, +promotion, regression rollback or qualification-backed catalog fallback. +Changed final/correction/attempt evidence invalidates a prior evaluation; it +does not rewrite it or automatically change a binding. Exact completed binding +decision replay returns its original receipt without another mutation. + +R3b.2's agreed implementation contract is an explicit append-only review with +`policy evaluate --experiment FILE --revision-id ID --previous-evaluation-sha256 SHA`. +Both revision arguments are required together. The predecessor must be the +latest saved evaluation, checked in the same write transaction. A revision +uses the original frozen spec and assigned runs, never relabelled new samples. +Schema 15 adds revision history beside the unchanged original `experiments` +row. A repeated revision ID is idempotent only for the same experiment/spec +and predecessor, and still checks present evidence. Responses identify the +revision and predecessor; qualification and decision records pin the existing +evaluation SHA256, which identifies either the original or reviewed revision. + +New qualifications must match the full tested candidate fingerprint and all +assigned candidate tasks' declared role/task class. Qualification replay and +every new binding mutation recheck current evidence within their write +transaction. A new promotion must compare the tested control fingerprint with +the current incumbent, not merely find a candidate that passed against some +other profile. Regression rollback requires complete evaluation and held-out +pairs and both exact tested fingerprints. A proven bootstrap predecessor without qualification retains its +existing explicit baseline contract. Installation of this schema remains +gated on old active/recoverable-run upgrade safety; adding the migration does +not establish that installation gate. Implementation status is in RESUME.md. + +## Bounded manual Council extension (R7 source gate) + +`council-decision` is an explicitly enabled, automatic-OFF, read-only workflow. +Its strict `CouncilSpec` freezes a bounded question, evidence IDs/hashes, +rubric, seeded labels, two valid proposers, one distinct verified critic and a +worker-invocation cap. Three exact distinct entitled model IDs under the same +verified Codex subscription harness suffice; cross-family preference cannot +substitute for verified entitlement/identity. One round is supported, with no +internal revision or silent retry beyond the frozen fallback budget. + +Each proposer sees only its frozen common evidence. The trusted coordinator +commits each immutable proposal before the critic starts; the critic receives +sanitized seeded A/B proposals without raw identity provenance. The reversible +mapping, actual identities, native IDs, nullable usage, checks and artifacts +are retained separately for audit. Partial anonymity is not a quality claim. +The sole existing lead chooses A, B or synthesis, identifies supported claims, +discarded alternatives, all unresolved objections and validation. Missing, +empty, invalid, cancelled, quota-exhausted or failed roles cannot become quorum; +agreement cannot override failed mandatory checks or integrity constraints. + +Native participant processes require the frozen macOS default-deny Seatbelt +profile and actual per-run own-read/peer-runtime-denial probes. Native Codex +permissions also remain read-only. Saved-artifact MCP defense applies only to +the explicit Council worker context, preserving ordinary worker read behavior; +the OS boundary remains authoritative even without that context marker. +Unsupported OS/capabilities fail unavailable, never launch without isolation. +Bootstrap success alone is not subscription/network or generating readiness. + +Stages reuse existing versioned run/attempt/account ownership and artifact +tables. Submitted inputs have a separate immutable origin digest; stage +projections and imported artifacts are checked under existing transaction +fences. Claim/submission recovery retains exact saved evidence and decision; +guided host intent belongs to the actual acquired/taken-over claim event, not +an owner label or independent intent artifact. Terminal Council reports project +exactly one R5 final outcome with all contributions and truthful missingness. + +Council workflow comparison is separate from R3 profile-binding eligibility: +predeclare matched and held-out workflow cases, preserve input/profile/prompt/ +evidence versions and actual receipt references, and report benefit, harm or +inconclusive without invented scores. Fixture mechanics cannot establish native +quality; automatic triggering remains OFF until a separately accepted gate. +Installed cross-version contract epoch and native/comparison acceptance remain +open at this source checkpoint; see RESUME.md. diff --git a/docs/plans/engineering-team/DECISION-CLASSIFIERS.md b/docs/plans/engineering-team/DECISION-CLASSIFIERS.md new file mode 100644 index 0000000..6dc81f2 --- /dev/null +++ b/docs/plans/engineering-team/DECISION-CLASSIFIERS.md @@ -0,0 +1,254 @@ +# Typed decision classifiers: Jev, Laya and DevSquad + +Decision date: **2026-09-26**. Status: **source evaluation and planning +amendment; runtime integration and performance evaluation are pending**. + +## Recommendation + +Evaluate a small typed classifier as an optional **decision helper**, not a +replacement lead, coding worker or authority boundary. Per the user's September +26 direction, start with a **single capped Jev synthetic pilot** because it +avoids Laya's local model setup. The pilot is limited to one billable request, +no retries, a $0.01 ceiling and no repository/private task content. Move to a +local, default-off Laya trial if Jev becomes materially costlier, cannot be +accessed, or fails the quality/latency gate. Existing paid coding subscriptions +do not establish access to the Jev API. Do not add a mandatory model dependency +or delay M5's delivery/revision loop after the bounded probe. + +The current [router](../../../plugin/core/src/devsquad/router.py) is deterministic +and makes no model call. A classifier adds routing latency. Its potential value +is better task/profile matching, fewer unnecessary tool loads, less irrelevant +context and less downstream rework. No DevSquad speedup, allowance saving or +number of saved subscription windows has been measured. + +The initial evaluation inspected public documentation, benchmark reports and +pinned Laya source. It did **not** install weights or run Laya inference. The +tracked [Jev pilot specification](experiments/jev-pilot-v1.json) and validated +[probe](../../../test/core/probes/jev_decision_eval.py) subsequently completed +their one-request live smoke on October 1. See the +[redacted receipt summary](evidence/M6-D2-jev-pilot-2026-10-01.json): exact pinned +model, 5,373 reported input tokens, estimated $0.00022567, 8/8 task-family and +skill labels, but 6/8 execution-tier labels. The two tier disagreements remain +failures against the frozen expectations; the classifier is still off. + +This closes M6-D2's synthetic mechanics/access measurement only. It does not +prove production quality, calibration, speedup or adoption. The cost/access +Laya trigger did not fire; this smoke has no frozen numeric production accuracy +threshold from which to claim a quality-triggered switch. Broader evaluation +needs predeclared data, held-out gates and separately authorized resources. +The existing one-request allowance is spent: do not rerun the live command +below without new authorization. + +## What was verified + +| Candidate | Relevant capability | Constraint for this project | +|---|---|---| +| **Jev, TypeSafe System One** | Hosted typed choice, rubric-score and boolean decisions; multiple questions over shared state | Not a code generator. A separate network/API dependency with separate billing and privacy decisions | +| **Laya** | Apache-2.0 local typed decision models with English, multilingual and typed-workflow variants | Optional heavy inference environment and model assets; local latency, memory, calibration and engineering-task quality are unverified | + +Jev's documented current version is `jev-1.13.0`; `jev-latest` is mutable. +Published pricing is $0.042 per million input tokens, with output tokens free. +The documented limits are 64k aggregate context and 32k for state plus the +longest question. These are provider claims/current specifications, not local +measurements. Pin the concrete model and recheck terms at integration time. +[Jev models and pricing](https://docs.typesafe.ai/models). + +Choice/score confidence describes the output distribution; it is not proof +that a decision is correct. Jev documents weaknesses in exact numeric tasks, +indirection and adversarial input. Its coding-agent guidance explicitly +distinguishes typed decisions from generation and tool execution. +[Confidence](https://docs.typesafe.ai/confidence), +[known weaknesses](https://docs.typesafe.ai/model-jaggedness/jev-1.13), +[coding-agent boundary](https://docs.typesafe.ai/introduction/coding-agents). + +The inspected Laya source is package `0.3.20`, commit +`4066d5d5fbf08b66c6757ddeedbd797bd7655bc0`. Its dependencies include PyTorch and +Transformers, unlike DevSquad's dependency-free core. Revision/hash pinning is +supported but must be selected explicitly. Its `Router` chooses among Laya +checkpoints using language/script and optional question-ID heuristics; it is +**not a coding-provider selector**. The bundled Hub revision observed was +`55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851`; pin individual assets in the eventual +experiment manifest, not just a moving model name. +[Package](https://github.com/NandhaKishorM/laya/blob/4066d5d5fbf08b66c6757ddeedbd797bd7655bc0/pyproject.toml), +[router](https://github.com/NandhaKishorM/laya/blob/4066d5d5fbf08b66c6757ddeedbd797bd7655bc0/laya/router.py), +[revision handling](https://github.com/NandhaKishorM/laya/blob/4066d5d5fbf08b66c6757ddeedbd797bd7655bc0/laya/revisions.py), +[model](https://huggingface.co/convaiinnovations/laya). + +Laya's published 32.8 ms T4 one-question median is not Mac end-to-end latency. +Its Jev comparison uses third-party numbers, not a matched head-to-head run. +The reported model-routing task measures domain classification, not which +coding model delivers a correct patch. Results also show language/calibration +weaknesses and degradation with many labels. Its own guidance suggests small +option sets. These findings justify evaluation, not automatic adoption. +[Pinned benchmark report](https://github.com/NandhaKishorM/laya/blob/4066d5d5fbf08b66c6757ddeedbd797bd7655bc0/BENCHMARKS.md). + +Laya can truncate state/question content; long-input aggregation retains a +window's confidence, not a calibrated document-level probability. Choice/score +entropy confidence and maximum answer probability are different fields. Keep +raw scores and our separately evaluated calibration, rather than normalizing +every provider's “confidence” into a supposed universal success probability. +[Pinned inference code](https://github.com/NandhaKishorM/laya/blob/4066d5d5fbf08b66c6757ddeedbd797bd7655bc0/laya/agent.py). + +## Other places to use this pattern + +Priority is an evaluation order, not a promise to enable every use case. +P1/P2/P3 below indicate sequence, not defect severity. + +| Priority / use | Bounded classifier output | Potential benefit and required boundary | +|---|---|---| +| P1 — Task intake and routing hints | Task family, complexity band, ambiguity/missing-information flags; rank already-qualified profiles | Improve template/profile matching. The lead validates the task; hints cannot lower its quality floor, change scope or invent capabilities | +| P1 — Skill and tool shortlist | Up to a few relevant IDs, including `none`/`uncertain`, from an approved catalog | Avoid irrelevant tool/skill loading. Read selected instructions fully; retain catalog access, validate actual arguments and keep worker permissions unchanged | +| P1 — Context and retrieval ranking | Relevance labels over bounded file, diff, log or retrieved-passage candidates | Reduce irrelevant context. Never remove mandatory instructions, acceptance criteria, failed checks or review dissent; retain source references and measure evidence recall | +| P2 — Failure triage | Semantic category for otherwise unclassified diagnostics, plus suggested next action | Group unknown failures for investigation. Exact auth/rate/timeout codes, exit status and protocol evidence remain authoritative; no automatic retry or paid fallback | +| P2 — Review and test attention | Finding clusters, subsystem/risk tags, relevant optional test IDs | Focus a reviewer and suggest extra checks. Preserve original findings and provenance; never suppress a blocker, waive independent review or skip required tests | +| P2 — Outcome learning and documentation | Proposed repair/failure labels, affected requirement/doc IDs | Make comparable cohorts and identify documentation gaps. Labels need evidence and correction; they cannot declare acceptance, author documentation or promote defaults | +| P3 — Council/escalation suggestions | Disagreement/ambiguity flags linked to evidence | Help the existing lead spot cases worth deliberating. No classifier-only council trigger, extra worker launch or change to C1's explicit budget and evaluation gate | + +Skill suggestion has unusually relevant prior evidence: TypeSafe reports fewer +wrong and unnecessary skill loads in a 488-request, 182-skill experiment. It +used Jev 1.12 and synthetic requests, not DevSquad; the agent retained access to +the full skill index. Treat this as a testable hypothesis, not a reproduced gain. +[Official skill-suggestion experiment](https://docs.typesafe.ai/cookbooks/skill_suggestion). + +The target is catalogs passed to DevSquad-managed workers. This does not allow +DevSquad to bypass the host's skill instructions, alter global AI settings, or +replace provider-native internal tool selection it cannot control. Context +selection is ranking, not generative summarization; complicated synthesis and +patch writing remain lead/worker tasks. + +Do **not** use a probabilistic classifier for quota arithmetic/reset times, +authentication truth, observed model identity, capability verification, lock +ownership, crash recovery, Git/candidate integrity, permission grants, mandatory +check acceptance, or the sole security/redaction gate. Existing exact checks +are cheaper and authoritative for these jobs. + +## Minimal architecture amendment + +This extends M6 evaluation and M7 optional packaging. It does not reopen M1–M5 +or remove any existing acceptance gate. Automatic self-training and unreviewed +learned policy changes remain deferred. + +1. **Three explicit modes:** `off` (default, current behavior), `shadow` + (save suggestions without changing execution), and `advisory` (only a + reviewed, versioned policy after the use-case gate passes). Classification + is optional preprocessing; the router consumes frozen validated data and + remains deterministic. A cache miss never silently enables inference. +2. **Constrain before ranking:** exact policy filters establish permissions, + billing, capabilities, verified identity, quality and current capacity. + Only permitted candidates may be scored. Honor pins and explicit fallback; + revalidate volatile capacity at reservation. Suggestions cannot change task + class/requirements or expand the eligible set. No eligible candidate still + means blocked, not “let the classifier choose.” +3. **Typed adapter contract:** versioned purpose/question/rubric, input and + evidence hashes, candidate IDs, model/runtime revision, language and + truncation metadata; output labels, raw probability vector, provider-specific + confidence, optional separately versioned calibration and an abstain reason. + Validate schema, known IDs, finite/ranged scores and distributions before + use. Invalid, unsupported, truncated, late or uncertain output falls back + to the unchanged deterministic behavior and is recorded. +4. **Freeze and replay:** save observations with the run/experiment. Cache keys + cover scope/evidence, candidate and catalog hashes, question/options/rubric, + schema, model/runtime, calibration and language. Changed inputs invalidate + reuse; resume reuses the saved observation instead of paying again. Keep + raw private content outside tracked evidence. Predictions are untrusted + data and cannot supply shell commands, policy text or new permissions. + An interrupted external call with an unknown outcome is not proof of no + charge: record it as indeterminate and abstain rather than automatically + resubmitting on resume. Any explicit retry needs its own budget reservation. +5. **Bound the extra work:** explicitly budget calls, input size, wall time, + local memory and any authorized API cost. Account for classifier calls, + worker launches and native usage separately. Batch related questions only + within budget; use bounded retries and cancellation. A timeout must not + outlive its run or consume the worker's entire remaining deadline. +6. **Keep installation optional:** no heavy imports in core CLI or hooks; no + hook network calls, inference or model downloads. Use an isolated optional + environment and pinned assets. An owned bounded process can reuse a loaded + model within a run; no permanent service/extra MCP server is required. Model + downloads are explicit setup, never an automatic fallback. Missing extras, + unsupported hardware or a failed helper must preserve ordinary operation. + +Jev experiments additionally require an approved endpoint/model, allowed data +classes, redaction policy, spending ceiling and explicit retry policy. Its API +supports typed responses and reports usage; record actual returned usage, +errors and concrete model identity rather than inferring usage from a request. +[Official API](https://docs.typesafe.ai/api). + +### Immediate capped Jev pilot + +The v1 pilot batches eight synthetic task cases and 24 task-family, +execution-tier and specialist-skill choices into **one** Jev 1.13 request. The +dry run is 19,219 request bytes. At the frozen published price, even the model's +documented 64k aggregate context limit would cost about $0.002688, below the +$0.01 ceiling. This calculation is only a preflight cap; the receipt must use +the API's returned `input_tokens`. The probe has no retry path, never accepts a +key on the command line and emits only case IDs, choices, probability vectors, +latency and usage—not task text. + +```bash +python3 test/core/probes/jev_decision_eval.py +# Fill TYPESAFE_API_KEY in the Git-ignored local .env using your editor. +# For a fresh checkout, copy the blank .env.example to .env first. +chmod 600 .env +# This validates local configuration without calling the API or printing the key: +python3 test/core/probes/jev_decision_eval.py --env-file .env +# Live execution is a separate, explicitly selected step after pricing review: +python3 test/core/probes/jev_decision_eval.py --env-file .env --execute \ + --output "$HOME/.devsquad/private-probes/jev-pilot-v1.json" +``` + +The probe supports a single-line plain or quoted `TYPESAFE_API_KEY` assignment +and optional `export`; it ignores unrelated variables, rejects duplicate key +assignments, and never evaluates shell substitutions. The env file must be a +private regular UTF-8 file, at most 16 KiB, with no group/other permissions. +An already exported non-empty key takes precedence. Files are loaded only +when `--env-file` is supplied; this does not enable the runtime classifier or +pass the key to coding workers. A dry run reports only whether a key is present, +not whether TypeSafe has authenticated it. The local `.env` and `.env.*` files +are Git-ignored; `.env.example` contains no secret. + +Do not copy the key or private receipt into Git. If the provider's current price +makes one maximum-context request exceed $0.01, if returned usage breaches the +cap, or if access requires purchasing a larger commitment, stop without retry +and start the pinned Laya local trial. A smoke result only answers whether Jev +can follow this schema on synthetic cases; it cannot enable runtime routing. + +## Spec-based iterations and acceptance + +These are small work packages within M6's existing experiments/learning work, +not new prerequisite milestones. M4's external live-host blocker stays open; +independent offline evaluation can proceed once the M5 evidence shape is stable. +M5 disposition, bounded revisions and receipts remain the immediate next work. + +| Package | Deliverable | Required proof before advancing | +|---|---|---| +| **M6-D1 — Contract and baseline** | Strict optional decision schema, fake adapter, run/cache accounting, redacted labeled corpus and frozen experiment spec | Default-off equivalence; malformed/unknown/NaN output, candidate/pin/quality/permission attacks, input drift, cancellation and crash/resume tests. Zero unauthorized selections or duplicated paid calls on replay | +| **M6-D2 — Capped Jev pilot** | One-request synthetic smoke, then a larger shadow comparison only if separately budgeted and justified | Exact model/usage/latency/cost receipt; strict response validation; no retry, task disclosure or runtime effect. Missing access, cost breach or poor results select the Laya fallback rather than weakening the gate | +| **M6-D3 — Local fallback and adoption decision** | Pinned optional Laya trial when triggered, followed by evidence-backed keep-off or narrowly scoped advisory policy and rollback receipt | Same cases and end-to-end accounting for any Jev/Laya comparison; use-case quality/cost gate before opt-in. Failed/inconclusive experiments remain off; model/rubric/calibration changes rerun held-out and boundary tests | + +Freeze the dataset split, metrics, thresholds, sample-size rationale and resource +ceilings **before** evaluating the held-out set. Split by issue/repository family +to prevent near-duplicate leakage. Include real engineering tasks, no-match +cases, long/noisy inputs, English/Hinglish/other supported languages, adversarial +instructions and quota-constrained candidate sets. Synthetic fixtures test +mechanics; they do not establish user-facing quality. + +Measure task-family/shortlist accuracy, abstention coverage and calibration, +context evidence recall, inappropriate downgrade rate, actual check/review +outcomes, escaped defects, lead repairs, total latency, peak memory and observed +worker/token usage. The cheapest sufficient profile is established from verified +outcomes, not a model-brand label or another classifier's answer. Report sample +sizes, uncertainty intervals and unavailable usage; do not convert worker counts +or tokens into fixed Plus-window consumption. + +Adoption requires all authority/integrity tests passing, no observed critical +evidence loss or constraint violation, quality meeting the predeclared +non-inferiority margin, and a material end-to-end benefit after helper overhead. +Use a predeclared target (initial proposal: at least 10% less observed scarce +worker usage **or** lead rework time) with adequate uncertainty bounds; a tiny +pilot cannot prove the production gate. Scarce usage must be observed, not an +invented quota conversion. Limited or inconclusive data means keep shadow/off. +Jev and Laya need the same inputs/options and end-to-end accounting to support +any head-to-head claim. + +Expand to P2/P3 only through separate small specs and measurements. A successful +skill selector does not establish a safe model router, failure handler or judge. diff --git a/docs/plans/engineering-team/IMPLEMENTATION.md b/docs/plans/engineering-team/IMPLEMENTATION.md new file mode 100644 index 0000000..e84cde9 --- /dev/null +++ b/docs/plans/engineering-team/IMPLEMENTATION.md @@ -0,0 +1,156 @@ +# Implementation work packages + +**All milestones are pending.** This is the execution sequence for [ADR-002](../../adr/ADR-002-surface-independent-engineering-team.md) and [contracts v1](CONTRACTS.md). The existing 177 offline assertions passed during architecture preparation; they validate legacy behavior, not the proposed runtime. + +## Sequence and stopping points + +```mermaid +flowchart LR + M1[M1: Invocation contract] --> M2[M2: Durable runner] + M2 --> M3["M3: Branch review
First useful product"] + M3 --> M4[M4: Local app access] + M3 --> M5[M5: Code delivery] + M4 --> M6[M6: Capacity and learning] + M5 --> M6 + M6 --> M7[M7: Install and prove] +``` + +Implement sequentially through M3. M4 and M5 may proceed in parallel after agreeing on the frozen M3 public service API; assign separate files. M6 integrates their evidence. Each milestone can contain small reviewable commits. A milestone is complete only when its required gate has evidence; an unavailable provider or app leaves the relevant live gate open. + +## M1 — Make invocation truthful and reusable + +**Outcome:** A versioned profile selects a verified model/effort/permission combination, and the core can prepare/classify a call without owning a second watchdog. + +**Existing files:** `plugin/lib/adapter.sh`, `codex-wrapper.sh`, `gemini-wrapper.sh`, `grok-wrapper.sh`, `model-catalog.sh`; `test/test_wrapper_contract.sh`, `test/test_models.sh`. + +**New files:** `plugin/core/pyproject.toml`, `bin/squad`, `src/devsquad/{cli,contracts,adapters,codex_protocol}.py`, `schemas/`, `adapters/{codex,antigravity,grok}/`; `test/core/test_contracts.py`, `test/core/test_adapters.py` and fake executables/protocol servers. + +Work in this order: + +1. Establish package/import layout, Python 3.11 floor, `squad --version`, initial `doctor`, strict v1 schemas and fixture loading. Record package/source fingerprints. +2. Extract reusable argv-building and classification helpers without changing sourced-wrapper signatures or error prefixes. Implement `prepare`/`classify` for CLI transport; add the version-verified Codex stdio app-server transport and event normalization. Preserve legacy telemetry only on legacy invocations. Use upstream native-protocol patterns from the model lifecycle amendment without importing a second job coordinator. +3. Fix legacy portable-watchdog completion delay and descendant cleanup with timing/process assertions. Preserve Bash 3.2 and jq-absent legacy tests. Do not make legacy operation require installing Python. +4. Add per-call explicit model/effort/tool/permission settings; verify installed CLI mappings with help/documentation and bounded probes. Discover Codex models/efforts through app-server metadata; define stable alias/templates separately from concrete profiles. Remove cross-family numeric tier ranking from new selection; correct legacy resolution with compatibility tests. Paginated catalog refresh retains last-good data on error/incomplete responses. Catalog drift produces unknown/revalidation rather than silent identity substitution. +5. Replace the Gemini extension whitelist and whitespace splitting with scoped, bounded context enumeration. Document when native file access replaces prompt concatenation. + +**Acceptance gate:** Existing suite passes. Fake immediate CLI returns promptly with a 2-second timeout (target under 1 second on normal local CI); a hanging CLI and descendant are gone by timeout plus grace. Fixtures cover empty exit-0, exit-0 auth banner, denied tool, malformed output, spaces/TSX inputs, explicit unsupported effort, unavailable capability and cross-family catalog entries. Overrides do not mutate global/project config. Doctor distinguishes supported, unverified and unavailable settings. One short read-only real adapter probe validates the chosen starting profile; save a redacted receipt and version facts. + +**Boundary:** No learned routing, MCP, worktree edits or universal model catalog. Manifests for unprobed harnesses remain visibly unverified. + +**Native/discovery gate:** A fake app-server exercises thread/review/turn events, explicit supported/unsupported effort and native IDs. Acknowledged start/interrupt is not reported as terminal completion. Paginated metadata assembles one complete snapshot; parse/auth/timeout failures preserve the previous snapshot. An added model becomes an unqualified candidate, and an unknown family cannot inherit capabilities from its name. Native model-list schema availability alone does not count as live entitlement or quality evidence. + +## M2 — Persist jobs and own their processes + +**Outcome:** Start/status/cancel/resume operate on the same saved job, including after the initiating shell exits. + +**New files:** `src/devsquad/{store,supervisor}.py`, database migrations, artifact storage helpers; `test/core/test_store.py`, `test/core/test_supervisor.py`. + +1. Implement SQLite ledger/projections, request hashing, schema migration and atomic artifacts under the machine-local runtime directory. Canonical Git common directory identifies a project; project state and shared account pools remain distinct. +2. Add start/status/events/result/cancel/resume service functions and CLI handlers. Persist the task before spawn; recover a crash between enqueue and spawn. Use one detached supervisor per run and fixed package snapshot. +3. Implement process-group ownership, bounded output capture, timeout/cancel/reaping, supervisor claims and a single active writer constraint. Native output goes to artifacts, not an unbounded in-memory buffer. +4. Implement recovery classification and host handoff storage/claim fencing. The public API needs these invariants before apps consume it. Expired heartbeat must not trigger blind relaunch. + +**Acceptance gate:** Two concurrent identical starts return one run; same key/different body conflicts. Event/projection mutations remain consistent under concurrent writers. Kill the supervisor while its child stays alive: resume launches no second writer. Reused/ambiguous PID is not signalled. Kill between artifact write and DB commit: no false completion. Repeated cancel is harmless; cancellation waits for confirmed child cleanup. Close the launching shell and inspect the job from another shell. Apply/test a database migration against a fixture of the preceding schema; refuse a newer unsupported schema. All tests use temporary runtime directories. + +**Boundary:** A fake-step harness exercises lifecycle; no claim yet that an engineering workflow works. No HTTP listener or permanent daemon. + +## M3 — Ship a useful branch review + +**Outcome:** `squad start` reviews a real committed diff and returns findings, check results and a receipt without modifying the user's checkout. + +**New files:** `src/devsquad/{workflows,router,reports}.py`, `policies/`, role templates, branch-review schema/fixtures; `test/core/test_review_workflow.py`. + +1. Implement profile/policy loading, alias binding resolution, version/hash snapshots and capability/permission/quality filters. Select automatically from a small static candidate list, with validated per-role profile overrides and explicit fallback semantics from the selection amendment. Freeze concrete profiles and fallback sets during preflight; never resolve a changed alias halfway through a run. Add basic shared-pool concurrency and typed unknown capacity now; richer observations arrive in M6. +2. Resolve commit refs and task scope; create a frozen review workspace. Reject intersecting dirty inputs rather than silently omitting them. +3. Implement reviewer → trusted checks → lead disposition, with `required_to_pass` semantics. Support host handoffs through CLI and a headless lead profile. Expose clear `awaiting_host` packets. Distinguish ordinary native review from a steerable adversarial-review prompt, recording the mode and supported controls; both remain review-only against the frozen candidate. +4. Produce receipt JSON/Markdown, events export and artifact manifest for every terminal outcome. Record unknown host usage honestly. + +**Acceptance gate:** An actual branch review returns actionable findings or a supported clean verdict, against recorded base/target hashes. A report-only failed check is included in a successful review; a required failed check prevents acceptance. A denied reviewer write or missing output cannot count as a valid review. A moving branch does not change the frozen run input. A second terminal claims a saved host handoff; a late completion from the first is fenced out. Verify original checkout/index/HEAD are unchanged. + +**First product stop:** At this point DevSquad is usable from terminal for a bounded job. Demonstrate it before adding more architecture. + +**Selection gate:** Identical task/policy/availability snapshots produce the same automatic selection. Pin one role and verify others remain automatic. Unsupported pins fail; an unavailable pin with `fallback:none` blocks without substitution. `fallback:policy` selects only a qualified alternative. No override broadens permissions or billing authority. Receipt explains effective model, effort and toolbox; a worker can only call permitted tools. Fake fixtures exercise an explicit bounded escalation and preserve each attempt's original profile. + +**Accounting gate:** `max_worker_invocations` limits adapter launches, including retries and the headless lead. Native internal model/tool turns and observed token/quota usage are separate fields; unknown stays null. A simulated worker with several native turns must not be reported as one measured model request or a fixed allowance charge. + +## M4 — Use that same run from local apps + +**Outcome:** Apps are interchangeable clients of the saved run. + +**New files:** `src/devsquad/mcp_server.py`, `integrations/{codex,claude-code,antigravity,grok}/`, optional MCP dependency/lock, generated tool-reference docs; `test/core/test_mcp.py`. + +1. Pin a currently supported official MCP Python SDK version and test it with the chosen Python floor and local runtime. Keep core CLI imports free of optional MCP dependencies. +2. Map the contract's operations directly to service functions. Paginate events, cap artifact previews and return IDs/paths/hashes. Start/cancel return promptly; app tool timeout is never the worker lifetime. +3. Add minimal host instructions: construct task/criteria, submit once with idempotency key, inspect status, claim a handoff, consume receipts. No host-specific router or copied provider matrix. +4. Create explicit setup templates for the documented local stdio integration. Detect existing/inherited registrations and stable executable resolution; preserve unrelated settings. Doctor reports what each installed app actually loads. +5. Enforce the worker recursion guard at MCP service and legacy hook boundaries. Worker sessions with inherited global MCP configuration must not create nested teams. + +**Acceptance gate:** MCP schema/conformance tests cover malformed requests and short tool timeouts. Start in terminal, inspect from Codex, complete a host handoff from Claude Code: the run ID, input hashes and ledger are identical. Close the MCP client and confirm the worker survives. Two hosts cannot both advance a handoff; stale host output is retained only as audit evidence. Prove one inherited duplicate server is detected and worker-origin mutation is rejected. Save installed-version receipts; config syntax alone is insufficient. + +**Boundary:** Antigravity and Grok configs can be prepared here; M7 requires their real local smoke receipts. No promise of hosted/cloud access. + +## M5 — Deliver a bounded engineering change + +**Outcome:** One model implements, another model reviews, checks verify the exact candidate, and the lead resolves the result. + +**Existing files:** Adapter manifests/bridge, workflow service, role prompts; new `plugin/lib/claude-wrapper.sh` and corresponding manifest if Claude headless is not yet available. Extend `test/core/test_adapters.py` and add `test/core/test_delivery_workflow.py`. + +1. Add the Claude headless adapter using the same conformance contract and verified permissions. Do not confuse the Claude Code host with a separately spawned Claude worker. +2. Implement isolated delivery worktree, one-writer enforcement, scope validation and local candidate commits. Return patch/commit artifacts without automatic merge/push. +3. Select a reviewer with a different verified model identity. Prefer a different harness when a qualified alternative exists; same-family or unknown identity must not masquerade as independent model review. +4. Run checks in a separate candidate worktree. Bind review, checks and disposition to candidate hash. Implement bounded revisions/fallbacks/deadlines and dependent-step failure handling. +5. Preserve all attempts and repairs in the result. Do not allow a failed mandatory check or unsupported reviewer restriction to be overridden by lead prose. + +**Acceptance gate:** A real bounded issue completes implementation → different-model review → correction if needed → checks → receipt using at least two harnesses. A fake reviewer catches a seeded defect and the corrected candidate reruns checks/review. Change the patch after a valid review: stale evidence cannot complete the run. Simulate rate-limit fallback without widening permissions. Kill a live implementation supervisor and prove no duplicate writer on resume. Verify no original-checkout changes or remote publication. + +**Boundary:** No arbitrary DAG, broad autonomous project implementation or automatic integration service. + +## M6 — Make capacity and improvement evidence useful + +**Outcome:** Scheduling respects shared limits, and completed work produces actionable, traceable policy proposals. + +**New files:** `src/devsquad/{capacity,learning}.py`, observation/experiment/outcome schemas, learning templates; `test/core/test_capacity.py`, `test/core/test_learning.py`. + +1. Implement pool mapping, applicable quota windows, TTL/source/confidence and local reservations. Use documented provider observations, including Codex app-server rate limits when supported, and timestamped manual values otherwise. Status displays unknown and stale values explicitly. +2. Add bounded fallback, native retry metadata and explicit paid-API eligibility. Snapshot every routing decision with exclusions and policy version. +3. Add final/late outcome records, role contribution, lead repair and evidence references; produce comparison reports with sample sizes and missingness. Separate manually pinned decisions, normal automatic routing and experimental assignments to avoid treating selection bias as a profile improvement. +4. Implement experiment specs and held-out evaluation fixtures. `learn propose` generates a draft hypothesis/evaluation/decision packet; promotion remains a reviewed Git policy change with a rollback target. +5. Generate receipts/handoffs on run transitions, and curated documentation on explicit report/proposal operations. Record drift and affected evidence; do not introduce an unrequested scheduled automation. + +6. Implement the [model lifecycle](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md): budgeted qualification, limited trials, reviewed or explicitly enabled guarded automatic binding promotion, compare-and-swap binding versions, last-qualified fallback and rollback. Promotions stay within allowed templates, quality criteria, permissions and billing authority. They write local decision receipts and affect new runs only. The default remains reviewed until evaluation gates and policy enable guarded automation. + +7. Execute the optional [typed decision-classifier evaluation](DECISION-CLASSIFIERS.md), packages M6-D1–D3, within the existing experiment budget. Begin with strict fake-adapter contracts and the user-authorized one-request, $0.01 Jev synthetic pilot for task/profile hints and skill shortlists. Use no private task content and no retry. Move to a pinned local Laya shadow trial if hosted access/cost or measured quality warrants its setup; consider context ranking next. Compare with the existing no-model router and lead on held-out engineering outcomes, overhead and rework. Keep-off/inconclusive is a valid adoption decision, not a reason to relax the gate. + +**Acceptance gate:** Two projects sharing one pool obey a fresh exhausted weekly window despite available short-window capacity. Stale/unknown values never become zero; external usage changes do not get assigned to one worker. Concurrency reservations release only after ownership is reconciled. A paid API fallback is excluded unless allowed. A failed original attempt later repaired by another model produces final success without crediting the original as independently successful. A late escaped bug updates outcome history. A one-variable fixture experiment produces a traceable no-change or promotion proposal, with all failures and a rollback version; insufficient evidence leaves active policy unchanged. Re-run a held-out fixture after policy change and exercise rollback. + +**Boundary:** No autonomous learned-policy changes or model training. Experiment budgets and decision helpers default off; activate only through explicit versioned policy. The September 26 [classifier amendment](DECISION-CLASSIFIERS.md) permits frozen semantic suggestions and, after its use-case gate, reviewed opt-in ranking within an already-qualified set. The deterministic router, user pins, quality floors, capacity/permission/billing checks and lead acceptance remain authoritative. No classifier is required for normal operation. + +**Release gate:** A new model cannot become default from discovery alone. Insufficient evidence, unsupported effort or widened permissions/billing prevents automatic promotion. Changed metadata revalidates only affected profiles; same-ID backing changes retain unknown revision when unobservable. An approved binding update affects a newly started run while an existing run and concrete override remain pinned. Concurrent promotions conflict on stale binding versions. A regression reverts to an available qualified binding and records why. All qualification/trial launches share the configured experiment and account-pool budgets. + +## M7 — Package, migrate and prove every requested surface + +**Outcome:** A fresh install has one known runtime, reliable documentation and real smoke evidence. + +**Files:** New `scripts/install-core.sh`, packaging/release checks and install tests; update `install.sh`, `CONTRIBUTING.md`, `docs/ARCHITECTURE.md`, appropriate `plugin/commands/`/skills/hooks, release metadata and changelog when releasing. + +1. Package the core inside `plugin/`, with a stable standalone launcher and optional MCP environment. Preserve active run release pins. Installer is idempotent and reports source/plugin/standalone drift. +2. Keep legacy plugin mode available during migration. Reconcile hook registration only when duplicate evidence exists; September review found no current duplicate. New hooks call the shared route source after its tests pass and remain fast/network-free. No Python dependency imposed on legacy mode. +3. Create generated command/schema examples and concise install/operate/recover guides. Mark implemented vs deferred features; link completion claims to receipts. Keep historical ADR/audit statements dated. +4. Add offline CI for legacy and core suites (Bash 3.2/macOS compatibility and chosen Python floor/current version), package-content checks, native protocol compatibility fixtures and optional MCP tests. Live provider/app tests remain explicit bounded smoke runs. Document supported protocol ranges, capability drift and the optional native Claude→Codex session-import path, while retaining portable artifact handoffs for every host. +5. Verify terminal CLI, Codex App, Claude Code App local Code tab, Antigravity CLI (`agy`) and Grok Build against the same saved runtime. The user's October 2 clarification excludes Antigravity IDE trust/UI from this acceptance. Check native capabilities/profile identity as used, not by brand inference. + +6. If the classifier trial is retained, package it as an isolated optional extra with pinned model assets and explicit setup. Test absent extras, offline operation, unsupported hardware, cancellation and default-off equivalence. Hooks remain inference/download-free and network-free; neither a hosted key nor heavy model dependencies become a core-install requirement. + +**Acceptance gate:** Fresh standalone install works without Claude installed; existing Claude plugin install contains `plugin/core` contents correctly. Reinstall creates no duplicate hook/server registration and a release update does not break a running job. Record actual start/observe/handoff-or-cancel receipts from every listed surface. Complete one end-to-end delivery with a different-model reviewer after installation. Documentation commands run as written. If an installed host cannot support an operation, retain that item as blocked with exact evidence instead of declaring universal support. + +## Optional C1 — Selective Council decisions + +After M6, implement the separately gated [Council extension](SELECTION-AND-COUNCIL.md). Reuse the runner and native adapters for two independent proposals, a distinct critic and the existing lead. Preserve dissent, enforce evidence checks and account for every call. C1 does not block M7; its own acceptance/evaluation gate controls whether automatic council triggering is enabled. Track it under `extensions` in the backlog, preserving the seven core milestones. + +## Evidence and ongoing status + +Update [backlog.json](backlog.json) as work proceeds. Allowed states: `pending`, `in_progress`, `blocked`, `complete`. A blocked item needs a precise reason and remaining independent work; completion requires non-empty evidence. Each evidence item records `kind`, `revision`, `command_or_action`, `outcome`, `artifact`, `recorded_at` and `availability`. Use an implementation revision already committed when writing a subsequent completion receipt; do not insert a circular “this commit's hash” placeholder. + +Store portable redacted milestone receipts under `docs/plans/engineering-team/evidence/`; reference private local run IDs separately. Never commit provider secrets or unreviewed raw task logs. Record measurements honestly: fake tests prove mechanics; live runs prove integration; neither alone proves a profile is generally better. + +Before each commit, run `bash test/run.sh`, relevant new tests, schema/example validation and `git diff --check`. Broaden testing only for changed behavior or unresolved concerns. Before ending, checkpoint the verified work and leave a clean tree under the repository's git-safety instructions. diff --git a/docs/plans/engineering-team/M1-STATUS.md b/docs/plans/engineering-team/M1-STATUS.md new file mode 100644 index 0000000..2be580f --- /dev/null +++ b/docs/plans/engineering-team/M1-STATUS.md @@ -0,0 +1,37 @@ +# M1 implementation status + +M1 is **complete** for its bounded invocation and preparation scope at +`97a10f0`. Independent review accepted it after the 34-test core suite, the +202-assertion shell suite, and inspection of the saved native receipt. Probe +cleanup hardening is preserved at `7b5c41c`. M2 process ownership has not +started. + +| # | Requirement | Evidence | Status | +|---|---|---|---| +| 1 | Native framing, typed requests, handshake, lifecycle and correlation | Bounded buffered JSON-line peer; `initialize` then `initialized`; installed 0.135.0 request shapes; conforming fake server with interleaved notifications | verified offline | +| 2 | Truthful provider completion and malformed-frame handling | Codex requires correlated terminal `turn/completed`; provider-specific document/JSONL parsing; partial, startup-only and malformed fixtures | verified offline | +| 3 | Strict Task, Profile, Policy, LaunchSpec, identity and result inputs | Runtime validators, matching strict schemas, adversarial nested-type tests and independent reviewer regressions | verified offline | +| 4 | Model/version-scoped capability preparation | Complete paginated discovery feeds native preparation; explicit unknown model, effort and drift cases fail before launch | verified offline | +| 5 | Shared Bash/Python classification taxonomy | Packaged `classification-policy.conf` is consumed by both paths; auth-topic, ordering and empty-success regressions | verified offline | +| 6 | Portable watchdog ownership and cleanup | Forced portable Bash 3.2 tests cover fast return, descendant cleanup, and a ready TERM-ignoring root with an absolute fake executable | verified offline | +| 7 | Complete native pagination and last-good behavior | Empty pages, repeated cursors, malformed/disconnected pages and incomplete-refresh retention are exercised | verified offline | +| 8 | Requested and observed identity remain separate | `ExecutionIdentity`, `LaunchSpec` and `NormalizedResult` validation and round trips | verified offline | +| 9 | Execution, artifact and acceptance states remain separate | Normalized result contract and classifier tests | verified offline | +| 10 | Tracked, bounded Antigravity context | Literal Git inventory; ignored/binary/secret/oversize omissions; file and ancestor-symlink containment tests | verified offline | +| 11 | Package and CLI envelope | Temporary wheel installation resolves schemas, adapters and shared taxonomy; input/readiness exit behavior is tested | verified offline | +| 12 | Bounded real starting-profile smoke | Codex CLI `gpt-5.5`, low effort, read-only, ephemeral JSONL invocation completed on 2026-09-06 | verified live (CLI) | +| 13 | Integrated native preparation/protocol/classification probe | Saved probe at `97a10f0`: discovery → catalog → preparation → read-only gpt-5.5/low turn → correlated completion; private receipt and stream hashes retained | verified live | +| 14 | Independent gate review | Eight reviewer regressions pass; reviewer independently reran 34 core tests, inspected all private artifact hashes and verified requested gpt-5.5/low plus readOnly/networkAccess:false and correlated completion | verified | + +The Python bridge returns launch/protocol preparation and normalized parser +policy only. It does not spawn a worker or implement a second watchdog. M2 +remains the sole owner of process sessions, timeouts, cancellation, draining +and reaping for new runs. + +Declared limitations at this checkpoint: + +- Grok workspace-write preparation is unsupported; its verified profile is + read-only. +- Antigravity and Grok inference were not used for the live M1 smoke. +- Policy learning, routing semantics and session supervision remain deferred + to their specified later milestones. diff --git a/docs/plans/engineering-team/M2-STATUS.md b/docs/plans/engineering-team/M2-STATUS.md new file mode 100644 index 0000000..0296094 --- /dev/null +++ b/docs/plans/engineering-team/M2-STATUS.md @@ -0,0 +1,56 @@ +# M2 implementation status + +M2 is **complete** at implementation checkpoint `ddb6f51`. The final gate +passes 121 core tests with `ResourceWarning` promoted to an error and all 10 +Bash regression files/202 assertions. No provider invocation was used for the +M2 gate. + +| Requirement | Evidence | Status | +|---|---|---| +| SQLite ledger, WAL and packaged migrations | Concurrent fresh-database initialization; source fixtures migrate from schemas 1, 3 and 4; a wheel installed outside the checkout applies all packaged migrations through schema 5 and rejects a future schema | verified offline | +| Canonical start idempotency and project identity | Independent processes return one run for identical requests and conflict for a changed body; linked Git worktrees share the canonical common-directory project ID | verified offline | +| Preparing-owner recovery | Interrupted preflight replays the immutable submitted request under a rotated fence; competing reclaimers yield one owner; stale owners and repository retargeting are rejected | verified offline | +| Transactional state, events and artifacts | Projection/version compare-and-swap and events share a transaction; artifacts are atomically finalized and hash checked before reference | verified offline | +| Supervisor and writer fencing | Database claims permit one active worktree writer; strong process identity distinguishes live, dead and ambiguous/reused processes | verified offline | +| Durable gated launch | The persisted runner cannot open the inner worker gate until runner identity and spool paths are committed; a crash after identity but before gate release requeues without executing the command | verified offline | +| Coordinator-loss recovery | A launching shell can exit while another process observes/imports the run; a live runner is not relaunched; completed receipts import exactly once across processes | verified offline | +| Orphan-child recovery | A dead runner with a live recorded child blocks relaunch; owned recovery cancellation signals only the verified child group and confirms absence before terminal state | verified offline | +| Interrupted recovery cancellation | A spawned canceller exits immediately after persisting `recovery_cleanup`; the next public cancel resumes cleanup, emits one terminal event/receipt and repeated cancel is harmless | verified offline | +| Timeout, cancellation and stdin | Exact regular-file stdin bytes reach the child; a timeout stays failed even when TERM exits zero; cancellation intent precedes bounded TERM/KILL and terminal state follows confirmed cleanup | verified offline | +| Cross-process races | Independent-process tests cover identical/conflicting starts, worktree writer claims, queued/running cancel, receipt import, host claims and handoff completion replay | verified offline | +| Host handoffs | Schema-5 packets, bounded claims, renewal/takeover, submission replay, late audit, awaiting-host cancellation and CLI/service operations are fenced and durable | verified offline | +| Lease authorization under contention | Real independent SQLite writer locks force renewal and completion to wait across expiry; authoritative time is sampled only after the write transaction is acquired | verified offline | +| CLI and durable reads | Exact v1 envelopes/exit mappings, `start --wait`, observation-only Ctrl-C, transactional status, stable cursors and hash-verified terminal results | verified offline | +| Public workflow boundary | Public branch review fails explicitly with `CAPABILITY_UNAVAILABLE`; only the private fake-step hook exercises M2 lifecycle | verified offline | + +## Acceptance mapping + +- Concurrent identical starts converge on one run; a different request under + the same key conflicts. +- Event/projection mutations and terminal artifact references remain + transactional under independent writers. +- Killing a coordinator or runner never creates a second writer. Ambiguous or + reused process identities are not signalled. +- A crash after artifact finalization but before its database transaction + creates no false completion; resume imports the same finalized bytes once. +- Cancellation survives caller loss and never records terminal cancellation + before the verified child group is absent. +- Schema migration is exercised from the preceding schema through an installed + wheel, and newer unsupported schemas are refused. +- A second process can inspect and operate on the same saved run and can claim + a host handoff; stale, expired and lock-delayed completions cannot advance it. + +## Independent review closure + +The final bounded review found two P1 races and no additional defect in its +targeted scope: lease time was sampled before SQLite lock acquisition, and an +interruption after orphan-cancel intent could leave public cancel stuck. Both +were fixed at `ddb6f51` and have deterministic contention/process-crash +regressions in the 121-test gate. + +## Boundary + +M2 proves the durable execution substrate, not a working engineering workflow. +`branch-review` intentionally remains unavailable until M3 implements frozen +review workspaces, deterministic profile routing, reviewer/check/lead flow and +bound reports. M3 is the first user-usable product checkpoint. diff --git a/docs/plans/engineering-team/M3-STATUS.md b/docs/plans/engineering-team/M3-STATUS.md new file mode 100644 index 0000000..7bac96d --- /dev/null +++ b/docs/plans/engineering-team/M3-STATUS.md @@ -0,0 +1,60 @@ +# M3 implementation status + +M3's reopened **candidate-integrity repair R1 is verified in source**. Public +review/delivery regressions now block source-changing and undeclared-input +checks through host/headless disposition, retain evidence and skip dependent +checks. The final core gate ran 330 tests successfully (two optional SDK skips). +See [R1 evidence](evidence/R1-candidate-integrity-2026-09-29.json). Installed +refresh/revalidation remains R8 work; the old installed payload is unchanged. + +M3 was accepted at implementation checkpoint `1737667`. That historical gate +passes 188 core tests with `ResourceWarning` promoted to an error and all 10 +Bash regression files/202 assertions. The bounded live Codex review evidence +remains the successful subscription-backed run recorded at `9478796`; no +additional provider turn was used for the closeout. + +| Requirement | Evidence | Status | +|---|---|---| +| Deterministic selection | Versioned profile aliases, exact pins, explicit `none`/`policy` fallback, permission/billing filters and typed capacity produce stable frozen routing snapshots | verified offline | +| Frozen branch input | Base/target refs resolve to exact OIDs; committed config hashes, candidate hash and detached review/check workspaces remain stable while the submitted checkout, index and HEAD are preserved | verified offline | +| Reviewer evidence | Strict ordinary/adversarial prompts, review/check/evaluation schemas, candidate binding, read-only identity checks and malformed/denied/disconnected output faults prevent unsupported success | verified offline | +| Check and lead gates | Source/HEAD/index/mode/undeclared-input mutation invalidates evidence even for report-only checks; host/headless refuse acceptance and later checks are skipped | R1 verified offline; installed refresh pending R8 | +| Durable host handoff | Waiting JSON/Markdown packets, claim leases, renewal/takeover, stale completion fencing, replay and crash-resume continuation share the saved run ledger | verified offline | +| Terminal reporting | Success, rejection, preflight failure, worker failure, cancellation, waiting cancellation, timeout and budget exhaustion publish receipt JSON/Markdown, events, manifest and result receipt | verified offline | +| Headless leadership | Offline and native headless leads run as separate fenced attempts, verify their own frozen identity/evidence and terminalize without host intervention | verified offline | +| Runtime fallback | Reviewer and lead failures advance only through the frozen qualified fallback order; `fallback:none`, invocation/wall budgets and recovery-before-launch are enforced without profile substitution | verified offline | +| Capacity and accounting | Reservations are transactional, unknown pools permit one unresolved trial, live shared-pool reservations are observed, adapter launches are distinct from native usage and unavailable counts remain null | verified offline | +| Live branch review | Bundled Codex 0.153.4, gpt-5.5/low, ephemeral read-only execution found one supported regression, passed the required check and produced all five terminal report hashes | verified live | + +## Acceptance mapping + +- The live public branch review is bound to recorded base, target and candidate + hashes and produced an actionable supported finding. +- Offline gates separately prove a supported clean verdict, report-only and + required-check behavior, moving-ref stability, denied/missing output faults, + two-host fencing and source-checkout preservation. +- Receipts retain the effective profile/model/effort/toolbox snapshot, every + executed fallback profile, earlier review revisions and lead dispositions. +- `max_worker_invocations` counts durable runner launches, not reservations + abandoned before their launch fence. Revision, fallback and wall budgets all + end in explicit terminal reports rather than stranded resumable states. +- Standard and adversarial review prompts are distinct while sharing the same + read-only evidence and candidate-binding rules. + +## Independent audit closure + +The bounded M3 audit at `84deb77` found four defects: a recovered pre-launch +reservation consumed a fallback slot, headless-lead budget exhaustion could +strand a run, unknown capacity allowed more than one unresolved trial, and a +later failed/cancelled revision omitted earlier attempts and dispositions. +`1737667` fixes all four and adds direct reproductions. The complete 188-test +gate and 202 Bash assertions pass after those fixes. The requested follow-up +agent re-run could not start because its shared Plus window was exhausted; the +finding-specific regressions and full local gates are the closure evidence. + +## Boundary + +M3 proves a useful saved branch review from terminal/service APIs. It does not +claim M4 local-app MCP access, M5 issue implementation, M6 lifecycle learning, +M7 installation/all-surface receipts or C1 Council. Those remain required by +the full assignment. diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md new file mode 100644 index 0000000..e0ab10d --- /dev/null +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -0,0 +1,153 @@ +# M5 implementation status + +M5 is **in progress with independent repairs available**. The September 29 +review at `f4fa657` reopened candidate integrity (F1), observed Claude identity +(F2) and complete normal-command check coverage (G4). F1/R1 is now repaired +and verified offline; R2 is also source/offline verified at `ef98889`. Execute R6 in +[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The genuine Claude-to-Codex +live gate additionally remains blocked on normal Claude login. + +R2 now enforces native reported model identity, strict imported evidence bound +to the actual attempt, independent-review acceptance and durable failure/ +fallback/cancellation/legacy replay. Its 37 added regressions are included in +the final 367-test suite (OK, two optional-SDK skips); 227 Bash assertions and +the generated-reference check passed. The SQLite cleanup warning remains R5. +See [R2 evidence](evidence/R2-observed-identity-2026-10-01.json); the +[red baseline](evidence/R2-identity-red-baseline-2026-09-29.json) is retained as +history. This is not a corrected installed Claude live receipt or full M5 closure. + +| Requirement | Planned evidence | Status | +|---|---|---| +| Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | argv conformance verified at `d96e9e4`; R2 observed-model evidence verified offline; unreported effort remains unknown | +| Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | verified offline at `0e88d73` | +| Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | verified offline at `0e88d73` | +| Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | R2 public host/headless identity gates verified offline; installed/live proof pending | +| Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | R1 check-mutation repair verified offline; R6 normal-command check coverage remains | +| Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | correction/budgets verified at `4c76887`; same-permission delivery fallback verified at `b7d90cc` | +| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | R1 candidate integrity and R2 actual-model identity enforced offline; installed/live proof pending | +| Complete result history | Receipt retains every implementer/reviewer/lead attempt, failed fallback, repair, revision, candidate and evidence hash | success, repair, fallback, failure and cancellation history verified offline at `b7d90cc` | +| Crash recovery | Killing a live implementation supervisor cannot create a duplicate writer on resume | prelaunch and live revised-writer recovery verified offline at `f199cd2` | +| Live acceptance | One bounded issue completes across at least two authenticated subscription harnesses with different verified models | pending repairs and normal Claude CLI login | + +## Boundary + +M5 implements only the fixed `issue-delivery` sequence. It does not add an +arbitrary DAG, broad autonomous project implementation, automatic integration, +merge, push or publication. + +## Plan 07-01 checkpoint 1 + +The Claude CLI adapter is available through both the core manifest boundary and +the Bash 3.2 compatibility API. Its verified 2.1.220 profile uses structured +non-interactive output, safe mode, no session persistence, an empty strict MCP +configuration, no browser, explicit role tools and no blanket permission +bypass. The read-only profile exposes only Read/Glob/Grep; the write profile +adds Edit/Write but not Bash or Agent. Offline evidence is 3 focused tests, 215 +full core tests (2 optional-SDK skips), a fresh wheel containing the manifest, +and 220 Bash assertions. This does not claim a live Claude model invocation; +the installed CLI still requires normal provider login. + +## Plan 07-01 checkpoint 2 + +The delivery workspace starts from the frozen target in a detached, run-owned +Git worktree. Candidate freezing rejects out-of-scope edits and escaping +symlinks, stages only after validation, disables repository hooks/signing, +creates one coordinator-owned local commit, and returns stable commit/tree/ +patch/candidate hashes plus added-file evidence. A replay returns the same +candidate; any later workspace edit invalidates it. Five focused tests prove +file names with spaces, no-change rejection, scope and symlink failures, +source checkout/index/HEAD preservation and unchanged local-remote refs. + +The complete gate is 220 core tests (2 optional-SDK skips) and 220 Bash +assertions. The first full discovery observed the pre-existing coordinator- +crash race test return `live` once; that isolated test and the full discovery +rerun passed unchanged. Connecting this workspace to the durable implementer +attempt and its existing store-level writer fence remains the next slice. + +## Plan 07-01 checkpoint 3 + +The offline implementation worker now runs inside the durable M2 process and +writer fence. Its frozen prompt/profile evidence must validate before the +coordinator serializes candidate finalization, rechecks scope, commits locally, +creates independent review/check worktrees and atomically imports the +implementation, candidate and patch artifacts into the same saved ledger. +A concurrent resume observes the live owner and creates no second attempt. +An out-of-scope implementation terminalizes failed without publishing a +candidate. The original checkout and remote refs remain unchanged. + +The complete gate is 224 core tests (2 optional-SDK skips) and 220 Bash +assertions. During this slice a full run reproduced an existing macOS race in +which a zombie-only process group was classified `ambiguous`; `0e88d73` now +uses the non-zombie process-group inventory before declaring ambiguity and has +a direct regression. The isolated service/supervisor/delivery suites and the +complete discovery pass after that fix. Plan 07-02 now owns independent review, +checks, disposition and revision behavior. + +## Plan 07-02 checkpoint 1 + +At `30b98df`, review prompts/evidence accept `issue-delivery` only when the +workflow matches the frozen task. A pending offline review fixture is bound +to the actual candidate after implementation; the durable reviewer then runs +read-only, imports its candidate-bound evidence, executes trusted checks in +the separate worktree, and publishes the correct delivery lead handoff. + +The end-to-end offline regression proves implementation → explicit candidate +resume → review/checks → lead claim, with matching candidate hashes and no +source-checkout mutation. This is not native identity verification or a live +two-harness result. Lead completion, revise-to-implementer iterations, full +terminal history and the remaining M5 fault/live gates are still pending. + +The complete core discovery is **226 tests, suite OK with 2 optional-SDK +skips**, with `ResourceWarning` promoted to error; **220 Bash assertions** +passed. The next implementation slice is Plan 07-02 Task 3, retaining Task 2's +unproven live/stale-revision requirements rather than marking M5 complete. + +## Plan 07-02 checkpoint 2 + +At `a1199c6`, host accept/reject applies the trusted gate to `issue-delivery`, +blocks acceptance after a required-check failure, and emits the complete +five-report terminal set with implementation plus review evidence. + +At `4c76887`, `revise` atomically consumes the saved handoff, switches the +writer fence back to the delivery worktree, binds prior review/check evidence +into a frozen revision request and launches a new implementer. The next local +commit is parented by the prior candidate while its identity and patch still +describe the complete baseline-to-candidate change. Each iteration gets +separate review/check worktrees, so stale live evidence cannot validate +against a replacement candidate; terminal history validates each archived +candidate against its own saved workspace. + +The seeded repair passes implementation → failed mandatory check → host revise +→ new implementation → new review/check → accept with distinct candidate +hashes and roles `[implementer, reviewer, implementer, reviewer]`. Revision and +invocation exhaustion fail before another writer launches, a saved prelaunch +revision resumes to exactly one repair writer, and a headless delivery lead +terminalizes through its own fenced attempt. The complete gate is **237 core +tests (2 optional-SDK skips)** with `ResourceWarning` promoted to error and +**220 Bash assertions**. Delivery fallback/failure/cancellation history, live +Claude implementation and the live two-harness proof remain open. + +## Plan 07-02 checkpoint 3 + +At `b7d90cc`, the frozen delivery package can invoke the exact preflighted +Claude binary without importing source-tree adapter resources at runtime. It +retains the requested and observed identity, native session and usage evidence, +classifies provider faults through the frozen contract, and permits only the +predeclared same-permission fallback. A rate-limited implementer therefore +uses the one frozen fallback without widening tools or permissions, and the +terminal receipt retains both attempts. Delivery-specific cancellation keeps +the prior candidate, attempts and lead disposition visible instead of +collapsing history. + +At `f199cd2`, a real controlled subprocess kills the delivery supervisor while +a revised implementation child is live. Recovery reports retained ownership, +does not launch a duplicate writer, and cancellation reaps the process while +the source checkout and remote refs remain unchanged. The exact checkpoint +passes **241 core tests with 2 optional-SDK skips** and ResourceWarning promoted +to error. The compatibility gate remains **220/220 Bash assertions**; the last +code change after that run added only the delivery-specific Python regression. + +This checkpoint's independent-completion assessment was superseded by the +September 29 review. Repair F1/F2/G4 before the genuine implementation and +different-model review/check/disposition gate. Normal Claude login remains +required for that redacted two-harness receipt. diff --git a/docs/plans/engineering-team/M6-STATUS.md b/docs/plans/engineering-team/M6-STATUS.md new file mode 100644 index 0000000..6941d13 --- /dev/null +++ b/docs/plans/engineering-team/M6-STATUS.md @@ -0,0 +1,157 @@ +# M6 implementation status + +M6 is **in progress with independent repairs and integration work available**. +The component tests below remain useful historical evidence, but the +September 29 review found reused held-out evidence (F3), normal routing that +bypasses lifecycle bindings (F4), and missing public outcome/experiment, +catalog and quota connections (G1/G2). Execute R3–R5 in +[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The separate one-request Jev +smoke completed on October 1; it does not close this engineering work or enable +runtime routing. Its preserved two execution-tier disagreements remain visible. + +R3a now rejects reused outcomes and adds v2 concrete-profile/paired-input +contracts, a stable corpus identity distinct from runtime context, strict +normalized evidence chains and schema-14 prelaunch assignment persistence. +The focused 47-test gate and complete 402-test gate (two optional-SDK skips) +pass at `672e383`, with bounded independent follow-up and 227 Bash assertions. +The SQLite cleanup warning remains R5. This is partial: +the saved-run v2 reader, shared lifecycle eligibility, historical compatibility +audit and public trial execution are not complete. See +[R3a evidence](evidence/R3a-provenance-contract-2026-10-01.json); +do not treat legacy helper fixtures as public proof. + +| Requirement | Planned evidence | Status | +|---|---|---| +| Strict capacity evidence | Typed pool/window/scope/source/confidence/TTL validation; stale, estimated and incomplete measurements remain unknown | verified at `21b1331` | +| Shared transactional reservations | Two projects share one pool; one-slot races produce one owner; schema-8 active attempts survive migration; ambiguous ownership retains the reservation | verified at `26ce5cf` | +| Capacity-aware routing | All applicable windows and profile sublimits affect deterministic selection; a later exhausted observation blocks reservation; paid API remains policy-gated | verified at `d170c01` | +| Public observation/status surface | `squad capacity observe --file FILE`; replay-safe persistence; status shows frozen and current detailed evidence | manual path verified at `d170c01`; native ingestion pending R4 | +| Final and late outcomes | Preserve attempt contribution, lead repair, final success and escaped-defect corrections without crediting failed attempts | manual components verified at `d621df2`; terminal projection pending R5 | +| Comparison reports | Sample sizes, missingness and separated automatic/pinned/experimental evidence | components verified at `45ebc9e`; public runtime evidence pending R5 | +| Frozen experiment evaluation | One-variable paired evaluation/held-out cases, failure evidence, no-change or promotion-proposal verdict and rollback target; evaluation never changes active policy | reopened under R3 for reused/non-comparable evidence and R5 for real assignments | +| Draft proposals | `learn propose` emits content-addressed JSON/Markdown with hashes, sample sizes, missingness, failures and rollback; no evidence yields no-change | verified at `98c6685` | +| Held-out rerun and rollback | Post-change held-out evidence and exercised rollback through lifecycle bindings | historical components at `4b27e0c`; evidence/runtime chain reopened under R3/R5 | +| Model lifecycle | Templates, qualification budgets, reviewed/guarded-auto promotion, compare-and-swap bindings, new-run-only effects and rollback receipts | components through `ca4ee73`; qualification integrity and normal alias routing reopened under R3/R4 | +| Catalog drift and unavailable incumbent | Complete catalog drift scopes revalidation; added models stay unqualified; removed incumbents roll back only to a prior proven/qualified binding or block | component tests at `398ae6a` / `ca4ee73`; production discovery connection pending R4 | +| Decision helper M6-D1 | Default-off typed contract, fake adapter, cache/accounting and authority/integrity tests | verified at `87fa9cf` | +| Jev M6-D2 | One capped synthetic request with exact model/usage/latency/cost receipt | smoke complete October 1; 8/8 family and skill labels, 6/8 tiers; runtime remains off | +| Laya M6-D3 | Triggered pinned local comparison and measured keep-off/adopt decision | pending; run only if the declared Jev trigger fires | + +## Capacity checkpoint + +Schema 9 stores immutable capacity observations and explicit reservations. +Each observation retains its native unit and applicability scope; the latest +observation per scoped window is evaluated without inventing quota. A fresh +authoritative exhausted window wins over shorter available windows. Stale, +estimated, incomplete or absent evidence remains `unknown`, and `allow_bounded` +permits only one unresolved reservation. + +Reservations use the shared SQLite write fence across projects. Attempt +reservations are created atomically with the attempt and reconciled by the same +transaction that proves ownership released; `ownership_ambiguous` remains in +flight. Migration backfills active schema-8 attempts so an update cannot create +a duplicate allowance. + +Public preflight freezes detailed pool and per-profile evidence. Reservation +rederives current observations under the transaction, so evidence that changes +after preflight cannot launch against an exhausted pool. Profile/model-family +sublimits exclude only applicable candidates. The public CLI records strict, +replay-safe JSON, and run status returns both frozen and current windows. + +The checkpoint gate is **255 core tests with 2 optional-SDK skips** and +ResourceWarning promoted to error, plus **220/220 Bash assertions**. + +## Outcome and experiment checkpoint + +Schema 10 adds append-only final and late-correction outcomes bound to saved +runs and attempts. A repaired task can succeed without falsely crediting the +failed original attempt, while a later escaped defect remains attached to the +original final verdict. `squad report --project PATH` reports sample size, +missingness and explicit contribution credit separately for automatic, +pinned and experimental selections. + +Schema 11 adds immutable experiment specifications and replay-safe evaluation. +`squad policy evaluate --experiment FILE` compares declared control/candidate +pairs across evaluation and held-out splits, records missing and failed cases, +enforces non-inferiority/gain/escaped-defect gates and preserves an explicit +rollback version. Its output is only `no_change` or `promotion_proposal` and +always records `active_policy_changed: false`; reusing an experiment ID with a +different specification conflicts. + +The combined checkpoint gate is **263 core tests with 2 optional-SDK skips** +and ResourceWarning promoted to error, plus **220/220 Bash assertions**. + +`98c6685` adds `squad learn propose --project PATH`. It reads one consistent +ledger snapshot, verifies the latest experiment and evaluation hashes, and +writes local content-addressed JSON and Markdown drafts under the runtime +directory. The draft includes selection-mode sample sizes, missingness, all +recorded evaluation failures and the rollback version. Absent experiment +evidence produces an explicit `no_change`; even qualifying evidence produces +only `promotion_proposal`, with `active_policy_changed: false`. The gate is +**265 core tests with 2 optional-SDK skips** plus **220/220 Bash assertions**. + +## Profile lifecycle checkpoint + +Schema 12 persists strict versioned templates, immutable concrete profiles, +bounded qualification runs, compare-and-swap alias bindings, binding history +and decision receipts. Reviewed and opt-in `guarded_auto` promotion require +saved evaluation and held-out evidence. Guarded changes cannot alter policy, +permissions, billing, account route or expand tools. Binding changes affect +new runs only, while an exact concrete pin and an already frozen run do not +float. + +`4b27e0c` requires a saved post-change `no_change` experiment before ordinary +regression rollback. The experiment must compare the active profile with the +requested predecessor, and its hash, failures, metrics and reasons are retained +in the decision receipt. `398ae6a` records complete catalog drift without +changing a binding: only profiles using changed or removed model IDs are +affected, same-ID drift remains explicitly uncertain, and added models remain +unqualified. + +`ca4ee73` adds `squad profile binding-fallback --file FILE`. Complete, +profile-scoped removal evidence may roll an unavailable incumbent back to the +newest prior profile that is proven and, when applicable, backed by a still- +qualified run under the same lifecycle template. Removed, suspended, trial, +unqualified or authority-widening predecessors are skipped. If no safe +predecessor exists the transaction fails without changing the binding. The +catalog evidence and its hash are retained in the immutable rollback receipt. + +The combined checkpoint gate is **275 core tests with 2 optional-SDK skips** +and ResourceWarning promoted to error, plus **220/220 Bash assertions**. + +## Decision-helper checkpoint + +`e1afbc6` adds a strict optional policy and typed observation/response contract. +Omitting the policy or selecting `off` leaves routing unchanged and creates no +call. `shadow` saves a suggestion without changing execution. `advisory` +requires reviewed gate evidence and can only reorder the exact profiles already +accepted by deterministic permission, billing, capability, identity, quality +and capacity filters. Pins cannot move. Unknown IDs, non-finite scores, +distribution errors, adapter/language drift, truncation, lateness, abstention +and insufficient confidence all preserve deterministic routing. + +Schema 13 at `7e83b3a` stores content-addressed decision requests, per-run +links, response/usage evidence and an explicit billable-call count. The launch +fence is written before an adapter boundary. Resume reuses a completed result; +a call launched before a crash becomes `indeterminate` and cannot be silently +retried. Cancellation before launch records zero calls, invalid output is not +persisted as trusted data, and raw task content is represented only by hashes +and byte counts. + +`87fa9cf` integrates the helper into public preflight and status. The frozen +snapshot records the observation and any advisory effect before worker adapter +selection. Missing optional adapters record `unavailable`, zero calls and the +unchanged route. The deterministic fake adapter proves cache reuse across runs, +shadow equivalence and advisory ordering. The frozen +`decision-helper-baseline-v1.json` links only the synthetic public corpus and +explicitly does not authorize advisory adoption. The gate is **291 core tests +with 2 optional-SDK skips** and **220/220 Bash assertions**. + +## Exact next slice + +Finish R3b.1 from the current partial saved-run reader, then shared eligibility +and historical compatibility before wiring R4/R5 through public saved runs. +Jev M6-D2's one-request allowance is spent; preserve its +[receipt summary](evidence/M6-D2-jev-pilot-2026-10-01.json) and do not repeat it. +Keep runtime classification off. Do not install or run Laya unless its declared +trigger is established. A synthetic pilot does not prove production routing quality. diff --git a/docs/plans/engineering-team/M7-STATUS.md b/docs/plans/engineering-team/M7-STATUS.md new file mode 100644 index 0000000..dc0a670 --- /dev/null +++ b/docs/plans/engineering-team/M7-STATUS.md @@ -0,0 +1,99 @@ +# M7 implementation status + +M7 is **in progress with independent normal-entry and readiness work**. +Installation and the recorded Codex/Gemini operations work, but the September +29 review found routing, terminal handoff, authentication-readiness and check- +discovery gaps (F4/G3/G4). Execute R4/R6/R8 in +[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). Claude and Grok authentication +remain additional blockers for the corresponding live proofs. + +| Requirement | Evidence | Status | +|---|---|---| +| Fresh standalone install without Claude | Immutable base install and composite `--core-only` regressions | verified | +| Optional MCP environment | Exact offline lock, one-line JSON, `pip check`, official SDK tests | verified | +| Reinstall/update safety | Idempotence, drift detection, stable launcher, retained old release and active-run survival | verified | +| Legacy plugin migration | Complete `plugin/core` package, no duplicate global hooks, second setup unchanged | verified | +| Host registration | Four installed hosts load the stable launcher and report matching | verified | +| Terminal operation | Fresh saved run started and later cancelled at version 20 | verified live | +| Codex App/CLI operation | Ephemeral gpt-6-luna/low called installed `squad_status` on the saved run | verified live | +| Antigravity operation | Gemini 3.8 Flash Low called installed `squad_status` with one project-scoped grant | verified live | +| Claude Code operation | Registration matching; real operation requires normal login | blocked externally | +| Grok Build operation | Registration matching; real operation requires renewed authentication | blocked externally | +| Installed two-model delivery | Genuine Claude implementation followed by different-model Codex review/check/disposition | pending M5 repairs and Claude authentication | +| Documentation/CI | Runtime guide, generated command reference and macOS/Python/optional-MCP workflow | historical installation checks pass; readiness/quickstart updates pending R6 | +| Normal task entry | Installed `squad review --base ...` and `squad fix "..."` dry-runs plus an offline end-to-end delivery | reopened under R4/R6 for approved routing, terminal completion and check coverage | +| Live normal Codex review | Exact commit range, verified read-only gpt-5.5/low review, isolated trusted checks and artifact-bound host acceptance | verified live at `b1d52ad` | + +## Current installation + +The stable launcher is `/Users/Dikshant/.local/bin/squad`. The active immutable +release is +`0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef`, using Python 3.12.14 and +`mcp==2.2.0`. A repeated install reports `changed=false`, `pip check` passes, +and `squad setup` reports every host unchanged. + +Codex capability drift was revalidated rather than inferred. Bundled +`codex-cli 0.155.0-alpha.9.2` passed `initialize` and a complete seven-model +`model/list`; the runtime now chooses that verified binary over PATH Codex +0.135.0. Unknown versions remain fail-closed. + +## Test gate + +- 317 Python core tests passed with `ResourceWarning` promoted to error; two + optional-SDK tests skipped in the dependency-free interpreter. +- 227/227 Bash assertions passed across 11 files. +- Eight focused installer tests passed. +- Twenty-two focused tests passed in the installed official MCP environment. +- The installed runtime passed `pip check`; setup and doctor report ready. +- The installed `review` and `fix` dry-runs resolved exact Git commits and the + requested verified Codex profile without a model generation. The offline fix + simulation created one isolated writer, froze its candidate, ran independent + review and checks, preserved the source checkout and reached host handoff. +- The installed live `review` path selected verified Codex 0.155/gpt-5.5/low, + returned a clean exact-candidate review, passed both frozen checks and + terminalized `succeeded` after an artifact-bound host acceptance. + +The recurring interpreter-finalization `ResourceWarning` was printed as an +unraisable cleanup diagnostic during full discovery, but the warning-as-error +process completed successfully. It is retained as an observation, not +misreported as a failing test. + +## Live receipts + +The terminal-created run `f121e96b-502a-4986-a3c3-7ba8ab418bb3` was observed +through both Antigravity and Codex as `awaiting_host`, version 14, with next +action `claim_handoff`. Terminal cancellation then moved the same run and its +open handoff to `cancelled`, version 20. Raw provider output remains private; +hashes and redacted usage are in +[the portable M7 evidence](evidence/M7-installed-runtime-2026-09-29.json). +Normal-entry installation and test details are in +[the normal-entry evidence](evidence/M7-normal-entry-2026-09-29.json). The +subsequent installed live review is recorded in +[the live Codex evidence](evidence/M7-live-codex-review-2026-09-29.json). + +The live review run `c611ad4c-5473-4f05-a870-a136f24464d3` bound +`9d70888..b1d52ad`, observed verified gpt-5.5/low read-only execution, returned +zero findings and passed both `git diff --check` and `bash test/run.sh` inside +the trusted check workspace. Its host disposition accepted the exact four +artifact hashes and terminalized at version 22. This proves the installed +Codex review path; it does not replace M5's still-required genuine Claude +implementation half of the two-harness delivery gate. + +Antigravity headless mode requires an explicit project grant for unattended +inspection. The verified minimum is `mcp(devsquad/squad_status)`; no global or +blanket permission bypass was used. Codex used an ephemeral minimal config, +read-only sandbox, no conversation resume and only the DevSquad MCP server. + +## Exact remaining work + +1. Complete R4/R6 normal routing, terminal disposition, optional run resolution, + authentication readiness and committed-target check discovery offline. +2. After normal Claude login, execute the saved M4 handoff and the M5 genuine + Claude implementation to different-model Codex delivery. +3. After Grok login renewal, record one supported Grok Build operation against + the same installed runtime. +4. Run the affected final gates and mark M7 complete only when every required surface + receipt and installed delivery is present. + +`squad council` remains attached to the separately gated C1 implementation and +does not reopen M7. diff --git a/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md b/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md new file mode 100644 index 0000000..ea9fea7 --- /dev/null +++ b/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md @@ -0,0 +1,95 @@ +# Local MCP access — operator contract + +This is the minimal M4 host contract for the shared local DevSquad runtime. It +does not create a cloud service, select a model for the host, or copy routing +policy into app-specific prompts. + +## One durable run, interchangeable clients + +Every supported local app launches the same absolute executable over stdio: + +```text +/absolute/path/to/squad mcp serve --surface HOST +``` + +The four versioned templates are under `plugin/core/integrations/`. Setup +resolves both the host CLI and `squad` to existing absolute files before it +registers anything. Host labels are saved as provenance only; they never grant +authorization. Worker/delegation environment markers, not a caller-supplied +label, deny recursive workflow mutations. + +Use the tools in this order: + +1. Construct one strict v1 task with bounded criteria, scope, checks and + budgets. Call `squad_start` once with a stable idempotency key and retain the + returned run ID. +2. Use `squad_status` for the projection and next action. Page + `squad_events` with `after=next_cursor`; do not treat provider reasoning as + an event stream. +3. Use `squad_result` for receipt/artifact IDs, paths and hashes. Its optional + UTF-8 previews share one 16 KiB ceiling; the files remain the authority. +4. If status reports `claim_handoff`, call `squad_handoff_claim` with the + displayed run version and a stable local owner label. Submit the returned + claim unchanged with a candidate-bound decision to + `squad_handoff_complete`. +5. `squad_cancel` saves cancellation intent; it does not wait for a worker's + lifetime. Use `squad_resume` only with the recovery object requested by + status. + +Closing an app or its MCP client does not cancel a saved run. The detached +worker is owned by the machine-local runtime, and another client can continue +with the same run ID. Two hosts still cannot advance the same handoff because +the service checks its fencing token and run version. + +## Registration templates + +| Host | CLI used by setup | Scope | Inspection form | +|---|---|---|---| +| Codex App/CLI | `codex mcp add` | Codex user config | `codex mcp get ... --json` | +| Claude Code | `claude mcp add --scope user` | user | `claude mcp get` | +| Antigravity | `agy mcp add` | Antigravity user config | `agy mcp list` | +| Grok Build | `grok mcp add --scope user` | user | `grok mcp list --json` | + +Install the optional, exactly pinned MCP extra into the same immutable release +that owns the `squad` executable. The installer is offline-only and requires a +wheelhouse containing every locked dependency; then preview or apply +registration: + +```text +./scripts/install-core.sh --with-mcp --mcp-wheelhouse /absolute/path/to/wheels +squad setup --dry-run --json +squad setup --json +squad doctor --json +``` + +Use `--host codex`, `--host claude-code`, `--host antigravity` or +`--host grok` to limit setup; repeat `--host` for more than one. Setup calls +the installed host CLI with argument arrays, never a shell command, and then +re-inspects what that host reports as loaded. A second successful setup is a +no-op. Claude Code requires a targeted user-scope remove/add only when its +existing direct `devsquad` entry has drifted because its CLI does not replace a +named server in place. + +Setup fails closed instead of editing an ambiguous configuration when it sees: + +- the server in more than one direct scope; +- a loaded registration that differs from the one visible in the direct + config, indicating an inherited override; +- a project/local registration, malformed config or a config the host does + not load; +- an unresolved launcher, unavailable host CLI, missing SDK or any MCP SDK + version other than the supported `2.2.0` pin. + +`squad doctor --json` is read-only. It reports adapter versions, the resolved +launcher, installed SDK version, config paths, normalized host inspection and +whether each installed app loads the expected absolute command. Environment +maps are never returned. Arguments are returned only when they exactly match +the fixed DevSquad server arguments; drifted arguments are replaced by a count +and SHA-256 digest so credentials cannot be echoed. Doctor exits 1 when an +installed app is not ready, while unavailable apps are not treated as required. + +These commands prove local CLI registration and SDK conformance. Actual +in-app operation still requires the cross-surface receipts in M4/M7; config +syntax or a matching `mcp list` result is not presented as that live proof. +The full install, operate and recovery procedure is in +[the runtime guide](../../RUNTIME-GUIDE.md). diff --git a/docs/plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md b/docs/plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md new file mode 100644 index 0000000..de72939 --- /dev/null +++ b/docs/plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md @@ -0,0 +1,116 @@ +# Model releases without routine DevSquad code updates + +**Design amendment, 2026-09-06; implementation pending.** Builds on [automatic selection](SELECTION-AND-COUNCIL.md), [contracts](CONTRACTS.md) and [ADR-002](../../adr/ADR-002-surface-independent-engineering-team.md). Incorporate into M1/M3/M6/M7; this adds no prerequisite platform or new core milestone. + +## What to adopt from OpenAI's Claude Code plugin + +Reviewed [openai/codex-plugin-cc](https://github.com/openai/codex-plugin-cc/tree/db52e28f4d9ded852ab3942cea316258ae4ef346), pinned at `db52e28f4d9ded852ab3942cea316258ae4ef346`. Source inspection only: no installation, account changes, session imports or live model requests. + +| Source pattern | DevSquad application | +|---|---| +| App-server thread/turn operations and structured events | Prefer Codex's native protocol for new jobs; preserve the legacy shell wrapper | +| Native review and a separate steerable adversarial review | Distinguish ordinary defect review from a targeted challenge; record review mode | +| Stored job status/result and native thread IDs | Keep DevSquad run IDs plus native thread/turn IDs for recovery and reopening | +| Native turn interruption | Interrupt the owned turn before bounded process cleanup; confirm termination | +| Model/effort overrides with native defaults when omitted | Discover defaults as metadata; resolve a verified run profile before execution | + +The implementation uses `review/start` for ordinary review and `turn/start` with effort/output schema for task execution. It also supports native thread resume and session import. These are concrete capabilities to reuse through an adapter. [Native integration source](https://github.com/openai/codex-plugin-cc/blob/db52e28f4d9ded852ab3942cea316258ae4ef346/plugins/codex/scripts/lib/codex.mjs). + +The plugin distinguishes read-only reviews from delegated editing and provides structured findings plus instructions to preserve verdict/evidence boundaries. DevSquad should retain this distinction while applying the user's already-authorized workflow: a delivery task can include bounded corrections, whereas review-only scope cannot. [Review schema](https://github.com/openai/codex-plugin-cc/blob/db52e28f4d9ded852ab3942cea316258ae4ef346/plugins/codex/schemas/review-output.schema.json), [Result handling](https://github.com/openai/codex-plugin-cc/blob/db52e28f4d9ded852ab3942cea316258ae4ef346/plugins/codex/skills/codex-result-handling/SKILL.md). + +The plugin's optional stop-time review gate can repeatedly return work to Claude; its README warns of extended loops and usage drain. DevSquad's existing task-scoped review gate, finite revisions and budgets remain the design. The plugin also uses the user's local Codex authentication/configuration and contributes to the same Codex limits. It does not create another allowance pool. [README](https://github.com/openai/codex-plugin-cc/blob/db52e28f4d9ded852ab3942cea316258ae4ef346/README.md). + +Native Claude→Codex transcript import is an optional convenience when supported. It does not imply every host can import every other host's history. Keep portable artifact handoffs as the common interface. The source resolves the selected transcript under the real Claude projects directory before importing. [Transfer path validation](https://github.com/openai/codex-plugin-cc/blob/db52e28f4d9ded852ab3942cea316258ae4ef346/plugins/codex/scripts/lib/claude-session-transfer.mjs). + +### Adapter amendment + +The new adapter abstraction supports `transport: cli_exec | native_protocol`. Both expose discovery, validated preparation, execution events, interruption, result normalization and optional resume. Codex should use `native_protocol` when the installed app-server passes conformance. Other providers keep `cli_exec` unless they expose a verified suitable native interface. + +For Codex, use a **DevSquad-owned stdio app-server child** inside the existing per-run supervisor. Avoid adding the upstream plugin's separate shared broker/job database as a second coordinator. Python owns that child's lifecycle; the adapter sends typed protocol requests. Store native thread/turn IDs and wait for actual completion notifications; accepting a start or interrupt request is not completion. Handle server requests with the declared role policy, never blanket-approve them. Use the isolated run worktree as native cwd and validate effective permissions. Protocol failure must not silently launch a duplicate CLI writer. + +Generate/test protocol schemas against supported installed versions. Keep experimental methods behind capability checks. An explicit fallback transport must satisfy the same role/model/effort contract and be selected before launch, or after safe reconciliation. Standard `review/start` and custom `turn/start` do not necessarily accept identical effort controls; unsupported explicit combinations fail visibly. The Python coordinator remains unchanged in purpose; Bash callers keep their existing four-prefix API. + +If implementation reuses upstream code, retain Apache-2.0 license/NOTICE obligations and pin provenance. Adopting behavior does not require vendoring the entire plugin. The protocol patterns are the immediate benefit; DevSquad still owns cross-provider scheduling, outcome evidence and worktree concurrency. + +## Stable names above changing model releases + +Workflows refer to stable **profile aliases**, for example `implement.balanced`, `review.deep`, `research.current`. These are DevSquad policy names, not guesses about provider tiers. Each alias binds to a tested concrete execution profile. + +```mermaid +flowchart TD + W[Stable role alias
review.deep] --> B[Current qualified binding] + B --> P[Exact harness + model + effort + tools] + C[Live provider catalog] --> N[New / changed candidates] + N --> V[Compatibility checks + bounded evaluation] + V --> G{Promotion rules satisfied?} + G -->|yes| B + G -->|no / insufficient evidence| K[Keep known working binding] +``` + +| Layer | Changes when | +|---|---| +| Workflow and role requirements | Your engineering process changes | +| Adapter code/protocol mapping | A CLI, protocol or permission interface changes | +| Model catalog and effort/tool metadata | Models or account-visible capabilities change | +| Alias → concrete profile binding | A replacement qualifies under the update policy | + +**Routine model releases should be data updates.** A changed CLI/protocol or unsupported capability may still require an adapter update. This reduces maintenance; it cannot make third-party interfaces permanently stable. + +### Discovery, not name guessing + +Codex app-server documents `model/list` with supported/default reasoning efforts, modality and upgrade metadata, and `account/rateLimits/read` for quota observations. Its local generated schema also exposes these method types. This replaces DevSquad's current hardcoded `codex: unlistable` assumption. The upstream Claude plugin's execution code and the app-server's catalog API are distinct evidence sources; do not claim the plugin itself implements a model-release router. [Official app-server documentation](https://learn.chatgpt.com/docs/app-server). + +Local verification used `codex app-server generate-json-schema` with the installed CLI and inspected `ModelListResponse`, `ThreadStartParams`, `TurnStartParams` and `GetAccountRateLimitsResponse`. This verifies available schema shapes, not account entitlement, model quality or a successful live model invocation. + +DevSquad already has a [catalog implementation](../../../plugin/lib/model-catalog.sh) and detached refresh. Extend that intent, replacing numeric/keyword tier ranking with structured identity and compatibility. A model version number is not comparable across families, and a provider's recommended default/upgrade is a candidate recommendation, not a benchmark result. + +Discovery contract: + +- Cache per harness, installed version, authenticated account identity reference and configuration scope. Use native structured metadata first; otherwise a version-tested CLI parser or curated manifest. API catalogs cannot establish subscription-CLI entitlement. +- Refresh outside hooks on first use, stale cache (initial TTL 24 hours), a changed CLI/config fingerprint, or a relevant model-not-found signal. Deduplicate refreshes with a lease, timeout, backoff and pagination. No permanent polling service is required. +- Keep the last successful snapshot on timeout, auth failure, parse error or an incomplete response. Record staleness and errors separately. A failed refresh is never a mass-removal event. Confirm removal with a complete scoped catalog or an explicit model-unavailable response, distinguishing account/auth problems. +- Store exact IDs, provider family when verified, native effort options/defaults, modalities, tool evidence, deprecation/upgrade hints, source and timestamps. Missing fields remain unknown. Do not infer tools or reasoning levels from a model name. +- Generate a small number of candidate profiles from **existing allowed templates**. Start with the template's supported effort intent and at most one experimental alternative. Do not enumerate every model × effort × tool permutation. Unknown families or widened capabilities require a policy/template change. +- A provider alias can change behind the same public ID. Record reported effective identity/revision when available; otherwise label the backing revision unknown and use metadata drift plus periodic behavior checks. Reproducibility is best-effort when providers expose no immutable revision. + +## Review the update rules once; automate routine promotion + +This refines the earlier “update defaults after review” statement: **review can authorize a bounded promotion policy once**, rather than require a human to approve every qualifying model release. Start with reviewed promotions while calibrating the evaluation; enable guarded automation after those checks prove useful. + +```mermaid +flowchart LR + D[Discover automatically] --> T[Compatibility smoke checks] + T --> E[Bounded held-out evaluation] + E --> C[Limited eligible trial] + C --> P{Preauthorized gate} + P -->|pass| A[Promote binding for new runs] + P -->|fail / unknown| R[Retain or revert binding] + A --> O{Regression / incompatible drift?} + O -->|yes| R + O -->|no| K[Keep qualified binding] +``` + +`model_updates.mode` is `reviewed` initially, or opt-in `guarded_auto`. A versioned update policy specifies allowed alias/templates/harnesses/families, evaluation cases/rubric, minimum evidence, quality non-inferiority criteria, critical-defect rule, latency/usage tolerances, candidate/trial budgets, rollback target and stop conditions. These are task-class-specific thresholds, not a global model leaderboard. Missing thresholds or insufficient evidence mean no automatic promotion. + +Metadata refresh and candidate proposals are automatic. Inference-based probes/evaluations/trials require an enabled, bounded qualification budget, share the existing experiment allowance and respect account-pool limits. `guarded_auto` cannot silently enable paid APIs, new tools/permissions, a new account route or a different policy. Those changes remain reviewed. A policy may permit a low-risk candidate trial; high-risk work keeps the qualified incumbent until the gate passes. Ordinary model-launch counts and native usage remain distinct. + +Promotion atomically updates an alias binding under a policy revision, records the previous binding and evidence IDs, and affects **new runs only**. Every run freezes the resolved concrete profile/fallback set before its worker starts. Active runs and pinned-profile overrides never move just because a new catalog appears. Retired/unavailable incumbents use an already-qualified fallback or block; do not substitute an untested “latest” model. Rollback also affects new attempts only after ownership is reconciled, and chooses an available qualified predecessor rather than a removed model. + +Only revalidate affected combinations after effort/tool/protocol drift. Evidence remains attached to the original fingerprint; passing a previous model's evaluation is not inherited automatically. Reduce trial scope and retain the incumbent when budgets or usable cases are limited. Automatic promotion is a static gate over measured evidence, not an unconstrained learned router or a self-reported confidence score. + +### State and documentation + +Keep catalog snapshots and immutable concrete profiles in the runtime registry. Add `profile_templates`, `profile_bindings` and `qualification_runs` to the transactional store. Project policy references stable aliases and allowed templates; the local binding ledger determines the current qualified concrete profile for that installation/account. Resolve and snapshot binding versions with task preflight; idempotent start replay never re-resolves them. + +Each binding change writes a local decision receipt: alias, old/new concrete IDs and fingerprints, effective/unknown settings, source release/catalog, evaluation/trial evidence, measured limits/usage, policy gate, rollback target and actor (`human` or `guarded_auto`). Derived Markdown reports summarize discovered, trialled, adopted, rejected and retired profiles. Curated exports remain available for Git documentation; runtime promotions do not edit or auto-commit the user's checkout. No new schedule or automation is installed by this design. + +## Additions to existing milestone gates + +| Milestone | Addition | +|---|---| +| M1 | Native/CLI transport contract; Codex app-server metadata; last-good catalog, templates and alias schema; unsupported effort handling | +| M3 | Resolve stable aliases to immutable profiles, explain selection and snapshot bindings; standard/adversarial review distinction | +| M6 | Qualification budget, reviewed/guarded-auto promotion, evidence receipts and rollback; real Codex quota observations when available | +| M7 | Compatibility fixtures across supported CLI/protocol versions and installation drift; document optional native session handoff | + +Required cases: paginated discovery; incomplete/error catalog retains last-good entries; added model does not become default; changed effort support revalidates only affected profiles; same-ID drift retains uncertainty; alias promotion affects new runs only; exact pin never floats; insufficient trial evidence blocks promotion; permissions/billing changes cannot auto-promote; failed/retired incumbent uses only qualified fallback; concurrent promotions use compare-and-swap binding versions; regression rolls back and produces a decision receipt. Native adapter tests cover server disconnect, interruption acknowledgment without termination, isolated cwd/permissions and resuming a still-active turn without a duplicate writer. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md new file mode 100644 index 0000000..495acb3 --- /dev/null +++ b/docs/plans/engineering-team/RESUME.md @@ -0,0 +1,149 @@ +# Resume DevSquad after an interruption + +Read this checkpoint, backlog.json and SOL-HANDOFF.md, then compare Git status +and recent commits. Retain newer work. Continue codex/engineering-team; no stash, +reset or restart from main. Detailed receipts and failed gates remain in evidence. + +## Current checkpoint — October 2, 2026 + +The user authorized a public **GitHub release in joshidikshant/devsquad**. +Release v0.11.0 ships plugin0.11.0 and standalone core0.1.0; no registry or +hosted service. PR1: https://github.com/joshidikshant/devsquad/pull/1. +Publication is pending final public CI, merge and exact-commit release assets. +The whole engineering-team plan remains incomplete. + +Latest public CI37084805784 on5a576d3: Bash259/11 files (50 focused) and +optional MCP passed; Python3.14.7 FAILED486 tests/679.063s at coordinator +crash receipt recovery (fixed.6s sleep, live worker). Python3.11 PASSED604/ +1013.419s, two SDK skips/no errors/failures/unraisable on that earlier head. +Runtime's live-process refusal is correct. Isolated terminal/ +readiness repair is limited to positively synchronizing the test's child-start, +receipt publication and actual owned runner exit; no core change. Controlled +original red and synchronized green proven on3.12.14/local3.14.6 (not CI.7). +edfcf65 test review clean/agent33 module each3.12/3.14/Bash259 pass. +Root integration Bash259 passed;33 service tests FAILED1/32.435s at the +neighboring before-gate crash waiter598 (ambiguous !=dead). Concurrent Bash +alone is not causal proof. Isolated test-only exactdead wait repair retains +the5s boundary and all before-gate/exact-once assertions; no core edit. +e7b63b0 confirmed-dead waiter is integrated: both independent reviews clean, +same5s boundary/core unchanged, agent33 service tests3.12/3.14 pass; +root complete33 module PASSED29.820s/no failures/errors/skips. Root Bash259/ +11files/reference/JSON/diff passed. Corrected checkpoint/push/new exact-head +public CI pending. No merge/tag. + +Corrected remote candidate f40c618caefd100637c4686660673e159899ecb7 is now +pushed; PR CI37086958301: Bash/MCP passed, Python3.11.9 FAILED137/191.313s +at delivery repair recovery1001 (ownership_ambiguous !=retain_ownership). +Python3.14.7 passed604/697.362s/twoSDKskips/no failures/errors/unraisable; +whole f40 workflow failed311 and is not accepted. Duplicate push cancelled. +f40 is rejected as a publication candidate by that failed311 job. Retain it +as source baseline only. Next integrate the narrow delivery fixture repair, +push a new exact candidate and require final public gates before merge/tag. +Later docs-only checkpoints are recovery metadata, not publication targets. + +New bounded isolated repair: test SIGKILLs attempt_runner, not coordinator, +then assumes.1s implies blocked. Correct running import returns safe +ownership_ambiguous before typed blocked recovery. Positively wait for the +same dead runner and blocked/current attempt ownership_ambiguous before one +explicit retain request. Preserve exactly3 attempts/no new writer/launchedfalse/ +source/cancel assertions; no core edit.893490a phasefix is now integrated, +exact7723f6ce reviewed blob. Agent52 delivery/service tests pass each3.12/3.14; +root19 delivery tests48.911s pass; unchanged service33/29.820s already passed. +Final root Bash passed259 assertions/11files (50 focused), reference/JSON/ +staged+unstaged diff checks passed. Checkpoint/push and exact-head public CI +remain. No merge. + +Rejected private diagnostic RELEASE+cancel cleanup overlapped import and +raised stale-phase ConflictError. Independent read-only triage: fail-closed +completion fence, no duplicate-writer/false-cancel evidence. No error-time +ledger snapshot proves ordering/final rows/lost-intent liveness; retain as +diagnostic concurrency-conflict / possible follow-up, not confirmed defect. +Corrected helper waits import DONE before teardown; no product edit/probes. + +## Accepted gates — do not repeat unchanged + +- Frozen core b83dda7fbf691503d3adf3c9ea6ecebd4071ba16 passed604 tests/ + 793.462s, two optional SDK skips, zero failures/errors/unraisable diagnostics. + Core tree62eea7fa31153ef732fd5e9dfd97951d95b367d8 is unchanged by the + subsequent legacy Bash/CI repair and documentation commits. +- Actual pristine accepted-R5 trusted-check preflight passed60 tests/33.182s + with unchanged tracked/all-input fingerprints and zero undeclared bytecode. + Input fingerprints are not branch-review candidate identities. +- Exact native EOF/fixture audit cb68f3ae-bea0-4744-90c4-96f25418fb8c: + verified Codex0.159.2/gpt-6.1-sol/low/read_only, clean, succeeded22/host accept. + Four required checks and integrity passed;60 affected tests/36.416s. + Native95970 tokens are not an invoice or Plus-window forecast. +- R3/R4/R5 accepted closure evidence and earlier live Claude CLI handoff, + Claude implementation → different verified Codex review → tests, and + Grok MCP operation remain valid for their recorded candidates/scopes. +- Final legacy watchdog checkpoint3fdc6bd065c577a9ed59566573fdce399e9b585a + is integrated. Real-deadline FIFO timer, buffered cancellation and restored + caller EXIT trap preserve Bash3.2/optional-jq and original timing limits. + Root integrated259 Bash assertions/45 affected Python tests0.927s pass; + agent gates also pass; independent final + review clean with50 legacy assertions. Exact reviewed adapter/test blobs: + 1847357c19677f276b06886386e2ac93e586b203 / + 8cc3fe9f9ee9df77b2fcb2d75c76c236a1f8d13a. + Initial CI deadline drift, rejected timer designs, caller-path cleanup defect + and corrected scheduling-only fixture failure remain honestly recorded. + +## Installed runtime and surfaces + +Current immutable release: +0.1.0-py31214-303a0e0a6c87-mcp-a26bc88afbef; Python3.12.14/MCP2.2.0/schema17. +Stable launcher /Users/Dikshant/.local/bin/squad. Zero nonterminal runs before/ +after update; pre16 online backup0600/integrityOK and old releases retained. +Reinstallchangedfalse/all driftfalse/manifests match/pipcheck pass, zero downloads. +Nine actual installed SDK tests/3.126s pass with zero skips/failures/errors. +Doctor ready for branch review and Claude→Codex delivery; four registrations +match, subscription authentication verified for Claude/Codex. + +Updated Antigravity means **agy CLI1.2.14**, NOT IDE. Actual +Gemini3.8FlashLow devsquad/squad_status observed cb68 succeeded22 in18.654s. +It read only its own MCP schema/generated result, not project files/other +servers; no zero-file-read or automatic-worker claim. +Grok/agy automatic worker readiness remains unknown; MCP proof is separate. + +This chat's existing Codex MCP server returned SCHEMA_UNSUPPORTED because it +still uses an old package. New CLI and agy succeed on the same saved run. +A scoped DevSquad-only reconnect request is pending; do not change global +settings or repeatedly probe the unchanged old connection. This does not block +the public CLI release. Local Claude plugin is still0.10.0: after main is +released run its normal scoped plugin updater and verify0.11.0. + +## Exact next action + +1. Root integrated watchdog Bash259/affected45/reference/diff/JSON gates pass. + Both receipt/dead-wait test repairs are integrated/reviewed; root33 passes. + Final Bash/JSON/reference/diff checks passed; checkpoint the corrected candidate. + Core source/full/native/installed gates need no unchanged repeat. +2. Push codex/engineering-team, require final exact-head public CI green: + legacy Bash3.2, core Python3.11 and3.14, optional MCP. Earlier CI + 37082086201/37082130929 was cancelled after true legacy failure; optional + MCP passed, cancelled core is NOT accepted. Core now fetches full history + for immutable migration fixtures and uses the tracked failfast runner. +3. Merge PR1 with exact-head fence without deleting engineering branch. Build + source/plugin archives from exact merged commit; old b83 archives are stale. + The unchanged-core wheel may be reused only after tree/hash verification. +4. Publish v0.11.0 on that exact merge SHA with verified assets/SHA256SUMS. + Verify tag, published-not-draft state and downloaded asset hashes. Update + release evidence/backlog/this checkpoint and local Claude plugin. End clean. + +## Residual scope and boundaries + +Council mechanics/schema17/reconciliation are integrated and offline tested. +Final nongenerating diagnostic is exhausted: sandbox network_request_failed/ +noHTTP, unsandboxed control strict-response rejection; zero generating calls. +Native Council remains unavailable; fixture comparison inconclusive, automatic +OFF. No additional network attempts or permission widening. +Jev OFF; the single authorized pilot is spent. Laya is conditional, not adopted. +Claude desktop Code-tab proof remains unverified/TCC; CLI is not a substitute. +No credit purchase, reset redemption, paid API fallback or global AI settings +change. Raw provider diagnostics/credentials remain outside tracked evidence. + +Evidence: evidence/public-release-0.11.0.json, +evidence/legacy-watchdog-release-repair.json, R6-terminal-readiness-partial, +evidence/release-receipt-synchronization.json and +evidence/release-runner-exit-synchronization.json, +R7 reconciliation/final-network diagnostics, R8-installed-workflows and +R3/R4/R5 closure files. Historical recovery detail is preserved in Git. diff --git a/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md b/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md new file mode 100644 index 0000000..c3edf66 --- /dev/null +++ b/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md @@ -0,0 +1,135 @@ +# Automatic selection and selective councils + +**Design amendment, 2026-09-06. Nothing here is implemented yet.** Clarifies the [contracts](CONTRACTS.md) and adds a separately gated Council extension to [ADR-002](../../adr/ADR-002-surface-independent-engineering-team.md). It follows the user's question about automatic selection and LLM Council. + +## Who chooses what? + +**Everyday selection is automatic. Manual selection is an override. Changes to the selection policy are reviewed.** + +| Decision | Owner | +|---|---| +| Discover installed harnesses, models, supported efforts and tools | DevSquad's discovery/probes | +| Connect accounts, identify shared allowance pools, set spending/access limits | User setup, assisted by discovery | +| Frame the task, scope, acceptance criteria and required capabilities | Current host lead, or supplied terminal task | +| Suggest task labels, relevant skills/context or eligible-profile rankings | Optional evaluated decision helper; advisory data, never authority | +| Select model, effort and permitted toolbox for each role | Deterministic router applying versioned policy and current availability | +| Decide which permitted tool to call during work | Selected worker, inside the assigned permissions | +| Pin a particular configuration for this task | User override, resolved by the host into a validated profile | +| Collect outcomes and propose better configurations | DevSquad's evidence and evaluation loop | +| Promote a changed default policy | Reviewed versioned change; it may preauthorize bounded model-binding updates | + +You should usually say “fix this issue” or “review this branch,” not fill in a model matrix. The host prepares a task; the router selects an eligible profile. The router itself needs no model call. A profile packages an exact harness/model, native effort setting, tool access, permission policy and account pool. Selection among tested combinations keeps the search space manageable. + +```mermaid +flowchart TD + U[Task and requirements] --> O{Explicit role override?} + O -->|yes| P[Validate pinned profile] + O -->|no| F[Filter eligible profiles] + P --> Q[Check capabilities, permissions and capacity] + F --> Q + Q --> R[Use valid pin or choose from preferences] + R --> W[Run with an explanation of the choice] + W --> E{Acceptance met?} + E -->|no| X[Bounded retry / configured escalation] + X --> Q + E -->|yes| L[Record outcome and actual settings] + L --> H[Evaluate policy improvements] +``` + +Effort is selected with the profile. Easy work can use a proven lower-effort configuration; difficult work can use a stronger validated configuration. Escalation follows declared failure/rework rules and a finite budget. A higher effort label is not treated as a universal quality score, and no model gets an automatic maximum setting simply because it supports one. + +The router selects the allowed toolbox; the worker chooses actual calls within it. Required X/search/video access must exist on that harness/profile. It cannot grant an app-native capability to an API model by naming the vendor. Selecting a profile never installs tools, adds permissions or purchases API usage implicitly. + +### Override contract + +Add optional `routing.overrides`, keyed by model role. Each value contains `profile_id` and `fallback` (`none` by default, or explicit `policy`). Empty/missing overrides means automatic selection for every role. Pinning one role leaves the other roles automatic. Pinning every model role gives manual assignment. + +Illustrative fragment, using a fictional configured profile: + +```json +{ + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + "overrides": { + "reviewer": {"profile_id": "my-verified-review-profile", "fallback": "none"} + } + } +} +``` + +Validate role names against the selected workflow. A pin must meet the same capability, identity, quality, permission and billing constraints as automatic candidates. Invalid settings return a validation error; temporary unavailability blocks with a reason. `fallback:none` never silently substitutes another profile. `fallback:policy` permits only the ordinary qualified candidate list and logs the substitution. Profile edits produce a new version; no live model/effort/tool mutation mid-attempt. All surfaces submit this same task shape. + +Reports explain selections, excluded alternatives, explicit overrides, escalations and observed settings. An unmeasured or manually pinned trial must not be counted as proof of a general routing improvement. + +The [model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md) adds stable aliases, automatic discovery and bounded qualification. After calibration, an enabled `guarded_auto` policy can promote tested bindings without per-release manual edits. The reviewed policy remains the authority; discovery or council votes alone cannot promote a candidate. + +The September 26 [Jev/Laya decision-helper amendment](DECISION-CLASSIFIERS.md) +adds a default-off evaluation within M6: one tightly capped Jev synthetic +smoke first, with Laya as the local fallback if access, cost or measured quality +justifies its heavier setup. Frozen suggestions may +inform only a reviewed policy after use-case-specific held-out validation. +The router remains deterministic: validate requirements, pins and eligible +profiles before considering suggestions; invalid/uncertain/missing output uses +the existing policy. Helpers cannot lower task quality requirements, relax +capacity, grant tools or trigger extra workers. Skill/context recommendations +retain mandatory instructions and evidence. This is not a second planner or a +change to C1's independent evaluation and explicit-invocation gates. + +## What LLM Council actually contributes + +Studied **Karpathy's original repository**, pinned at [`92e1fcc`](https://github.com/karpathy/llm-council/tree/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131). This is a source review, not a performance benchmark or a survey of forks. Its author describes it as an exploratory, unsupported project. [Original README](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/README.md). + +```mermaid +flowchart LR + Q[One question] --> A[N independent answers] + A --> B[N anonymous-label peer rankings] + B --> C[Configured chairman synthesizes] +``` + +The code gives every ranker all successful answers, including its own, in the same order. The chairman sees named answers and raw critiques. Ranking uses permissive text parsing and average positions. For four members, a normal completed run makes **nine chat-completion requests**: four answers, four rankings, one synthesis. Ranker input repeats all answers, so aggregate answer-content volume grows roughly quadratically with membership. These are code-derived counts, not measured spending or quality improvements. [Council implementation](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/backend/council.py). + +Membership/chairman are configured explicitly; the API request supplies model and messages without effort, tool definitions or account-quota selection. It uses OpenRouter API access, so running this original app would not automatically draw from your CLI subscriptions or supply their native tools. [Configuration](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/backend/config.py), [API client](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/backend/openrouter.py). + +Conversation files store stage outputs; the reviewed implementation has no evaluation-to-policy learning loop. The next question is sent without the stored conversation history. The first message adds a title-generation call, making the four-member initial exchange ten calls. [Storage](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/backend/storage.py), [Request orchestration](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/backend/main.py). + +## Adopt the deliberation pattern inside DevSquad + +| Adopt | DevSquad adaptation | +|---|---| +| Independent first opinions | Same frozen brief and evidence packet; participants cannot read each other's initial proposals | +| Peer critique | Criterion-by-criterion findings, source/test references and unresolved objections | +| Anonymous presentation | Hide profile metadata; store per-judge order and label mapping; counterbalance order in evaluations | +| One synthesis role | Existing host/headless lead explains the chosen approach and retains meaningful dissent | +| Inspectable stages | Persist proposals, critiques, decisions, failures and later outcomes as ordinary run artifacts | + +My recommendation is **selective Council mode** for architectural tradeoffs, competing debugging hypotheses and consequential disputed reviews. Routine delivery keeps implementer → independent reviewer → checks. Agreement is useful evidence to examine; it cannot override failing tests or prove a claim true. Research on LLM judges documents position, verbosity and self-enhancement biases, supporting calibration rather than treating peer ranks as ground truth. [LLM-as-a-judge study](https://arxiv.org/abs/2306.05685). + +```mermaid +flowchart TD + T[Task] --> G{Configured council trigger?} + G -->|no| N[Normal engineering workflow] + G -->|yes| A[Two independent proposals] + A --> B["One independent critic
Evidence and objections"] + B --> C[Existing lead selects / synthesizes] + C --> D[Decision artifact with dissent and validation plan] + D --> N +``` + +Proposed starting budget: two proposers, one critic, and one headless lead = **four worker invocations before DevSquad-level retries**. With a host lead, there are three worker invocations plus separately recorded host work. A native CLI worker can perform several model/tool turns; these units are not comparable to the original Council's nine API requests. This smaller deliberation protocol needs its own quality/capacity evaluation. Count launches using `max_worker_invocations`, recording native usage and context bytes separately. Council participants stay read-only; it never creates several concurrent implementation writers. + +### C1 — Optional Council extension after M6 + +**Dependencies:** M3–M6 through M6. C1 is a separate pending extension and does not block M7 or first usability. Reuse the runner, adapters, artifacts, routing, cancellation and evidence store. No additional service or UI platform. + +Add a feature-gated `council-decision` workflow after the two core workflows are proven. M1–M7 schemas keep the original workflow enum until C1 is implemented; expose supported workflows through doctor. C1 adds the new enum and its strict configuration schemas together, with a documented contract revision. + +- **Inputs:** Existing task/criteria/scope/budget plus a CouncilSpec: proposer candidate profiles, critic candidate profiles, evidence artifact IDs/hashes, rubric, minimum valid proposals (two), critic requirement (one), maximum council invocations and a reason for invocation. Proposer/critic profiles use the same pin/fallback semantics as ordinary roles. The normal lead remains the sole decision authority. Stage barriers and role-scoped filesystem/MCP artifact access withhold peer proposals until every initial proposal is finalized; independence cannot rely on prompt wording alone. +- **Selection:** Automatically choose two distinct verified model identities and a critic whose model identity differs from both authors. Prefer independent families where qualified, without inferring independence from harness names. If the minimum eligible set or budget is unavailable, report the council as blocked; do not manufacture a quorum. The ordinary engineering workflow remains separately available. +- **Critique:** No self-scoring. Hide identity metadata, retain raw provenance separately, and record reproducible randomized presentation order. Anonymity is partial because writing/content can reveal identity. Require structured criterion assessments and evidence references; validate missing/duplicate/unknown candidate IDs. Avoid forcing an overall numeric rank. +- **Decision:** Save supported claims, chosen proposal or synthesis, discarded alternatives, unresolved objections and required validation. The lead cannot turn majority preference into test acceptance. Failed providers, missing evidence and critic failure are explicit; there is no silent “consensus” when participants drop out. +- **Tools:** A common baseline evidence packet makes proposals comparable. Additional native research is allowed only by profile and declared scope, with its sources retained. Unequal evidence access is recorded as a confounder for model comparisons. Check code claims through the normal trusted verification path. +- **Triggers:** Start with explicit user/lead request and a capped policy allowance. Automatic triggering remains disabled until C1's comparison gate passes. Later policy may trigger on declared architectural decisions or unresolved evidence-backed review conflicts; model self-confidence alone is not a trigger. One council per decision by default; additional rounds consume explicit allowance. +- **Learning:** Compare normal workflow vs selective council on matched/held-out cases: accepted quality, escaped defects, lead rework, latency and observed allowance. Keep question, profile, effort, prompt and evidence versions. Council rank or agreement never directly promotes a model globally. + +**Acceptance:** Proposers cannot see each other's draft artifacts; critic cannot author-score; shuffled labels map back correctly. Fake fixtures cover missing proposer, missing critic, invalid judgement IDs, empty output, quota exhaustion, cancellation/resume and recorded dissent. A seeded wrong majority cannot override a failed mandatory check. A live bounded council yields an inspectable decision using existing subscription adapters. A predeclared held-out comparison records benefit, harm or inconclusive results; automatic use stays disabled unless a reviewed policy change is supported. Receipt includes every worker attempt, failed attempts and available native usage, not just the selected answer. diff --git a/docs/plans/engineering-team/SOL-HANDOFF.md b/docs/plans/engineering-team/SOL-HANDOFF.md new file mode 100644 index 0000000..de1750e --- /dev/null +++ b/docs/plans/engineering-team/SOL-HANDOFF.md @@ -0,0 +1,111 @@ +# Sol execution handoff — build, test and make DevSquad usable + +**Current continuation:** Start with +[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The September 29 review at +`f4fa657` reopened shared M3 acceptance integrity and independent M5–M7 work. +That plan supplies the current findings, ordered repairs and acceptance gates; +the full scope and constraints below remain in force. + +This is the full execution prompt for Sol. It is an implementation assignment; the underlying runtime is still pending at handoff creation. Copy this document into Sol, or ask Sol to read this file and execute it in full. + +## Objective and persistence + +Build the complete DevSquad engineering-team plan into a simple, reliable local product I can actually use from terminal, Codex App, Claude Code App, Antigravity and Grok Build. Treat this as a persistent implementation goal. Execute, test, repair and document it through completion; do not stop after planning, scaffolding, M1, the first successful demo or passing existing tests. + +Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. +Canonical build branch after cleanup: `codex/engineering-team` (local and GitHub). +Complete architecture and original handoff checkpoint: `ff1fa60`. +Latest architecture amendment before this handoff: `bfaa390`. + +Inspect current Git state and files first. Continue `codex/engineering-team`, or use an isolated Codex worktree based on that branch. Verify `git merge-base --is-ancestor ff1fa60 HEAD` and preserve existing work. GitHub `main` remains the published runtime baseline and does not contain this build plan. Do not restart from `main`, resurrect an archived branch or merge the unrelated February backup history. Read the [branch consolidation record](../../audits/2026-09-06-branch-consolidation.md), repository instructions and `CONTRIBUTING.md`. Use one integration branch for milestone checkpoints; remove temporary task branches/worktrees after their work is integrated and preserved. + +## Read the complete specification + +Read these files relative to the repository, in order: + +1. `docs/plans/engineering-team/START-HERE.md` +2. `docs/adr/ADR-002-surface-independent-engineering-team.md` +3. `docs/plans/engineering-team/CONTRACTS.md` +4. `docs/plans/engineering-team/IMPLEMENTATION.md` +5. `docs/plans/engineering-team/SELECTION-AND-COUNCIL.md` +6. `docs/plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md` +7. `docs/plans/engineering-team/backlog.json` and `docs/plans/engineering-team/examples/` + +Use `docs/audits/2026-09-06-engineering-team-assessment.md` for verified starting defects/history. ADR-001 and the old `.planning/` records are historical context; they do not supersede the September decisions or prove implementation completion. Recheck source and installed provider capabilities instead of trusting dated model examples. + +**Scope clarification:** This assignment includes M1–M7 and C1. Council is optional to invoke in the product, but its implementation and acceptance gates are included in this delivery. M3 is the first usable checkpoint, not the finish line. Later amendments override older conflicting details. The simple task-entry requirement below extends the earlier deferral of natural-language task preparation; it does not authorize a general workflow engine or a second planner. + +Resolve ordinary implementation choices yourself. Where current evidence requires a contract change, make the smallest coherent amendment and update affected schemas, docs and tests together. Do not restart the architecture exercise or quietly remove hard requirements to obtain green tests. + +## Deliver the whole engineering team + +- Implement the shared Python runner, transactional event/state store, isolated worktrees, durable jobs and common CLI/MCP service. Preserve legacy Bash 3.2 wrapper callers, jq-absent behavior and the four error prefixes. +- Use a verified native Codex app-server adapter, keeping legacy CLI compatibility. Support Claude, Antigravity and Grok through tested adapters. Discover actual models, efforts, tools and permissions from each installed harness; preserve requested versus observed settings and unknown values. +- Deliver branch review and bounded issue implementation → different-model review → correction → tests → lead disposition. Bind acceptance evidence to the exact candidate. Keep one writer per worktree, one lead per run and bounded retries. Implement ordinary and adversarial review distinctly. +- Make every listed local surface operate on the same saved runs. Closing a client must not lose work. Provide status, results, events, cancel, safe resume and fenced host handoffs. Native transcript import is a capability-gated convenience; portable artifact handoffs are required across hosts. +- Route automatically among eligible model/effort/tool profiles, with exact per-role pins and explicit fallback. Account for shared subscription pools, every applicable quota window, latency, unknown/stale observations and local concurrency. Maximize verified outcomes and useful capacity; do not force every provider into every task. +- Implement stable profile aliases, automatic catalog refresh, last-good snapshot retention, bounded qualification/trials, guarded promotion and rollback. Model discovery alone must not change defaults. Freeze run bindings and concrete pins. Ship/test `guarded_auto` as an opt-in after calibration; permission or billing changes stay outside it. +- Record all attempts, failures, fallbacks, reviewer contributions, lead repairs and later corrections. Generate receipts, handoffs, evaluation/decision reports and ongoing documentation. Separate worker invocations from native internal model calls and quota usage. Proposed improvements need comparable evidence. +- Implement C1 using independent proposals, a distinct critic and the existing lead. Preserve dissent, evidence and budget limits. Test manual Council use; keep automatic triggering disabled until its evaluation gate and policy authorize it. + +## Make ordinary use simple + +Ship a guided local setup that discovers existing installations/authentication, explains readiness and configures DevSquad's required local integration without asking me to maintain a model matrix. Preserve unrelated app settings and credentials. Provide a clean readiness report when a capability cannot be verified. + +Deliver this small human-facing command surface over the same service and contracts: + +```text +squad setup +squad doctor +squad review --base main +squad fix "the bounded issue to resolve" +squad council "the decision to evaluate" +squad status [RUN] +squad result [RUN] +squad cancel RUN +squad resume RUN +``` + +These are target commands to implement, not commands that exist at handoff creation. Preserve the specified low-level JSON/API operations for automation. Optional omitted run IDs resolve only when the current project has an unambiguous relevant run; otherwise present the choices. Never silently target an unrelated job. + +Normal review/fix/council entry must not require manually writing JSON. In an app, the current lead constructs the task. In terminal, use the configured single headless lead, where necessary, for one bounded task-preparation step against approved templates and allowed checks. Record that preparation, its budget and output. Validate criteria, scope and commands before execution; generated task text cannot expand permission or spending authority. Reuse the same planning authority rather than spawning a second lead. Retain committed-input constraints unless explicitly amended with equivalent tested snapshot guarantees. + +Show the selected roles/profiles, why they were chosen, progress, the run ID and a clear next action. Provide one verified command per normal operation. Errors should say what failed, whether work is still running and how to recover. Keep model IDs, raw schemas and protocol details out of the normal user flow unless they help resolve an issue. + +Do not add a dashboard, cloud service, arbitrary DAG builder or extra configuration layer merely to present these features. + +## Execute in small verified increments + +Follow the dependency graph. Demonstrate M3 early, then keep going through M4/M5/M6/M7 and C1. Delegate bounded implementation/review/test tasks when useful; isolate concurrent edits and keep ownership clear. A delegated failure does not remove the requirement: continue independent work and use available execution paths. + +Use offline fake CLIs/protocol servers for development and fault injection. Use existing authenticated subscription harnesses for bounded real smoke tests and the required live acceptance demonstrations. Recheck current official capabilities at integration time. Do not purchase credits, consume usage resets, silently switch to paid APIs, publish, push, merge, deploy, send external messages or change unrelated account settings under this assignment. + +Local implementation dependencies, an isolated DevSquad installation and required local MCP registrations are part of delivery. Make them reversible/idempotent and preserve unrelated configuration. Handle authentication through normal provider flows. If quota/authentication or an unavailable host blocks a live gate, record the exact blocker, finish all independent work, and ask only for the missing action needed to complete that gate. Do not hammer a limited provider or substitute fixture results for live proof. + +## Test behavior thoroughly + +Derive a requirement-to-evidence matrix before implementing. Preserve every milestone gate in that matrix; tests must establish behavior, not mirror code structure. Run the existing `bash test/run.sh` before every commit as required, plus meaningful new tests for the changed behavior. The historical 177 assertions are a regression baseline, not evidence that the new product works. + +Required coverage includes: + +1. **Adapters:** argv/path handling including spaces and TSX; model/effort validation; auth/rate/timeout classification; empty/malformed/denied exit-0 results; native protocol events, disconnects and interruption; permissions and recursion guards. +2. **Durability:** concurrent idempotent starts, conflicting bodies, transactional events/artifacts, migrations, supervisor crash with a live child, reused PID protection, no duplicate writer, cancel/reap, restart/resume and stale host claims. Use real controlled subprocesses where fake clocks cannot prove cleanup. +3. **Engineering outcomes:** seeded defects detected by independent review, bounded repairs, mandatory failing tests blocking acceptance, report-only failures remaining visible, stale-patch evidence rejection, scope enforcement and preservation of the user's checkout. +4. **Routing/capacity:** deterministic selection from identical snapshots; exact pins/fallback; two projects sharing one pool; short and weekly windows; stale/unknown data; external account consumption; blocked paid-API fallback; native usage versus worker-launch counts. +5. **Model lifecycle/learning:** pagination, incomplete discovery preserving last-good data, affected-profile revalidation, same-ID uncertainty, qualified promotion, insufficient evidence, new-run-only binding changes, rollback, failed attempts later repaired and escaped-defect feedback. +6. **Council:** isolated first proposals, no author judging their own proposal, valid label mapping, missing participants, invalid critiques, recorded dissent and votes unable to override objective failure. Run the predeclared comparison and keep automatic use disabled when evidence is inconclusive. +7. **Packaging and real hosts:** fresh standalone install without Claude, plugin package contents, reinstall/update without duplicate hooks/MCP servers, active runs surviving package/client changes, and actual operation from terminal, Codex App, Claude Code App local Code tab, Antigravity and Grok Build. Parsing a config or passing an MCP unit test does not prove an app integration. + +Run one genuine bounded issue through at least two harnesses with a different verified review model. Save exact candidate/check/review evidence. Exercise a real cross-surface start → observe → handoff/finish and a cancel/recovery flow. Each provider adapter and each named local surface needs its own supported-operation smoke receipt; do not force all providers into one job to satisfy that coverage. + +Do a fresh-install usability walkthrough using only the quickstart and normal commands, without hand-editing task JSON. Fix setup friction, misleading status, confusing errors and documentation commands that fail. Obtain an independent implementation/UX review when available and resolve actionable findings; do not claim an independent review that did not occur. + +## Completion and handoff back to me + +Update `backlog.json` after each verified checkpoint with revision, command/action, result and portable redacted evidence. Keep a requirement matrix and concise implementation status beside it. Commit small verified increments, never stash work, and end with a clean tree. Keep secrets/raw private task content out of tracked evidence. + +Before declaring completion, audit the full requested scope against the actual installed product. Fix failures and rerun the affected checks. A partial implementation, unavailable live gate or unsupported required host is incomplete even if offline tests pass. Record precise residual blockers rather than weakening the gate or marking it done. Do not wait indefinitely when no process is live; continue remaining independent work. + +Deliver working code and local setup; reproducible tests and live receipts; an updated architecture/contract reference; a concise quickstart with one workflow visual; and a short recovery/troubleshooting guide. My final summary should state what works, exact commands to start using it, the evidence location, remaining limitations and the implementation branch/commits. Prefer a compact readiness table over a long narrative. + +Start by inspecting the current state and implementing the earliest unmet dependency. Continue until this whole assignment is complete or the remaining requirements are explicitly blocked by external state that you cannot resolve. diff --git a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md new file mode 100644 index 0000000..8916716 --- /dev/null +++ b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md @@ -0,0 +1,610 @@ +# Sol execution plan and review feedback — October 1, 2026 + +## Assignment and starting point + +Continue the existing DevSquad build on `codex/engineering-team`. Repair the +review findings below, finish the missing runtime connections, and prove the +requested product through the public service and installed normal commands. +This is a continuation of [SOL-HANDOFF.md](SOL-HANDOFF.md), not a redesign. +The full delivery remains M1–M7 plus C1; the immediate engineering priority is +saved-run evidence and public runtime integration in R3–R6. C1 is optional to +invoke but required to implement under the original assignment. + +Review baseline: `f4fa6577e2e891151231c9c7d3180be6e9e23faa`. Compare Git state +before acting and retain later work. The review made no source changes. +Installed release at review: +`0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef`, with no detected payload drift. + +Read `AGENTS.md`, `CONTRIBUTING.md`, [RESUME.md](RESUME.md), this document and +[backlog.json](backlog.json) first. Load the relevant contract/specification +for each work package. Do not reload the historical chat or restart M1/M2. +This review supersedes the earlier claim that only credentials remain. + +Current inspected checkpoint: **`b766f9e` on `codex/engineering-team`**, with a +clean tree before this planning update. The latest implementation is +`672e383`; preserve it and all later work. **R1, R2 and R3a are source/offline +verified; start at R3b, not R1.** R3 as a whole, R4–R8, C1 and affected +installed/live proofs remain open. See +[R1 evidence](evidence/R1-candidate-integrity-2026-09-29.json), +[R2 evidence](evidence/R2-observed-identity-2026-10-01.json) and the R3a record +below. The installed release named above predates these repairs. + +R3a source/offline checkpoint at `672e383`: outcome uniqueness, the v2 provenance +contract, separate corpus/pair hashes, schema-14 fenced prelaunch assignments +and normalized-chain checks are verified by 47 focused tests, migration checks, +the 402-test complete gate (two optional-SDK skips) and bounded independent +review/follow-up. The SQLite warning remains R5. See +[R3a evidence](evidence/R3a-provenance-contract-2026-10-01.json). +The saved-run v2 outcome reader and +shared lifecycle eligibility are still R3b; do not describe this partial +checkpoint as trustworthy end-to-end qualification or a public trial runner. + +This October 1 follow-up is a **source review and planning handoff**, not +another implementation or live verification. The recorded 402-test result +belongs to `672e383`, not a new run for this document. Two optional-SDK skips +and the SQLite finalizer warning remain explicit. No installation, login, +provider request or model-setting change is part of this planning pass. + +### Ready-to-execute handoff for Sol + +Do not reimplement R2. It now has a shared strict native parser, v2 imported +evidence, actual-attempt profile binding, actual-model independence and durable +failure/fallback/cancellation/history handling. The original 16 worker tests, +12 import/parser tests and nine public identity tests cover this repair. +Two independent-review findings were reproduced and repaired: another allowed +fallback profile cannot impersonate the actual attempt, and invalid UTF-8 or +CRLF native output retains its exact byte hash. The original +[red baseline](evidence/R2-identity-red-baseline-2026-09-29.json) remains history. + +Preserve the verification history: a prior 367-test run failed six cases; +all six passed unchanged with stable clocks, and the final full run passed +with UTC and monotonic elapsed times agreeing. Do not erase the failed run or +weaken budgets to make tests pass. The planning handoff at `b3de5c6` made no +implementation, installation or provider call; subsequent R3a source work is +recorded above. No installation or provider call was needed for that slice. + +| Next slice | Deliverable | Gate before claiming completion | +|---|---|---| +| R1 / R2 | Preserve verified source repairs | Recheck affected regressions when shared code changes; installed/live proof remains R8 | +| R3a — verified | Preserve provenance contracts and fenced assignments | Existing contract/store tests cover this foundation; normalized unit witnesses do not prove actual saved-run evidence | +| R3b | Durable validation and current-evidence eligibility | Replay, qualification, promotion, rollback and catalog fallback cannot reuse stale/unverified evidence | +| R3c | Public regression and migration evidence | Valid independent pairs work; legacy receipts remain readable; late corrections block new unsafe decisions | +| R4 → R5 → R6 | Routing, learning and normal terminal integration | Public-run evidence through the full chain, then fresh-install usability proof | +| R7 / R8 | Council and installed/live closure | Each separately required acceptance gate; external blockers remain explicit | + +Use the same spec-based loop for every slice: name the contract and public +entry point; reproduce the missing behavior; implement the smallest coherent +change; run focused tests; review the patch; record evidence and the exact +next action; checkpoint. Run the full integration gates at package boundaries, +not after every small edit. One integration owner controls shared files; use +bounded leaf reviews only when they can run alongside useful local work. + +## Feedback on what has been built + +Preserve the substantial working implementation: the durable runner and +ledger, process ownership and recovery, isolated worktrees, fenced host +handoffs, Codex protocol repairs, immutable installer and MCP integrations. +The successful installed Codex review receipt is real and remains valid for +that exact run. Its SHA256 was rechecked in the September 29 review: +`0103db19a528cff916ebef80c6ec94b682ed4801fc22db020cdbc21796040f70`. + +The main problem is the strength of completion claims. Several components +were marked complete because their unit fixtures passed, even though the +normal command path bypasses them or cannot produce their input evidence. +Passing counts do not demonstrate that the complete workflow enforces its +contract. Each new completion claim needs an actual runtime caller, a public +path regression and the corresponding installed/live proof where required. + +During the original September 29 review, the complete Python suite ran 317 tests successfully +(two optional-SDK skips), and all 227 Bash assertions passed. The generated +reference and installed-source comparison passed. A recurring unclosed +SQLite `ResourceWarning` appeared despite process exit zero; it remains an +open cleanup issue. Initial sandbox-restricted runs were interrupted because +process inspection was denied; the successful reruns had the process access +needed by the cancellation tests. Do not count the interrupted runs as passes. + +### Findings and evidence limits + +Source locations refer to the baseline above and will move as repairs land. + +| ID | Priority / affected scope | Observed behavior and evidence | Repair package | +|---|---|---|---| +| F1 | P1 / M3, M5 | `review_worker.py:169` never revalidates candidate source after checks. A real-Git direct worker/evaluator reproduction froze `VALUE='wrong'`, changed it to `fixed` in a required check, then returned `passed` and `accept_allowed:true` while the frozen candidate stayed wrong. The full service reproduction was environment-blocked, so terminal service acceptance is not claimed. Violates CONTRACTS §5. | R1 | +| F2 | P1 / M5 | `claude_delivery_worker.py:256` copies requested model/effort into observed identity and labels it verified. A mocked result reporting `different-model` still yielded requested-model/low/verified; absent model identity also passes. `workflows.py:572` checks only that identity is a dictionary. | R2 | +| F3 | P1 / M6 | `learning.py:396–414` checks unique case IDs but permits reuse of outcome IDs. A confirmed evaluator reproduction reused two outcomes across two evaluation cases and one held-out case and returned `promotion_proposal`. Qualification reads those inflated counts. | R3 | +| F4 | P1 / M6, M7 | `task_entry.py:136` selects the provider default or first model; `_managed_routing` uses singleton concrete profiles and empty bindings. `store.py:2588` skips lifecycle overlay without aliases. Catalog default changes can change normal routing without qualification. | R4 | +| G1 | Required integration / M6 | The only production read of `experiment_assignment` is in `store.py:1522`; there is no writer. Public runs cannot supply the experimental outcomes required by the evaluator. `record_outcome` is reached through manual `outcome_add`, not ordinary terminal completion. | R5 | +| G2 | Required integration / M6 | Last-good catalog persistence has no production caller. Normal discovery has no scoped persistent refresh/cache path. Native Codex quota observations are not ingested; observations require manual input. | R4 | +| G3 | Required usability / M7 | Normal tasks always use `lead.mode=host`; terminal completion requires manual claim/decision JSON. `status`/`result` require an ID. Doctor checks binaries/registrations, not provider authentication. All CLI responses are JSON even without `--json`. | R6 | +| G4 | Verification coverage / M5, M7 | Default check detection on DevSquad selects only `bash test/run.sh`; the Python core suite is omitted, including when the requested fix changes Python core code. Detection also inspects the current checkout rather than the selected target tree. | R6 | +| G5 | Remaining full-delivery scope / C1 | `squad council` is absent. Its implementation and acceptance are included in the full assignment. | R7 | + +The table records the original review baseline. G1's missing internal writer +is now partly addressed by R3a: `Store.complete_preparation` freezes schema-14 +assignments before attempts. The missing **public controller** and automatic +terminal-outcome projection remain R5; reuse the new writer, do not rebuild it. + +The optional decision helper currently executes only a fake adapter; +non-fixture requests record `unavailable`. This is consistent with the narrow +M6-D1 contract, not proof of a production Jev integration. The one-request +Jev pilot is separate from runtime adoption. Do not enable a classifier or +claim routing improvement from a synthetic smoke result. + +## Work order and ownership + +R1 and R2 source repairs are done; R3 is the next dependency. R4 precedes R5; R6 integrates the repaired public +paths. R7 follows the relevant M3–M6 repairs. R8 proves the installed product; +its authentication and classifier subgates may remain externally blocked. +Continue all independent work when a live subgate is blocked. + +Use one coordinator and small commits. A bounded independent review may run +alongside non-overlapping work, but do not recursively spawn agents or run +multiple full suites concurrently. `service.py`, `store.py`, `workflows.py` +and the ledger documents require a single integration owner. Prefer focused +offline regressions before a full gate and one justified live probe. + +### R1 — Preserve candidate identity through trusted checks + +**Files:** `review_worker.py`, `workspaces.py`, `workflows.py`, and review/ +delivery runtime tests. Shared M3/M5 fix; no provider dependency. + +1. Turn F1 into a regression through the public service for both review and + delivery. Retain the exact pre-check candidate, check output and mutation. +2. Revalidate HEAD/index/tracked content and candidate inputs across each check + boundary. A tracked edit, deletion, mode change or checkout cannot provide + passing evidence for the original candidate or contaminate the next check. + Distinguish ordinary permitted build outputs from candidate source changes. +3. Make the integrity failure visible in evaluation, handoff and terminal + reports; both host and headless disposition must refuse acceptance. + +**Acceptance:** source-changing check cannot accept; a second check cannot +validate an unnoticed changed tree; ordinary non-source build artifacts and +clean checks work; the original checkout is unchanged; recovery/replay cannot +restore invalidated evidence. Preserve per-check temporary HOME behavior. + +### R2 — Observe and validate Claude execution identity + +**Files:** `claude_delivery_worker.py`, `workflows.py`, adapter fixtures and +delivery tests. Implement conformance offline before spending a live turn. + +Investigation checkpoint (September 29): installed Claude `2.1.220` exposes +`session_id` and `modelUsage` in native result JSON. Preserve all reported model +entries; `canonicalModel` is pricing normalization, not serving-model proof. +The result does not report effective effort, so keep it unknown rather than +copying `--effort`. Multiple entries do not identify a unique writer. Pin the +parser to tested local capability; current online docs can describe newer CLI +features. Primary references: [result schema](https://code.claude.com/docs/en/agent-sdk/python#resultmessage), +[model aliases and effort limits](https://code.claude.com/docs/en/model-config), +and [usage attribution](https://code.claude.com/docs/en/agent-sdk/cost-tracking). +Do not replace subscription-compatible safe mode with `--bare`. + +1. Verify the supported CLI's native model/session/usage fields against its + installed schema/help and current official documentation when necessary. + Record requested settings separately from actual reported identity. +2. Parse effective model identity, including alias resolution and multiple + model entries where the CLI reports them. Preserve unavailable effort or + backing revision as unknown. Never manufacture an observed setting from + the requested argv. +3. Validate imported identity and enforce the existing independent-review + contract. Missing, contradictory or insufficient identity evidence must + not qualify as verified independence. Keep failed evidence in the receipt. + +**Acceptance:** matching exact identity, alias-to-effective model, unexpected +model, absent identity, multiple models, unavailable effort and tampered +evidence all have explicit tested results. The genuine two-harness live gate +remains open until the corrected adapter obtains a real receipt. + +#### R2 implemented contract to preserve + +The following requirements are implemented at `ef98889`, not a new to-do +list. Keep them enforced when R3–R6 touch shared import/disposition paths. + +- **R2a — parser/worker:** define a shared strict identity contract rather than + letting the worker and importer invent separate interpretations. Require a + valid success envelope, bounded nonblank session identity and well-formed + native usage. Reject non-object, duplicate-key and non-finite JSON without + leaking an `AttributeError`. Keep the frozen requested profile unchanged. + A sole concrete `modelUsage` key may establish the reported model; preserve + all entries and reject ambiguous writer attribution. Tested family aliases + (`sonnet`, `opus`, `haiku`) may resolve only to a reported concrete model of + that family, without a hardcoded current version. Pricing `canonicalModel` + must not replace native usage identity; a differing pricing-only value is + allowed and is not a serving-identity conflict. An optional top-level + `model` is not identity authority either, but a contradictory serving-model + claim fails closed. Keep effective effort/backing revision + null when unreported, and make the scope of verification explicit. The new + worker test file covers this layer only; import and public runtime suites + provide the additional coverage. Fixtures must contain valid native evidence. +- **R2b — import/acceptance:** validate the native evidence against the frozen + adapter, selected profile (including a legitimate fallback), session and + claimed observation. A dictionary or `verification="verified"` label is not + proof. Add regressions for forged model/effort/session/provenance and missing + native evidence. Enforce independence using verified reported identities, + not just requested aliases or family labels. Exercise same effective model + behind different profile names, unknown/ambiguous identity and a valid + different-model case through both host and headless delivery. Candidate + finalization/import and later disposition must not bypass these gates. +- **R2c — failures/recovery:** persist bounded, redacted identity diagnostics, + valid reported model/usage data and artifact references for rejected attempts + in the durable receipt; exception text alone is insufficient. Test failure, + fallback, cancellation and recovery/replay with no duplicate writer or lost + attempt. Historical terminal receipts remain readable and exact terminal + replays remain idempotent, but old unverified evidence cannot authorize a + new acceptance. Version evidence/contracts explicitly if needed. Preserve + R1's non-overridable check-integrity gates throughout. + +Run the preserved worker regression set with: + +```bash +PYTHONDONTWRITEBYTECODE=1 PYTHONPATH=plugin/core/src \ + python3 -m unittest discover -s test/core -p test_claude_identity.py -v +``` + +This command is now expected to pass. Its historical red result belongs to +`6848f11` and the saved baseline artifact. The added public tests, full core +gate and independent patch review have completed; do not substitute these +offline results for the separately blocked installed/live receipt. + +### R3 — Make experiment and held-out evidence independent + +**Files:** `learning.py`, experiment/qualification paths in `store.py`, and +learning/lifecycle tests. No key or provider required. + +1. Reject duplicate outcomes across cases/arms/splits. Persist enough frozen + case, run and split provenance to prevent the same underlying observation + being relabelled to inflate sample size or contaminate held-out evidence. +2. Validate that each arm belongs to its declared project, experiment, + concrete profile fingerprint and paired input contract. Review arms require + the same candidate; implementation arms require the same baseline, task + and checks. An experimental label alone is insufficient to establish a + controlled one-variable comparison. +3. Enforce these checks before evaluation persistence and qualification. + Audit any saved qualifying evaluations affected by the bug; preserve them + as historical evidence and prevent unsafe future use rather than deleting + or silently rewriting receipts. Late escaped-defect corrections must + trigger an explicit evaluation/qualification revision or review policy, + not leave stale success silently eligible for future promotion. + +**Acceptance:** the two-outcome/three-case reproduction is rejected; repeated +run evidence, crossed split membership, wrong-profile arms and mismatched +tasks fail; valid disjoint pairs qualify; failures/missingness remain visible; +replay cannot bypass the repaired gate. Promotion and rollback use the same +validated evidence path. + +#### R3 execution slices and review feedback + +Source inspection at `ef98889` confirms a broader problem than duplicate IDs: +`learning.py:343` accepts profile IDs without fingerprints; +`store.py:1719` loads outcome payloads without run/snapshot/attempt provenance; +evaluation replay (`store.py:1696`) and qualification replay +(`store.py:1994`) return before current evidence validation. Fix the complete +trust path, not only the sample counter. + +1. **R3a — contract and red tests.** Define a versioned frozen assignment: + experiment/spec hash, project, case, split, arm, tested role, concrete + profile hash and paired-input hash. Reject global outcome reuse, underlying + run reuse and evaluation/held-out input overlap. Review arms must share + the candidate; implementation arms must share baseline and normalized + task/scope/acceptance/checks. Explicitly define which incidental paths and + tested routing field are excluded from paired-input hashing; do not drop + substantive inputs to obtain a match. +2. **R3b — saved evidence and shared eligibility.** Join outcomes to saved + runs, project identity, frozen snapshots and actual role attempts. The + assignment/profile must agree with actual execution, including fallbacks. + Persist final-outcome and correction-chain hashes in a versioned evaluation. + One shared current-evidence gate must protect qualification, evaluation/ + qualification replay, new promotion, regression rollback and unavailable- + profile catalog fallback. A new correction invalidates old eligibility; + require an explicit new evaluation/review, not silent receipt rewriting. +3. **R3c — compatibility and public proof.** Preserve historical v1 records + and exact completed-decision replay without another binding mutation. + Label unverified historical evidence ineligible for new decisions. Replace + positive tests that SQL-terminalize empty runs with distinct saved runs, + frozen assignments and bound actual attempts. Keep R5's public trial + controller as a separate deliverable; R3 must not claim it already exists. + +Use this regression order: two outcomes reused across three cases; cross-arm +and underlying-run reuse; wrong project/experiment/case/split; same profile ID +with a different fingerprint; wrong actual fallback; mismatched candidate or +baseline/task/checks; valid disjoint pairs; failed/missing arms; atomic +persistence and replay; legacy read versus new decision; late escaped defect +after evaluation and after qualification; valid promotion/rollback and stale +catalog-fallback rejection. Record independent pairs, not the number of labels. + +R3 can use the existing run IDs, project Git common directory, request hash, +frozen task/config hashes, candidate/baseline identity, selected profiles and +attempt profile indices. Do not trust caller-supplied `experimental` labels +as proof. Failed and missing evidence must remain visible and ineligible for +unearned success credit. Keep schema/contract changes and migrations together. + +#### October 1 feedback: exact R3b/R3c implementation boundary + +The following gaps were rechecked in source at `672e383` / checkpoint +`b766f9e`. They are unfinished integration identified by this review, not new +claims of live exploitation or a reason to repeat the completed R3a suite. +Paths below are under `plugin/core/src/devsquad/` unless stated otherwise. + +| Current path | Concrete remaining gap | Required change | +|---|---|---| +| `store.py:1763`, `evaluate_learning_experiment` | Loads only outcome payloads; passes neither saved project identity nor normalized provenance required by v2 | Construct chains from immutable assignments and actual saved attempts; never accept a caller's witness as authority | +| `store.py:1780` and `store.py:2073` | Evaluation and qualification replay return before current-evidence checks | Separate immutable historical receipt from present eligibility; replay must not restore eligibility lost to corrections | +| `store.py:2114`, `record_profile_qualification` | Matches candidate by profile ID and alias, without binding the full tested profile/execution evidence | Check exact candidate fingerprint and the declared role/task context against validated saved evidence | +| `store.py:2248`, `store.py:2309`, `store.py:2532` | Promotion, regression rollback and catalog fallback rely on saved verdicts/qualification labels | Use the same current-evidence gate for every new decision, within the mutation transaction | +| `learning.py:690`, `build_learning_proposal` | Strict new-spec validation also parses historical v1 records; an old reused-outcome record can fail before a readable audit result | Preserve original bytes/hashes and report explicit historical-ineligible status without allowing new decisions | +| `test/core/test_learning.py:226`, `test/core/test_lifecycle.py:198` | Existing positive fixtures set terminal state directly with SQL | Replace positive authority proofs with distinct prepared runs, actual fixture attempts and public completion; retain SQL only for explicitly labelled migration/tamper negatives | + +**R3b.1 — authoritative saved-run reader.** Keep `learning.py`'s evaluator +pure and make Store responsible for constructing its trusted inputs. In one +consistent transaction, verify saved spec/assignment hashes and project +identity, recompute the original pair/corpus/execution fingerprints from +`experiment_assignments.snapshot_json` and its package digest, and bind every +relevant role attempt to that declaration. Include failed attempts, retries, +fallbacks and repairs, not only the eventual successful worker. An actual +different-profile fallback cannot be credited as the declared arm. Reconcile +durable attempt profile/index/package and frozen adapter identity; retain +native observed evidence without manufacturing missing identity. Preparation +must precede attempts by the existing fence/order, not timestamp comparison. +No attempt or missing partner is missing evidence, never an invented exposure. +Verify and hash final outcomes plus ordered corrections from saved rows. + +First acceptance: a minimum qualifying set of disjoint public offline runs +can be evaluated through Service/Store without supplied provenance; changing +the saved actual profile, assignment, package or paired input fails atomically, +including when the other arm is missing. A test-only preflight seam may attach +the predeclared manifest until R5 supplies the public controller, but it must +still use fenced preparation, real fixture attempts and public disposition. +Document that seam: it does not prove normal public experiment entry exists. + +**R3b.2 — one eligibility gate and explicit revisions.** Define one reusable +current-evidence result with concrete ineligibility reasons, bound to spec, +evaluation and evidence hashes. Keep historical decoding separate. Protect +new qualification and binding mutations against a correction arriving between +validation and commit; do not run a read check and then write from stale data. +Qualified profile fingerprints must match the tested candidate, with role and +task-class applicability checked rather than inferred from the alias alone. + +Settle the correction/re-evaluation contract before implementation: +`migrations/011_experiments.sql` permits only one evaluation per experiment ID, +and preparation already freezes that spec. A correction therefore cannot be +handled by silently overwriting its evaluation or repeatedly replaying the +same stale receipt. Use an explicit append-only evaluation/review revision +referencing the original spec and evidence, with qualifications/decisions +pinning the reviewed revision/hash. Update contract, migration and public +response semantics together. Do not reassign old runs to a new experiment to +make corrected evidence appear to be new independent samples. + +| Consumer | Acceptance for current evidence | Acceptance after a late correction / for unverified v1 history | +|---|---|---| +| Evaluation and its replay | Saved-run-derived evidence and deterministic hashes | Preserve receipt; expose stale/ineligible status or a specific error, never silently re-authorize it | +| Qualification and its replay | Exact candidate/context, measured pairs and current evidence match | Cannot return an unqualifiedly current `qualified` result; require explicit review/revision | +| New promotion | Validated qualification plus current evidence, in the same write transaction | Block without incrementing binding version | +| New regression rollback | Validate both regression evidence and any qualification-backed target | Stale/legacy regression or target cannot authorize the change | +| New catalog fallback | Select only a currently eligible, available predecessor | Skip stale qualification-backed predecessors; block if none remain | +| Report / proposal / historical audit | Show evidence scope and current eligibility | Remain readable; no promotion proposal from historical-ineligible evidence | +| Exact completed binding-decision replay | Return the original receipt without another mutation | Still return historical receipt; do not retroactively undo the completed decision | + +Preserve the existing explicitly proven bootstrap-baseline contract for a +predecessor without a qualification record; do not invent an experiment for +it or treat any arbitrary legacy profile as proven. A new correction blocks +future unsafe use; it does not itself authorize automatic binding changes. + +**R3c — public and upgrade proof, then close R3.** Cover branch-review and +issue-delivery input identities, honest failed/missing arms, correction after +evaluation and after qualification, valid promotion/rollback, rejected stale +catalog fallback, and repeated completion with no duplicate state mutation. +Upgrade a historical database and exercise actual report/proposal/read paths, +including previously saved duplicate-outcome evidence, not only SQL byte +preservation. Positive fixtures must meet the original sample/budget gates; +do not lower those gates or replace a valid-success test with rejection to +make the suite green. Run focused tests per change, one full integration gate +at the coherent R3 boundary, then a bounded independent review. Record R3 +closure separately from R5 controller and R8 installed/live acceptance. + +### R4 — Connect normal routing, catalog lifecycle and quota + +**Files:** `task_entry.py`, `catalog.py`, `capacity.py`, `router.py`, +`service.py`, `store.py`, adapter metadata and relevant tests. + +1. Resolve ordinary roles through approved stable aliases and existing + policy. Preserve explicit concrete pins, their pinned-selection provenance + and qualified fallbacks. A normal `--model` or `--review-model` selection + must not be reported as automatic merely because task preparation embedded + it as a singleton profile. Bootstrap + without evidence must remain an explicitly labelled bounded trial and + cannot become a proven default through catalog ordering or provider hints. +2. Connect discovery to account/config/version-scoped last-good snapshots, + TTL, a refresh lease, bounded timeout/backoff and existing protocol pagination. + Reuse `codex_protocol.py` pagination rather than replacing it. Wire complete + drift into affected-profile revalidation and qualified fallback/block. + Incomplete discovery must preserve the incumbent and prior snapshot. +3. Normalize documented native Codex quota observations into existing typed + pools/windows, with timestamps, confidence and unknown values preserved. + Use the existing reservation fence for launch decisions and keep manual + observations available for unsupported providers. + +**Acceptance:** changed provider default does not change an approved alias; +approved promotion affects new runs only; exact pins and old runs stay fixed; +partial/auth-failed discovery does not remove models; two projects honor +fresh weekly exhaustion despite short-window availability; stale/unsupported +quota remains unknown; no permission or billing expansion occurs. Exercise +these through normal task entry, not only router/store helper calls. +Include concurrent refreshes (one lease owner), interrupted refresh, and +account/config/version switches: incompatible scopes cannot reuse each +other's catalog or quota, and failed refresh preserves the last-good incumbent. +The existing normal profiles are already labelled `trial`; the defect is +bypassing stable alias/qualification policy, not a missing trial label. + +### R5 — Feed learning and trials from actual saved runs + +**Files:** terminal transition/report paths, `service.py`, `store.py`, +`learning.py`, `lifecycle.py`, migrations if needed, and public service tests. + +1. Project objective terminal facts into an idempotent final outcome record, + including failed attempts, repairs, disposition and evidence references. + Keep subjective later corrections explicit. Recover safely from a crash + between terminalization and outcome projection without duplicate records. +2. Freeze the experiment, cases, splits, profiles and budgets before either + arm launches. Reuse R3a's schema-14 assignment writer and R3b's saved-run + reader/eligibility; do not introduce another assignment authority. Run bounded paired + trials through the existing runner, account reservations and experiment + budgets. Derive consumed budget from durable attempts, not imported counters. + Carry assignment, case/split and profile provenance into outcomes. Use a + thin bounded controller, not a new general-purpose scheduler. +3. Connect public reporting, evaluation, qualification and proposal generation + to those outcomes. Trace and fix the SQLite cleanup warning without merely + suppressing it. + +**Acceptance:** a public offline run completes and appears in `report` without +manual outcome import; real public fixture runs yield evaluator-eligible +paired evidence without direct SQL state fabrication; a failed original +attempt repaired later is not credited as independently successful; late +correction, crash/replay, budgets and promotion/rollback all remain truthful. +Exercise the complete public chain: freeze experiment → run both arms → record +terminal outcomes → evaluate → qualify → promote for a new run → rerun held-out +cases → roll back. Count independent completed pairs, not labels or retries. + +Use branch-review runs over one already-frozen candidate for reviewer trials, +and issue-delivery runs over one baseline/task/check contract for implementer +trials. Delivery preflight has no produced review candidate yet; assigning two +whole delivery runs as a controlled reviewer comparison would not establish +the same candidate. Assignment must still precede every attempt. + +Test outcome projection from each terminal origin: preparation failure/cancel, +worker failure, host completion, headless completion, and crash/replay around +the projection boundary. All yield exactly one objective final outcome, with +missing attempt/criteria evidence explicitly absent. A prelaunch failure must +not become a completed experiment exposure. Do not project success merely +because one successful terminal path passes. + +### R6 — Finish the normal terminal and readiness experience + +**Files:** `cli.py`, `task_entry.py`, `diagnostics.py`, integrations, runtime +guide and public CLI tests. Use the existing service and single lead authority. + +1. Provide a supported terminal path from normal review/fix entry through + disposition without hand-written JSON, using the configured existing + headless lead or an explicit guided host path. Keep low-level JSON APIs. + Resolve omitted status/result IDs only for an unambiguous current-project + run; otherwise show choices. Supply readable output and a real next action. +2. Separate installed/registered/supported/authenticated/operation-verified + readiness. Use non-generating auth checks where supported and report + unknown otherwise. An unverified CLI version must not count as supported + adapter readiness. Missing auth must not advertise an unavailable workflow + as ready. Keep required logins in the user's normal provider flow. +3. Discover checks from the selected committed target and approved project + contracts. For DevSquad, include the Bash suite, relevant Python core suite + and generated-reference check. Do not guess unittest from a directory + named `tests` in an arbitrary non-Python project. Keep scope/check approval + bounded and do not let generated preparation expand permissions. + +**Acceptance:** fresh temporary install plus normal CLI review/fix reaches an +offline terminal receipt without manually authored task/decision JSON; zero, +one and multiple current-project runs behave correctly; unrelated runs are +never selected; Claude logged out is visible; target-ref check discovery is +stable; a Python defect cannot pass the DevSquad default delivery gate through +the Bash-only path. Update the quickstart to the verified commands. + +### R7 — Complete C1 within the existing runner + +Read [SELECTION-AND-COUNCIL.md](SELECTION-AND-COUNCIL.md) before this package. +It is separate from M7; its offline implementation can proceed while live +authentication gates wait. No new service, dashboard or multiple writers. + +1. Add the gated workflow/schema and `squad council` normal entry with two + distinct verified proposer identities, a critic distinct from both, and + the existing sole lead. Apply shared profile/capacity/budget rules. +2. Enforce independent proposals through stage barriers and artifact access, + then structured critique with validated labels/order and retained dissent. + Persist every attempt and support cancellation, recovery and missing quorum. +3. Run adversarial offline fixtures and the predeclared matched/held-out + comparison; obtain the required bounded subscription live receipt when + eligible identities are available. Keep automatic triggering disabled + unless its separate evidence and reviewed policy authorize it. + +**Acceptance:** no peer draft access or self-scoring; missing/invalid critic +cannot become consensus; label shuffling is correctly reversed; a wrong +majority cannot override failed checks; budgets and restart preserve history. +Inconclusive comparison is recorded as inconclusive, with automatic use off. + +### R8 — Install, prove live gates and audit closure + +1. Install the repaired immutable release and run the quickstart walkthrough, + dependency/drift/idempotence checks and required installed-SDK tests. + Independently review the repaired paths. Preserve previous releases and + saved receipts; a new release does not erase earlier failures. + **Before updating the real installation**, prove safe upgrade behavior + with active or recoverable old-package runs in a temporary install: + compatible coexistence across a real schema migration, or explicit upgrade + deferral until those runs are safely reconciled. Test continuation, cancel + and recovery for the selected approach. Frozen daemons load their old package + (`service.py:963`, `detached.py:84`), while `Store.migrate` rejects a newer + schema (`store.py:175`). The existing installer survival fixture changes + package versions, not schema (`test/core/test_install_core.py:58`). This is + a source-backed compatibility risk and missing test, not a reproduced live + failure. Cover schema 13 → 14 and any R3b evaluation-revision migration; + implement safe compatible coexistence or an explicit non-destructive + deferred upgrade, without stranding active runs or relaxing schema checks. + Deferral is a documented safety state, not proof of cross-schema + coexistence or completion of the installed-upgrade gate. +2. After normal Claude login, prove the M4 real-host handoff and one bounded + installed Claude implementation → verified different-model Codex review → + mandatory checks → disposition. After Grok login, prove its supported + operation. Recheck affected Codex/Gemini surfaces when their paths change; + unchanged historical receipts are not a reason for repeated model calls. +3. Handle the classifier gate under its existing authorization: if the key + becomes available, run the prepared synthetic Jev request once, no retry, + with a $0.01 maximum and verified current pricing. Run pinned local Laya + only on the declared trigger. A larger hosted comparison needs separate + authorization. Audit every requirement and report measured limitations. + +**Acceptance:** requirement-to-evidence matrix contains exact revisions, +commands, outcomes, candidate/receipt hashes and verified scope. M5/M6/M7 and +C1 close only against their own required gates. External failures remain +explicit; keep-off is a valid classifier adoption outcome, not missing-proof +permission to claim routing improvement. + +## Verification and delivery discipline + +- First write a behavior regression that fails on the reviewed baseline, + then implement the repair. Test public transitions, not only dictionary + constructors or helper output. Keep fixture evidence distinct from live. +- Run relevant Python tests for each code slice and `bash test/run.sh` before + every commit. Run complete Python and generated-reference gates at coherent + integration boundaries. Do not repeat unaffected suites just to accumulate + counts. Use the installed SDK environment to resolve relevant optional skips. +- Cancellation tests need process inspection. If a sandbox denies it, record + that limitation and obtain the narrowly needed execution permission; do not + weaken recovery logic to make sandbox-restricted tests appear green. +- Update the milestone's review item, evidence and `RESUME.md` per coherent + checkpoint. Preserve historical receipts and report failures accurately. + End a pause with a clean committed tree; never stash. No push/merge/deploy. +- Use existing subscription harnesses for bounded required live proofs. + No credit purchase, reset redemption, paid API fallback or global AI-setting + change is authorized. Credentials and raw provider output stay outside Git. +- Report remaining work as concrete acceptance gates, not invented completion + percentages or fixed five-hour-window estimates. Shared quota depends on + actual models, context and account usage. +- Keep one short recovery entry with the source revision, exact active test + session/log if any, observed failure, next command and changed files. Resume + from that entry rather than repeatedly loading the full conversation. Do + not launch a second full suite while the first remains live. Distinguish + account quota, tool timeout, runtime budget expiry and host clock movement; + they are not interchangeable explanations for an interruption. + +## First action for Sol + +Read the recovery files and verify current Git state. Preserve `b766f9e`, its +`672e383` source checkpoint and later work. R3a's review and full gate are +verified; implement **R3b.1**'s authoritative saved-run reader first, then +**R3b.2**'s shared current-evidence gate and explicit evaluation/review revisions. +Use the consumer acceptance matrix above. Complete R3c before R4. +Preserve R1's check-integrity gate and R2's actual-attempt/native-identity gate. +Continue R3–R6 without waiting on the Claude login or Jev key. Retain R7/R8 +and C1 in the full scope. Report each slice as red baseline, verified offline, +installed proof, or externally blocked; never collapse those into one claim. + +Suggested instruction to SOL: “Execute this plan from R3b.1 on +`codex/engineering-team`, in small spec → failing public test → implementation +→ verification → review → checkpoint cycles. Do not redo R1/R2 or reload the +old chat. After each slice report the evidence, open gates and exact next +action. Keep the classifier off until its separate adoption gate passes.” diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md new file mode 100644 index 0000000..a8128db --- /dev/null +++ b/docs/plans/engineering-team/START-HERE.md @@ -0,0 +1,90 @@ +# DevSquad: coding-agent entry point + +**Build status: implementation in progress.** After an interruption, read +[RESUME.md](RESUME.md) first and compare it with current Git state. M1/M2 are +accepted; the September 29 review reopened M3 acceptance integrity and M5–M7 +implementation gaps. M4's actual Claude handoff remains externally blocked. +Execute [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md) and use +[backlog.json](backlog.json) for evidence. Do not restart the architecture +exercise. + +**Full-build assignment:** Use [SOL-HANDOFF.md](SOL-HANDOFF.md) for the user's request to have Sol execute everything, test thoroughly and make normal use simple. It includes M1–M7 plus the opt-in Council feature, and adds guided task entry over the same contracts. + +**Git starting point:** `codex/engineering-team` is the shared local/GitHub build branch. `main` remains the published runtime baseline. The [branch consolidation record](../../audits/2026-09-06-branch-consolidation.md) documents the preserved history and recovery paths. Continue this branch, or base a Codex worktree on it; verify the complete handoff checkpoint `ff1fa60` is an ancestor before coding. + +> Build an AI engineering team that can be operated from terminal, Codex, Claude Code, Antigravity and Grok. Use each eligible model/effort/tool configuration where it produces the best verified outcome; account for shared subscription limits. Preserve work across surfaces, learn from attempts and maintain the evidence automatically. + +```mermaid +flowchart LR + A[Fix invocation] --> B[Persist and recover runs] + B --> C[Usable branch review] + C --> D[Same run from any local app] + C --> E[Implement → review → verify] + D --> F[Capacity + learning] + E --> F + F --> G[Install and prove the full workflow] +``` + +## Read and execute + +1. Read [ADR-002](../../adr/ADR-002-surface-independent-engineering-team.md) for boundaries and decisions. +2. Implement the [contracts](CONTRACTS.md), using the [examples](examples/branch-review.json) as fixtures, not live model configuration. +3. Work through [IMPLEMENTATION](IMPLEMENTATION.md), one milestone at a time. [backlog.json](backlog.json) is the current completion record; resume the earliest unfinished requirement whose dependencies are ready. +4. Consult the [assessment](../../audits/2026-09-06-engineering-team-assessment.md) for verified defects and history, and [ADR-001](../../adr/ADR-001-contract-and-ledger-core.md) for legacy constraints retained by ADR-002. + +Selection is automatic by default, with validated per-role profile overrides. Read the [selection and LLM Council amendment](SELECTION-AND-COUNCIL.md) for the clarified contract. Its optional C1 extension follows M6 and does not block the seven core milestones. + +Also read the [native adapters and model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md): use verified Codex app-server capabilities, stable profile aliases, automatic catalog updates and qualified binding promotions. These refine M1/M3/M6/M7; they add no prerequisite milestone and do not require rewriting workflows for each model release. + +The September 26 [Jev/Laya evaluation and decision-helper amendment](DECISION-CLASSIFIERS.md) +adds optional M6 experiments for routing hints, skill selection and context +ranking, with further bounded uses prioritized. It changes no runtime defaults +and does not delay M5. The only current hosted authorization is the explicitly +capped one-request, $0.01 synthetic Jev pilot; it completed on October 1 and +that allowance is now spent. See the [receipt summary](evidence/M6-D2-jev-pilot-2026-10-01.json). +No repeat request, purchase, retry or private task upload is authorized; runtime +classification remains off. + +## Copyable execution brief + +```text +Implement DevSquad's September engineering-team plan in this repository. +Read docs/plans/engineering-team/RESUME.md, current Git state, +SOL-REVIEW-FOLLOWUP.md, START-HERE.md and the relevant contracts first. +Preserve existing implementation and historical receipts. +R1's source repair is verified; start with R2's observed-identity regressions, then execute the remaining +review packages through M5–M7 and C1. Do not restart completed M1/M2 work. +Preserve existing Bash 3.2 wrapper callers and their four error prefixes. +Keep all distributable core files inside plugin/core; add no cloud service. +Use fake CLIs for development; real provider runs are bounded smoke tests. +Do not change global AI account settings or silently switch to paid APIs. +Before marking a milestone complete, record the revision, checks, outcome, +and a reproducible receipt in backlog.json and update the relevant docs. +Run bash test/run.sh before each commit as CONTRIBUTING.md requires. +Commit each verified milestone. Continue through all ready work; +report exact blockers rather than claiming unsupported integrations work. +``` + +The initial architecture is decided. Routine implementation choices need no renewed architecture approval. If a discovery changes a contract, document the small proposed change and its impact in an ADR amendment; preserve compatibility or version the contract. + +## Completion means evidence + +| Claim | Required proof | +|---|---| +| Adapter works | Fake-binary conformance, supported flag mapping, one bounded real smoke receipt | +| Recovery works | Crash with live child; resume never creates a second writer | +| Surface works | Start/status/handoff or cancel from that installed application | +| Delivery works | Exact patch, independent review, checks and disposition all linked | +| Router improved | Comparable cases, all attempts, failures and rework; versioned evaluation | + +Keep planned and implemented features visibly separate. Do not mark M4/M7 complete merely because a config file parses, or a host advertises MCP support. Do not rewrite `.planning/STATE.md`'s historical February “100%” into a claim about this build. + +## Scope guard + +First usable product: **a saved branch review**. Next: **one bounded code change reviewed by another model**. Two functioning harnesses are sufficient to prove the engineering workflow; M7 verifies access from every requested local surface. Do not force every provider into every run. + +Defer a dashboard, universal DAG builder, remote execution service, automatic model training or unreviewed learned policy changes, plugin marketplace, autonomous merges and scheduled documentation jobs. Optional evaluated decision hints are bounded by the classifier amendment, not a replacement for deterministic policy. Existing plugin behavior remains available while the new runner is opt-in; switching hook suggestions to the new route source happens only after its gate passes. + +This packet began as architecture only. Current implementation and live-probe +evidence are tracked in RESUME.md, the milestone status and backlog; they do +not yet establish that the full engineering-team product is usable. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json new file mode 100644 index 0000000..7cc1106 --- /dev/null +++ b/docs/plans/engineering-team/backlog.json @@ -0,0 +1,509 @@ +{ + "schema_version": 1, + "plan_id": "surface-independent-engineering-team", + "created_at": "2026-09-06", + "baseline_revision": "c7f5930", + "architecture": "../../adr/ADR-002-surface-independent-engineering-team.md", + "contracts": "CONTRACTS.md", + "implementation": "IMPLEMENTATION.md", + "selection_and_council": "SELECTION-AND-COUNCIL.md", + "model_lifecycle_and_native_adapters": "MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md", + "decision_classifiers": "DECISION-CLASSIFIERS.md", + "execution_brief": "SOL-HANDOFF.md", + "review_followup": "SOL-REVIEW-FOLLOWUP.md", + "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], + "status": "in_progress", + "next_milestone": "M7", + "next_work_package": "R8", + "partial_implementation_checkpoint": { + "recorded_on": "2026-10-02", + "baseline_revision": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", + "status": "partial", + "artifact": "evidence/public-release-0.11.0.json", + "scope": "R3-R6 core runtime accepted: frozen604-test core, exact native EOF audit clean/accepted, pristine trusted-check integrity, installed schema17/no drift, nine actual installed SDK tests, updated agy1.2.14 actual MCP status. Legacy watchdog3fdc6bd independently reviewed and integrated without core source changes. Earlier accepted Claude delivery/handoff and Grok receipts retained separately.", + "next_action": "Complete final integrated Bash/affected checks, exact-head public CI, PR1 merge and verified v0.11.0 GitHub release; normal local Claude plugin update. Preserve original R7/C1 and R8 residual gates separately; do not repeat unchanged accepted proofs.", + "limitations": "Whole-plan closure is not claimed. Council is integrated/offline-tested but native-unavailable, comparison inconclusive and automaticOFF. Claude Code-tab proof remains unverified/TCC. Antigravity means agy CLI, not IDE; Grok/agy MCP status does not prove automatic writer/reviewer roles. Current CodexApp MCP connection needs DevSquad-only reconnect after schema17 upgrade. Jev off/single pilot spent; no paid API fallback or reset use authorized." + }, + "planning_checkpoint": { + "recorded_on": "2026-10-01", + "inspected_checkpoint": "b766f9e", + "implementation_revision": "672e383", + "artifact": "SOL-REVIEW-FOLLOWUP.md", + "scope": "Source review and planning only; no runtime edits, installation or provider requests", + "next_action": "R3b.1 authoritative saved-run reader, then R3b.2 current-evidence eligibility and explicit evaluation/review revisions, then R3c public and historical compatibility proof", + "additional_acceptance": "Reuse R3a assignment authority in R5; cover all terminal outcome origins; prove old-package active-run schema-upgrade safety through compatible coexistence or explicit deferral before real R8 installation", + "verification": "Source/test hashes match R3a evidence; JSON, generated reference and diff checks pass. Restricted Bash run was interrupted after process-inspection-related cancellation failures; permitted rerun passed 227 assertions in 11 files. No full Python rerun for documentation-only changes." + }, + "review_checkpoint": { + "reviewed_revision": "f4fa6577e2e891151231c9c7d3180be6e9e23faa", + "recorded_on": "2026-09-29", + "artifact": "SOL-REVIEW-FOLLOWUP.md", + "verification": "317 Python tests completed successfully with 2 optional-SDK skips; 227 Bash assertions passed; recurring SQLite cleanup warning remains; new integrity reproductions expose missing coverage", + "assessment": "M3 acceptance integrity and M5-M7 independent engineering work are reopened. Historical receipts remain evidence for their recorded scope; credentials are not the only remaining work." + }, + "review_work_packages": [ + {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, + {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, + {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, + {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, + {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "complete", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-closure-2026-10-02.json", "checkpoint": "Accepted dfe9976/source equivalent f87060b and installed schema16. Initial native findings repaired; exact follow-up e046a174 clean/accepted. Full 504 tests/511.241s, two optional SDK skips/no errors/failures/unraisable; nine actual installed SDK tests pass/no skips. Public complete trial/evaluate/qualify/promote/new-run/held-out regression/rollback, shared call/active deadline and all terminal projection origins pass. Safe old-schema upgrade and idempotence/drift gates pass. Failed history retained; automatic experimentation and Jev stay off."}, + {"id":"R6","title":"Normal terminal experience and readiness","status":"complete","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/public-release-0.11.0.json","checkpoint":"Accepted core terminal/readiness package: exact native EOF cb68 succeeded22/clean, pristine60 trusted-check integrity, frozen604 full tests/793.462s and installed schema17/nine SDK tests/no drift. Independently reviewed legacy watchdog3fdc6bd integrated with unchanged core tree;259 Bash/45 affected agent gates pass, root integrated gates recorded in release evidence. Public CI/publication remain release gates; desktop UI/Council residuals remain separate."}, + {"id":"R7","title":"Complete Council within existing runner","status":"in_progress","milestones":["C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":["G5"],"checkpoint":"Council reconciliation0e7dd319/EOF4f integrated, schema17 actually installed; mechanical comparison/cancel/recovery/old16 fences and frozen604 core gate pass. Final nongenerating diagnostic exhausted: sandbox network_request_failed/noHTTP and unsandboxed strict-response rejection, zero generating calls. Native quality/usage unavailable; fixture comparison inconclusive; automaticOFF. No more network attempts or permission widening. Original C1 native acceptance remains incomplete, not a core-publication blocker under user-selected release scope."}, + {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Safe updated schema17 install/no drift/nine actual SDK tests and updated agy1.2.14 actual saved-run MCP proof accepted; earlier live Claude handoff/delivery and Grok MCP retained. PR1/v0.11.0 public release pending exact-head CI/merge/assets. Existing CodexApp old MCP server needs scoped reconnect; Claude desktop Code-tab/TCC and whole-plan closure remain separate."} + ], + "milestones": [ + { + "id": "M1", + "title": "Truthful reusable adapter invocation", + "depends_on": [], + "status": "complete", + "acceptance_section": "M1 — Make invocation truthful and reusable", + "evidence": [ + { + "kind": "implementation_checkpoint", + "revision": "97a10f0", + "command_or_action": "34 core tests, 202 shell assertions, temporary wheel install, and saved integrated native read-only smoke", + "outcome": "All enumerated offline and live M1 gates pass; independent review accepted the bounded invocation/preparation scope", + "artifact": "evidence/M1-invocation-core-2026-09-06.json", + "recorded_at": "2026-09-07T08:20:00+05:30", + "availability": "portable_redacted" + } + ], + "blocker": null + }, + { + "id": "M2", + "title": "Durable runs, cancellation and recovery", + "depends_on": ["M1"], + "status": "complete", + "acceptance_section": "M2 — Persist jobs and own their processes", + "evidence": [ + { + "kind": "implementation_checkpoint", + "revision": "df955f4", + "command_or_action": "62 core tests, 202 shell assertions, real subprocess lifecycle tests, migration-1 fixture upgrade and installed-wheel migration 2", + "outcome": "Store and bounded supervisor lifecycle pass; durable detached service operations remain pending", + "artifact": "M2-STATUS.md", + "recorded_at": "2026-09-08T00:00:00+05:30", + "availability": "tracked tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "f0d29a7", + "command_or_action": "82 core tests, 202 shell assertions, installed-wheel schema-3-to-4 migration, two-process receipt import, repository-shadow and timeout adversarial regressions", + "outcome": "Durable import and CLI/wheel slices pass; M2 remains in progress for crash-phase recovery, all-terminal receipts, cancellation/stdin races, host claims and independent re-review", + "artifact": "M2-STATUS.md", + "recorded_at": "2026-09-09T21:58:27+05:30", + "availability": "tracked tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "f9f2ffa", + "command_or_action": "94 core tests with warnings enabled, 202 shell assertions, independent preparation-recovery review, real reservation-process crash, durable stdin and project-retarget regressions", + "outcome": "Preparing and pre-gate launching recovery, pre-attempt receipts, predecessor validation, exact durable stdin and detached-process reaping pass; M2 remains in progress for host claims and remaining process races", + "artifact": "M2-STATUS.md", + "recorded_at": "2026-09-15T11:06:00+05:30", + "availability": "tracked tests" + }, + { + "kind": "milestone_acceptance", + "revision": "ddb6f51", + "command_or_action": "121 core tests with ResourceWarning promoted to error, 202 shell assertions, spawned-process crash/cancel recovery, independent-process races and installed-wheel schema-4-to-5 migration", + "outcome": "All M2 durability, cancellation, recovery and host-handoff gates pass; both final independent-review P1 findings are fixed with regressions; real branch review remains the explicit M3 boundary", + "artifact": "evidence/M2-durable-runs-2026-09-16.json", + "recorded_at": "2026-09-16T00:41:03+05:30", + "availability": "portable_redacted" + } + ], + "blocker": null + }, + { + "id": "M3", + "title": "Usable branch review from terminal", + "depends_on": ["M2"], + "status": "in_progress", + "acceptance_section": "M3 — Ship a useful branch review", + "open_review_items": [], + "review_resolution": "F1/R1 repair is now installed and the complete combined branch review is accepted with all checks passing. Milestone closure audit remains separate; old pending-install wording is superseded by R8-installed-workflows evidence.", + "evidence": [ + { + "kind": "implementation_checkpoint", + "revision": "ca55990", + "command_or_action": "131 core tests and 202 shell assertions covering strict deterministic routing, versioned aliases, pins/fallbacks, independent review and typed pool capacity", + "outcome": "M3 routing decisions are deterministic, fail closed and retain exact profile/policy identity; workflow execution remains unavailable", + "artifact": "../../../test/core/test_router.py", + "recorded_at": "2026-09-16T00:52:00+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "1c7b614", + "command_or_action": "132 core tests and 202 shell assertions covering public-preflight routing snapshots and replay after mutable config corruption", + "outcome": "New runs freeze exact config hashes, bindings and selected concrete profiles before execution", + "artifact": "../../../test/core/test_service.py", + "recorded_at": "2026-09-16T00:58:00+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "cd9a881", + "command_or_action": "144 core tests with ResourceWarning promoted to error and 202 shell assertions covering exact commit inputs and run-owned detached worktrees", + "outcome": "M3 preflight freezes base/target/config identity, rejects dirty scoped inputs and escaping symlinks, and preserves the submitted checkout/index/HEAD; reviewer/check/lead execution remains pending", + "artifact": "../../../test/core/test_workspaces.py", + "recorded_at": "2026-09-16T01:08:00+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "a756307", + "command_or_action": "154 core tests with ResourceWarning promoted to error and 202 shell assertions covering strict candidate-bound review, check, evaluation and attempt evidence", + "outcome": "Reviewer output and trusted check results are schema-bound to the frozen candidate; required failures cannot be overridden and usage remains explicitly unknown when unavailable", + "artifact": "../../../test/core/test_review_workflow.py", + "recorded_at": "2026-09-16T01:21:23+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "97d2c6c", + "command_or_action": "159 core tests with ResourceWarning promoted to error and 202 shell assertions covering durable offline review, isolated declared checks and atomic host handoff publication", + "outcome": "A frozen offline reviewer and separate check worktree now produce durable evidence and a claimable host packet without changing the submitted checkout", + "artifact": "../../../test/core/test_review_runtime.py", + "recorded_at": "2026-09-16T01:34:41+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "30cf49e", + "command_or_action": "164 core tests with ResourceWarning promoted to error and 202 shell assertions covering host accept/reject/revise, retry budgets, exact evidence binding, crash resume and terminal reports", + "outcome": "The offline M3 path now completes end to end with attempt-specific evidence, required-check enforcement, clean retry workspaces and durable receipt.json, receipt.md, events.jsonl, artifact manifest and result receipt; real provider execution remains pending", + "artifact": "../../../test/core/test_review_runtime.py", + "recorded_at": "2026-09-17T16:35:24+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "459ff3f", + "command_or_action": "167 core tests with ResourceWarning promoted to error and 202 shell assertions covering the public native Codex reviewer and provider protocol faults", + "outcome": "M3 can run one frozen, ephemeral, read-only Codex app-server review inside the M2 process fence, verify exact execution identity, validate candidate-bound structured output and record native usage; the bounded real Codex gate remains pending", + "artifact": "../../../test/core/test_review_runtime.py", + "recorded_at": "2026-09-17T16:55:37+05:30", + "availability": "tracked_tests" + }, + { + "kind": "live_acceptance", + "revision": "9478796", + "command_or_action": "Run the public branch-review service with the isolated subscription-backed codex-cli 0.153.4 app-server, gpt-5.5/low, read-only permissions, one required check and host acceptance", + "outcome": "Succeeded with one supported finding, a passing required check, verified observed identity, 53,700 native-reported tokens and all five terminal report hashes; M3 remains in progress for waiting/failure reports, headless lead and final audit", + "artifact": "evidence/M3-native-codex-review-2026-09-17.json", + "recorded_at": "2026-09-17T21:34:40+05:30", + "availability": "portable_redacted" + }, + { + "kind": "milestone_acceptance", + "revision": "1737667", + "command_or_action": "188 core tests with ResourceWarning promoted to error, 202 shell assertions, bounded reviewer/lead fallback, headless budget terminalization, transactional unknown-capacity trial limit and independent audit closure", + "outcome": "All M3 branch-review, routing, selection, accounting, headless-lead, terminal-report and live Codex acceptance gates are covered; four independent-audit findings are fixed with direct regressions", + "artifact": "evidence/M3-branch-review-2026-09-22.json", + "recorded_at": "2026-09-22T20:42:05+05:30", + "availability": "portable_redacted" + } + ], + "blocker": null + }, + { + "id": "M4", + "title": "Shared local MCP access and host handoffs", + "depends_on": ["M3"], + "status": "blocked", + "acceptance_section": "M4 — Use that same run from local apps", + "evidence": [ + { + "kind": "implementation_checkpoint", + "revision": "643910d", + "command_or_action": "200 core tests with ResourceWarning promoted to error, 208 shell assertions, Python 3.11 lock resolution and 12 focused official-SDK tests for the optional MCP boundary, saved-run tools and recursion guards", + "outcome": "M4 Plan 06-01 provides dependency-isolated stdio MCP access to all eight saved-run operations with strict envelopes, bounded previews, caller provenance and worker-origin mutation denial; setup/doctor and cross-surface live gates remain open", + "artifact": "../../../test/core/test_mcp.py", + "recorded_at": "2026-09-22T21:14:30+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "9abad1e", + "command_or_action": "211 core tests with ResourceWarning promoted to error, 208 shell assertions, 21 pinned-SDK tests and isolated-HOME add/no-op/drift-repair checks through the installed Codex, Claude, Antigravity and Grok host CLIs", + "outcome": "M4 Plan 06-02 provides strict host templates, stable launcher resolution, idempotent setup, inherited/duplicate fail-closed detection, redacted installed-app doctor reporting and the squad_doctor MCP tool; real in-app cross-surface receipts remain open", + "artifact": "MCP-LOCAL-ACCESS.md", + "recorded_at": "2026-09-23T04:22:26+05:30", + "availability": "tracked_tests" + }, + { + "kind": "blocked_acceptance", + "revision": "aa0fef5", + "command_or_action": "212 core tests with ResourceWarning promoted to error, 208 shell assertions, 22 pinned-SDK tests, stable four-host setup, real Codex MCP read and a terminal-to-stdio-client saved-run handoff with disconnect survival and claim fencing", + "outcome": "All independent M4 implementation and cross-client gates pass; Codex live operation is proven, while the exact milestone gate remains blocked because Claude Code requires normal provider login before it can perform the real host handoff", + "artifact": "evidence/M4-local-mcp-2026-09-23.json", + "recorded_at": "2026-09-23T06:59:59+05:30", + "availability": "portable_redacted" + } + ], + "blocker": "Claude authentication and actual CLI/MCP handoff now pass. Claude local Code-tab operation remains unverified. The user's October 2 clarification excludes Antigravity IDE: verify the updated agy CLI against the final accepted installation. Portable Claude CLI proof is not substituted for the local Code-tab proof." + }, + { + "id": "M5", + "title": "Bounded implementation with independent review", + "depends_on": ["M3"], + "status": "in_progress", + "acceptance_section": "M5 — Deliver a bounded engineering change", + "open_review_items": ["G4"], + "review_resolution": "F1/R1 and F2/R2 repairs, G4 discovery, installed real Claude-to-independent-Codex 477-test delivery and complete combined review now pass. Whole M5 closure audit and broader R6 scope remain separate.", + "evidence": [ + { + "kind": "implementation_checkpoint", + "revision": "d96e9e4", + "command_or_action": "215 core tests with ResourceWarning promoted to error, 220 Bash assertions, 3 focused Claude adapter tests and a fresh wheel content check", + "outcome": "The bounded Claude 2.1.220 headless adapter now uses explicit read/write tool profiles, structured non-interactive output, strict empty MCP config, no blanket permission bypass and shared legacy error classification; delivery workspaces and workflow execution remain open", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-23T12:19:00+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "6a7e849", + "command_or_action": "220 core tests with ResourceWarning promoted to error, 220 Bash assertions and 5 focused delivery-workspace tests", + "outcome": "A run-owned detached delivery worktree now freezes only validated in-scope edits into a replay-safe local candidate commit and binary patch identity while preserving the source checkout and remote refs; durable implementer execution remains open", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-23T18:00:27+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "0e88d73", + "command_or_action": "224 core tests with ResourceWarning promoted to error, 220 Bash assertions and focused delivery/service/supervisor suites", + "outcome": "The durable implementer is fenced to one writer, validates frozen attempt evidence, serializes recovery-time candidate freezing and atomically publishes local candidate/patch artifacts plus separate review/check worktrees; invalid scope cannot publish a candidate and the source checkout/remote remain unchanged", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-23T18:37:00+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "30b98df", + "command_or_action": "226 core tests discovered with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions and exact-candidate delivery/review regressions", + "outcome": "Offline delivery now advances from a fenced local candidate through read-only review and separate trusted checks to a candidate-bound issue-delivery lead handoff; disposition, bounded revisions, complete receipts and live different-model/two-harness proof remain open", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-26T15:01:21Z", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "4c76887", + "command_or_action": "237 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions and seeded two-candidate repair/headless/crash regressions", + "outcome": "Issue-delivery lead accept/reject and bounded revise now terminalize with complete candidate-bound reports; revise returns atomically to one fenced implementer, creates distinct candidate review/check workspaces, rejects stale live evidence and enforces revision/invocation budgets. Delivery fallback/failure/cancellation history and live two-harness proof remain open", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-26T23:58:00-07:00", + "availability": "tracked_tests" + }, + { + "kind": "offline_completion_checkpoint", + "revision": "f199cd2", + "command_or_action": "241 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions, 19 focused delivery regressions and a controlled live-child supervisor kill", + "outcome": "All independent M5 behavior is implemented offline: the frozen Claude bridge records identity/session/usage, same-permission rate-limit fallback and cancellation retain complete history, and revised live-writer recovery cannot launch a duplicate or mutate the source checkout/remotes. Only the genuine authenticated Claude-to-Codex two-harness acceptance receipt remains.", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-27T08:10:15-07:00", + "availability": "tracked_tests" + } + ], + "blocker": "Claude login, genuine delivery and G4 check coverage are no longer blockers. Remaining acceptance audit and broader R4/R5/R6 dependencies remain open; see installed-workflows evidence." + }, + { + "id": "M6", + "title": "Shared capacity and evidence-based improvement", + "depends_on": ["M4", "M5"], + "status": "in_progress", + "acceptance_section": "M6 — Make capacity and improvement evidence useful", + "open_review_items": ["F3", "F4", "G1", "G2"], + "decision_helper_work_packages": [ + { + "id": "M6-D1", + "title": "Optional typed decision contract and frozen evaluation baseline", + "status": "complete", + "specification": "DECISION-CLASSIFIERS.md", + "evidence": [ + { + "kind": "experiment_preparation", + "revision": "70e59cb", + "command_or_action": "One-request dry run plus 5 focused probe tests, 231-test complete core discovery with ResourceWarning promoted to error, and 220 Bash assertions", + "outcome": "The pinned Jev 1.13 synthetic pilot has 8 cases and 24 typed choices, a $0.01 ceiling, no retry, strict model/probability/usage validation and private-output enforcement; no provider request ran because TypeSafe login/API key is unavailable", + "artifact": "experiments/jev-pilot-v1.json", + "recorded_at": "2026-09-26T17:26:40Z", + "availability": "tracked_fixture_and_tests" + }, + { + "kind": "offline_completion_checkpoint", + "revision": "87fa9cf", + "command_or_action": "291 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions and focused contract/store/service decision-helper tests", + "outcome": "Off is a zero-call no-op; shadow is observation-only; advisory is constrained to the deterministic eligible set. Schema 13 provides replay-safe call/cache accounting, launched-unknown fencing, cancellation evidence and public preflight/status integration without persisting raw task text.", + "artifact": "experiments/decision-helper-baseline-v1.json", + "recorded_at": "2026-09-29T02:57:24-07:00", + "availability": "tracked_fixture_and_tests" + } + ], + "checkpoint": "Default-off typed contracts, strict authority bounds, deterministic fake adapter, schema-13 call/cache accounting, crash/cancel/replay fencing, public preflight/status integration and a frozen synthetic baseline are verified at 87fa9cf", + "completed_revision": "87fa9cf" + }, + { + "id": "M6-D2", + "title": "Capped Jev synthetic pilot for task/profile and skill hints", + "depends_on": ["M6-D1"], + "status": "complete", + "specification": "DECISION-CLASSIFIERS.md", + "evidence": [ + { + "kind": "local_env_setup", + "recorded_on": "2026-10-01", + "artifact": "../../../test/core/test_jev_decision_probe.py", + "command_or_action": "14 offline Jev probe tests, explicit --env-file dry run, private local .env and Git-ignore checks; 227 Bash assertions in 11 files, generated-reference, JSON and whitespace checks passed", + "outcome": "Explicit private env-file loading is verified without shell evaluation, key disclosure or a live API call. Local key field remains blank; authentication and the capped live pilot are still pending.", + "availability": "tracked_tests_and_blank_example" + }, + { + "kind": "live_synthetic_smoke", + "revision": "0d37764", + "recorded_at": "2026-10-01T20:35:46.511565+00:00", + "artifact": "evidence/M6-D2-jev-pilot-2026-10-01.json", + "command_or_action": "One pinned Jev 1.13 API request after official pricing/API recheck; no retries; private env-file and redacted receipt outside Git", + "outcome": "Typed authenticated smoke completed: 5373 input tokens, 1762 output tokens, 187 ms and estimated $0.00022567. Task-family and skill labels matched 8/8 each; execution-tier matched 6/8. This closes the single-request smoke only, not production quality or routing adoption. Runtime remains off and the one-request allowance is spent.", + "availability": "portable_redacted_with_private_receipt_hash" + } + ], + "checkpoint": "One-request smoke complete; preserved two tier disagreements, exact model/usage/cost evidence and private receipt hash. No automatic routing change or additional hosted request authorized.", + "blocker": null + }, + { + "id": "M6-D3", + "title": "Triggered local Laya fallback and measured adoption decision", + "depends_on": ["M6-D2"], + "status": "pending", + "specification": "DECISION-CLASSIFIERS.md", + "evidence": [], + "fallback_rule": "Use pinned Laya locally if Jev access requires a larger purchase, price/usage exceeds the frozen cap, or measured quality/latency fails the predeclared gate" + } + ], + "evidence": [ + { + "kind": "capacity_checkpoint", + "revision": "d170c01", + "command_or_action": "255 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions, two-project reservation race, schema-8 active-attempt migration and public capacity CLI/service/routing tests", + "outcome": "Schema 9 persists strict scoped quota-window observations and reservations; fresh exhaustion, stale/unknown evidence, model-family sublimits, post-preflight changes and ambiguous ownership now affect deterministic routing and transactional launch without widening billing or permission eligibility.", + "artifact": "M6-STATUS.md", + "recorded_at": "2026-09-27T08:41:45-07:00", + "availability": "tracked_tests" + }, + { + "kind": "learning_evaluation_checkpoint", + "revision": "edfb3f3", + "command_or_action": "263 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions, replay-safe outcome/experiment persistence and public report/evaluate CLI tests", + "outcome": "Schemas 10 and 11 preserve final and late outcomes, truthful attempt contribution, selection-mode-separated reports and immutable paired evaluation/held-out experiments. Missing or failed evidence produces no-change, qualifying evidence produces only a promotion proposal with rollback, conflicting replay is fenced and evaluation never mutates active policy.", + "artifact": "M6-STATUS.md", + "recorded_at": "2026-09-29T08:42:48Z", + "availability": "tracked_tests" + }, + { + "kind": "learning_proposal_checkpoint", + "revision": "98c6685", + "command_or_action": "265 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions and focused pure/store/service/filesystem/CLI proposal tests", + "outcome": "squad learn propose writes content-addressed local JSON and Markdown from a consistent ledger snapshot, verifies report/experiment/evaluation evidence, preserves sample size, missingness, failures and rollback, returns explicit no-change without evidence and never changes active policy.", + "artifact": "M6-STATUS.md", + "recorded_at": "2026-09-29T08:50:09Z", + "availability": "tracked_tests" + }, + { + "kind": "profile_lifecycle_checkpoint", + "revision": "ca4ee73", + "command_or_action": "275 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions, focused catalog/lifecycle/CLI tests and a complete-catalog removal fallback", + "outcome": "Schema 12 lifecycle templates, qualification, compare-and-swap promotion, new-run-only binding changes, held-out regression rollback and immutable receipts are verified. Catalog drift revalidates only affected profiles, keeps additions unqualified, and an unavailable incumbent rolls back only to the newest prior proven/qualified binding under the same template or blocks without mutation.", + "artifact": "M6-STATUS.md", + "recorded_at": "2026-09-29T02:31:32-07:00", + "availability": "tracked_tests" + }, + { + "kind": "decision_helper_checkpoint", + "revision": "87fa9cf", + "command_or_action": "291 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions, strict off/shadow/advisory authority tests, schema-13 cache/call fencing and public preflight/status integration", + "outcome": "The optional helper defaults to zero-call off mode; shadow cannot alter execution and advisory can only reorder already-eligible unpinned profiles under reviewed gate evidence. Invalid or uncertain output, missing adapters, cancellation and launched-unknown recovery preserve deterministic routing without duplicate paid calls or raw task persistence.", + "artifact": "experiments/decision-helper-baseline-v1.json", + "recorded_at": "2026-09-29T02:57:24-07:00", + "availability": "tracked_fixture_and_tests" + }, + { + "kind": "early_experiment_checkpoint", + "revision": "70e59cb", + "command_or_action": "Prepared and offline-validated the explicitly authorized capped Jev pilot before M6 runtime integration", + "outcome": "Synthetic probe mechanics are ready and ordinary runtime behavior is unchanged; the billable live smoke is blocked on TypeSafe login/API-key setup and cannot count as classifier quality evidence", + "artifact": "experiments/jev-pilot-v1.json", + "recorded_at": "2026-09-26T17:26:40Z", + "availability": "tracked_fixture_and_tests" + } + ], + "blocker": "R3-R5 evidence/routing/learning/outcome integration is accepted and installed. Whole-milestone closure audit remains separate. Jev single-request smoke does not prove production quality; runtime off and Laya conditional." + }, + { + "id": "M7", + "title": "Installation, migration and all-surface proof", + "depends_on": ["M6"], + "status": "in_progress", + "acceptance_section": "M7 — Package, migrate and prove every requested surface", + "open_review_items": [], + "evidence": [ + { + "kind": "packaging_checkpoint", + "revision": "2275b49", + "command_or_action": "299 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 227 Bash assertions, 8 focused installer tests and a fresh offline MCP install", + "outcome": "The Claude-independent immutable installer, stable launcher, idempotent update, payload drift checks, legacy-launcher migration, active-run release retention, one-line JSON and pip dependency validation are verified.", + "artifact": "M7-STATUS.md", + "recorded_at": "2026-09-29T10:34:00Z", + "availability": "tracked_tests" + }, + { + "kind": "installed_runtime_checkpoint", + "revision": "ccba125", + "command_or_action": "300 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 227 Bash assertions, exact Codex 0.155 native handshake/model catalog, four-host unchanged setup and live Terminal/Antigravity/Codex saved-run operations", + "outcome": "The current immutable MCP release is installed with no drift and all four registrations matching. Terminal started and cancelled one saved run; scoped Gemini and ephemeral Codex observed the identical run/version through the installed MCP server. Claude login, Grok authentication and the installed two-model delivery remain open.", + "artifact": "evidence/M7-installed-runtime-2026-09-29.json", + "recorded_at": "2026-09-29T10:52:08Z", + "availability": "portable_redacted" + }, + { + "kind": "normal_entry_checkpoint", + "revision": "85aa378", + "command_or_action": "311 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 227 Bash assertions, offline end-to-end managed delivery and installed no-generation review/fix dry-runs", + "outcome": "Normal review and bounded fix now require no hand-written task/routing JSON. Exact commits, embedded hash-frozen routing, runtime-discovered Codex selection, one Claude writer, independent Codex review, checks, saved-run continuation and compact next-action output are verified; the updated immutable release passes pip check, idempotent reinstall, four-host setup and doctor.", + "artifact": "evidence/M7-normal-entry-2026-09-29.json", + "recorded_at": "2026-09-29T11:17:22Z", + "availability": "portable_redacted" + }, + { + "kind": "live_normal_review_checkpoint", + "revision": "b1d52ad", + "command_or_action": "317 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 227 Bash assertions, immutable MCP reinstall and one installed exact-commit gpt-5.5/low review with host disposition", + "outcome": "Codex terminal output selection and redacted diagnostics are compatible with 0.155, each trusted check receives a fresh isolated HOME, the exact candidate received a verified clean read-only review, both frozen checks passed and the artifact-bound handoff terminalized succeeded. Claude/Grok live gates remain external.", + "artifact": "evidence/M7-live-codex-review-2026-09-29.json", + "recorded_at": "2026-09-29T12:13:54Z", + "availability": "portable_redacted" + } + ], + "blocker": "Core R4/R6 and updated schema17 installation are accepted. Actual Claude CLI delivery/handoff, Grok MCP and updated agy1.2.14 MCP receipts pass. Claude desktop Code-tab remains unverified/TCC; current CodexApp DevSquad-only MCP reconnect and final public CI/publication remain. No automatic Grok/agy worker claim." + } + ], + "extensions": [ + { + "id": "C1", + "title": "Selective evidence-based Council decisions", + "optional": true, + "depends_on": ["M6"], + "status": "in_progress", + "specification": "SELECTION-AND-COUNCIL.md", + "acceptance_section": "C1 — Optional Council extension after M6", + "evidence": [{"kind":"integrated_partial_acceptance","revision":"b83dda7fbf691503d3adf3c9ea6ecebd4071ba16","artifact":"evidence/r7-final-nongenerating-network-diagnostic.json","outcome":"Integrated schema17 Council mechanics/offline tests and actual install pass. Native generating acceptance unavailable; fixture comparison inconclusive, automaticOFF. No more network attempts authorized."}], + "blocker": "Native Council network/response attestation unavailable after final nongenerating diagnostic; original C1 acceptance incomplete, not a user-selected core release blocker." + } + ] +} diff --git a/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json b/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json new file mode 100644 index 0000000..2051734 --- /dev/null +++ b/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json @@ -0,0 +1,52 @@ +{ + "schema_version": 1, + "milestone": "M1", + "status": "complete", + "implementation_revision": "97a10f0", + "verification_revision": "7b5c41c", + "recorded_at": "2026-09-07T08:30:00+05:30", + "evidence": [ + { + "kind": "offline_test", + "command_or_action": "/bin/bash test/run.sh", + "outcome": "10 test files passed; 202 shell assertions passed, including forced portable timeout, fast cleanup, descendant cleanup and TERM-ignoring-root escalation", + "availability": "tracked tests" + }, + { + "kind": "offline_test", + "command_or_action": "PYTHONDONTWRITEBYTECODE=1 python3 -m unittest discover -s test/core -v", + "outcome": "34 tests passed, including eight independent reviewer regressions", + "availability": "tracked tests" + }, + { + "kind": "package_test", + "command_or_action": "bundled Python 3.12: pip wheel --no-deps plugin/core; install into temporary venv; import packaged classifier and resolve shared policy resource", + "outcome": "wheel built and installed; shared taxonomy loaded outside the checkout", + "availability": "reproducible locally" + }, + { + "kind": "live_metadata", + "command_or_action": "Codex 0.135.0 app-server initialize then model/list with limit 100 on 2026-09-06", + "outcome": "initialized; complete single page with 5 account-visible models and model-scoped reasoning efforts; default gpt-5.5/medium", + "availability": "redacted portable summary" + }, + { + "kind": "live_smoke", + "command_or_action": "codex exec --json --sandbox read-only --ephemeral --model gpt-5.5 with model_reasoning_effort=low and a no-tools response probe on 2026-09-06", + "outcome": "exit 0; thread.started, turn.started, item.completed and turn.completed observed; final message DEVSQUAD_M1_PROBE_OK", + "availability": "redacted portable summary" + }, + { + "kind": "live_integrated_probe", + "command_or_action": "production preparation and JSON-line peer: initialize response, initialized notification, then model/list or prepared thread/start; bounded per-process low-effort override", + "outcome": "Passed at 97a10f0: complete discovery selected gpt-5.5/low, catalog-backed preparation launched a fresh server, and the read-only no-tools turn returned the exact expected output with correlated terminal completion. Both owned process groups were confirmed absent after cleanup. Receipt SHA256: 324c2ce6154936ecf71a8bf913a195befd4251efc6dbb8bbab3dc0dbe1b86df8", + "availability": "private receipt and stream hashes retained locally; redacted summary only" + } + ], + "independent_review": { + "status": "accepted", + "scope": "bounded M1 invocation and preparation", + "outcome": "Reviewer reran 34 core tests, relied on the required 202-assertion shell run, verified the receipt and all five private artifact hashes, and inspected requested gpt-5.5/low, readOnly with networkAccess false, and correlated turn completion." + }, + "residual_blockers": [] +} diff --git a/docs/plans/engineering-team/evidence/M2-durable-runs-2026-09-16.json b/docs/plans/engineering-team/evidence/M2-durable-runs-2026-09-16.json new file mode 100644 index 0000000..a1065e3 --- /dev/null +++ b/docs/plans/engineering-team/evidence/M2-durable-runs-2026-09-16.json @@ -0,0 +1,47 @@ +{ + "schema_version": 1, + "milestone": "M2", + "status": "complete", + "implementation_revision": "ddb6f51", + "recorded_at": "2026-09-16T00:41:03+05:30", + "provider_invocations": 0, + "evidence": [ + { + "kind": "offline_test", + "command_or_action": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning python3 -m unittest discover -s test/core -v", + "outcome": "121 tests passed, including real subprocess crash recovery, independent-process races, deterministic SQLite lock contention, exact stdin, bounded cleanup and installed-wheel migration through schema 5", + "availability": "tracked tests" + }, + { + "kind": "offline_test", + "command_or_action": "bash test/run.sh", + "outcome": "10 Bash test files passed; 202 assertions preserved the Bash 3.2, optional-jq, wrapper and legacy contracts", + "availability": "tracked tests" + }, + { + "kind": "crash_and_race_matrix", + "command_or_action": "Spawned-process tests interrupt preparation, attempt reservation, post-runner-identity/pre-gate launch, live-runner coordination, orphan-child cancellation, exit import and artifact finalization before the database commit; independent processes race starts, writers, cancellation, claims and completion", + "outcome": "No duplicate run/writer/cleanup, no execution before the gate, no blind relaunch or ambiguous signal, no false completion and idempotent recovery", + "availability": "tracked tests" + }, + { + "kind": "host_handoff", + "command_or_action": "Service, CLI and Store tests publish, claim, renew, take over, complete, replay, audit late submissions and cancel schema-5 host handoffs", + "outcome": "Run-version and fencing-token ownership is durable; expiry is evaluated after acquiring the SQLite write transaction, including a real lock wait crossing the deadline", + "availability": "tracked tests" + }, + { + "kind": "package_test", + "command_or_action": "Build an offline wheel, install it into a fresh temporary environment outside the checkout with PYTHONPATH removed, and migrate an independent schema-4 fixture", + "outcome": "Packaged migrations through 005 are present, schema 4 upgrades to 5, and future schema versions are refused", + "availability": "reproducible locally" + } + ], + "independent_review": { + "status": "completed_with_findings_resolved", + "scope": "M2 crash recovery, host-handoff fencing and final artifact-import additions", + "outcome": "The reviewer found two P1 races and no other defect in the bounded target. Checkpoint ddb6f51 moves lease authorization after lock acquisition and makes recovery_cleanup resumable through public cancel; both findings have failing-boundary regressions in the final gate." + }, + "milestone_boundary": "Real branch-review workflow execution begins in M3; public M2 requests return CAPABILITY_UNAVAILABLE.", + "residual_blockers": [] +} diff --git a/docs/plans/engineering-team/evidence/M3-branch-review-2026-09-22.json b/docs/plans/engineering-team/evidence/M3-branch-review-2026-09-22.json new file mode 100644 index 0000000..02f29c8 --- /dev/null +++ b/docs/plans/engineering-team/evidence/M3-branch-review-2026-09-22.json @@ -0,0 +1,58 @@ +{ + "schema_version": 1, + "milestone": "M3", + "status": "complete", + "implementation_revision": "1737667a3a847bfd09559abd5a53f8ec88745982", + "recorded_at": "2026-09-22T20:42:05+05:30", + "offline_gate": { + "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONPATH=plugin/core/src python3 -W error::ResourceWarning -m unittest discover -s test/core -q", + "result": "188 tests passed", + "bash_command": "bash test/run.sh", + "bash_result": "10 test files and 202 assertions passed" + }, + "live_gate": { + "revision": "9478796", + "artifact": "M3-native-codex-review-2026-09-17.json", + "result": "subscription-backed Codex gpt-5.5/low read-only review succeeded with one supported finding, a passing required check, 53700 native-reported tokens and five terminal report hashes" + }, + "closeout_checkpoints": [ + { + "revision": "9dc795a", + "result": "waiting handoff and rich early-terminal report sets" + }, + { + "revision": "8f3fa56", + "result": "durable host and headless lead paths with distinct fenced attempts" + }, + { + "revision": "e1afa4d", + "result": "cumulative wall budgets, transactional live pool reservations and complete waiting cancellation reports" + }, + { + "revision": "84deb77", + "result": "bounded frozen reviewer and lead runtime fallbacks with schema-8 attempt identity" + }, + { + "revision": "1737667", + "result": "four independent-audit defects fixed with deterministic regressions" + } + ], + "independent_audit": { + "target_revision": "84deb77", + "findings": [ + "pre-launch recovery consumed an unused fallback slot", + "headless-lead invocation exhaustion stranded an awaiting run", + "unknown capacity allowed more than one unresolved trial", + "later early-terminal reports omitted completed earlier revisions" + ], + "closure_revision": "1737667", + "closure": "all four reproduced by tracked tests; focused and complete offline gates pass", + "follow_up_limit": "the same audit agent could not rerun after the fixes because the shared Plus window was exhausted" + }, + "provider_blockers_outside_m3": [ + "Claude CLI is not logged in", + "Grok authentication is expired", + "Antigravity is authenticated but headless scoped trust/permission remains blocked" + ], + "residual_m3_blockers": [] +} diff --git a/docs/plans/engineering-team/evidence/M3-native-codex-review-2026-09-17.json b/docs/plans/engineering-team/evidence/M3-native-codex-review-2026-09-17.json new file mode 100644 index 0000000..2f27c6c --- /dev/null +++ b/docs/plans/engineering-team/evidence/M3-native-codex-review-2026-09-17.json @@ -0,0 +1,64 @@ +{ + "schema_version": 1, + "milestone": "M3", + "status": "in_progress", + "implementation_revision": "9478796247486ead4e62578cff4035f418534ac7", + "recorded_at": "2026-09-17T21:34:40+05:30", + "evidence": [ + { + "kind": "offline_test", + "command_or_action": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning python3 -m unittest discover -s test/core -p 'test_*.py' -q", + "outcome": "169 core tests passed after the current-version, isolated-auth and provider-compatible structured-output fixes", + "availability": "tracked tests" + }, + { + "kind": "offline_test", + "command_or_action": "bash test/run.sh", + "outcome": "10 Bash test files passed; all 202 Bash 3.2 and optional-jq assertions passed", + "availability": "tracked tests" + }, + { + "kind": "live_public_branch_review", + "command_or_action": "test/core/probes/m3_codex_review.py --run-live --codex-binary /Applications/ChatGPT.app/Contents/Resources/codex --model gpt-5.5 --effort low --timeout 180", + "outcome": "Public Service.start reached a host handoff, the required unit-tests check passed, the native reviewer returned one supported finding, the host accepted the evidence and the run terminalized succeeded with all five required reports", + "availability": "private artifacts retained locally; portable redacted hashes below" + }, + { + "kind": "execution_identity", + "command_or_action": "Freeze and re-observe the selected app-server identity before accepting review evidence", + "outcome": "codex-cli 0.153.4; OpenAI gpt-5.5; low effort; read_only; verified; correlated native thread and turn IDs present; ephemeral thread; isolated temporary CODEX_HOME with only existing subscription auth exposed and apps/plugins/browser/computer-use/multi-agent disabled", + "availability": "portable redacted summary" + }, + { + "kind": "native_usage", + "command_or_action": "Read the correlated thread/tokenUsage/updated event from the successful review turn", + "outcome": "52,271 input tokens; 1,429 output tokens; 53,700 total tokens; source native_reported", + "availability": "portable redacted summary" + } + ], + "candidate_sha256": "2b3bc90aee56082b4df943e4f5877c83ee41fb75a760980d1364e06820efabe7", + "successful_private_probe_receipt_sha256": "a9166be2c2e1a7946d5288a83685d96049e69ccba833a0634d1a4844ad23dd53", + "terminal_report_sha256": { + "artifact-manifest.json": "b2a84aea8419b451bc12ce3c1e9f00be4ecc16b70475366c166fc458af13850e", + "events.jsonl": "24442a87aedc085fcfc038d5382870105a585d8642aa4f8b72e524b21c270529", + "receipt.json": "ba1e13861f4e5b02e6ed22e2118b6a4e125872c7452eb2ebea3007408130f2a3", + "receipt.md": "802594961f53387bfa201fdf4bbb9af871f422e6f835309314f37aa3738b36dd", + "result-receipt.json": "ba1e13861f4e5b02e6ed22e2118b6a4e125872c7452eb2ebea3007408130f2a3" + }, + "failed_probe_evidence": [ + { + "receipt_sha256": "7e096bb8e47cd2081067f535fae4bc71616d1b9c3386927a8024840a82388026", + "outcome": "codex-cli 0.135.0 could no longer decode the current server model catalog after max reasoning metadata was introduced; no valid review evidence was accepted" + }, + { + "receipt_sha256": "bd7d43e7900810307eaba7b5400d218be0427d7204c4222f7b5bd67a3ae1f5fb", + "outcome": "codex-cli 0.153.4 reached the provider but strict structured output rejected the untyped schema_version property with invalid_json_schema; no valid review evidence was accepted" + } + ], + "residual_blockers": [ + "Materialize waiting handoff JSON and Markdown reports", + "Generate rich terminal reports for failures before host disposition", + "Implement and verify the configured headless-lead disposition path", + "Complete an independent M3 acceptance audit before marking the milestone complete" + ] +} diff --git a/docs/plans/engineering-team/evidence/M4-local-mcp-2026-09-23.json b/docs/plans/engineering-team/evidence/M4-local-mcp-2026-09-23.json new file mode 100644 index 0000000..f5d8fd7 --- /dev/null +++ b/docs/plans/engineering-team/evidence/M4-local-mcp-2026-09-23.json @@ -0,0 +1,111 @@ +{ + "schema_version": 1, + "milestone": "M4", + "status": "blocked", + "implementation_revision": "aa0fef5e6e734f51bb1ba6fbcfa9e659daea6e01", + "recorded_at": "2026-09-23T06:59:59+05:30", + "offline_gate": { + "command": "PYTHONDONTWRITEBYTECODE=1 python3 -W error::ResourceWarning -m unittest discover -s test/core -q", + "result": "212 tests passed; the two optional-SDK tests were skipped in the dependency-free base environment", + "optional_sdk_command": "PYTHONDONTWRITEBYTECODE=1 ~/.devsquad/releases/0.1.0+5b82f53/venv/bin/python -W error::ResourceWarning test/core/test_mcp.py -q", + "optional_sdk_result": "22 tests passed, including official in-memory schema calls and a real stdio-client disconnect while a detached worker survived to success", + "bash_command": "bash test/run.sh", + "bash_result": "10 test files and 208 assertions passed", + "clock_note": "One earlier Bash pass reported portable timeout elapsed 369s while the complete suite wall time was 25s; the isolated test and two subsequent complete Bash gates passed without a code change" + }, + "installed_runtime": { + "launcher": "~/.local/bin/squad", + "immutable_release": "~/.devsquad/releases/0.1.0+aa0fef5/venv/bin/squad", + "devsquad_core": "0.1.0", + "mcp_sdk": "2.2.0", + "pip_check": "passed", + "doctor": "ready; 4 installed local apps and 4 matching registrations" + }, + "installed_hosts": { + "codex": { + "version": "codex-cli 0.153.4", + "executable": "/Applications/ChatGPT.app/Contents/Resources/codex", + "registration": "matching", + "live_operation": "squad_status succeeded" + }, + "claude_code": { + "version": "2.1.220", + "registration": "matching", + "live_operation": "blocked: normal provider login required" + }, + "antigravity": { + "version": "1.2.3", + "registration": "matching", + "live_operation": "blocked: headless scoped trust/permission unresolved" + }, + "grok": { + "version": "0.2.111", + "registration": "matching", + "live_operation": "blocked: provider authentication expired" + } + }, + "registration_evidence": { + "first_setup": "all four hosts updated to the immutable aa0fef5 release and re-inspected matching", + "second_setup": "all four hosts returned unchanged; configuration hashes were unchanged", + "codex_version_drift": "PATH Codex 0.135.0 could not parse the desktop app's ultra reasoning setting; the explicit template now prefers bundled Codex 0.153.4 without changing the global setting" + }, + "cross_surface_run": { + "run_id": "c88c025e-9dc8-4b5a-ac3b-1d1d0c4c9c20", + "terminal_start_state": "awaiting_host", + "terminal_start_version": 14, + "terminal_state": "succeeded", + "terminal_version": 22, + "packet_sha256": "0b3d04320f3cb1404c3364a0b6fafa6d7d2fd2cd51d58ef601bb860b27fd0453", + "candidate_sha256": "4c911e531fda1d2110294f6485755bc2ddff4d9c414799c48c4a1d6414e828f1", + "event_count": 22, + "same_run_id": true, + "preclaim_ledgers_identical": true, + "competing_claim_error": "CONFLICT", + "sdk_client_surfaces": [ + "codex-app", + "claude-code", + "antigravity" + ], + "private_probe_receipt_sha256": "1d0e5aab27ca18934c9b0b8cf02e5141d0269304121a0f79b6c70ff9eff2f05a", + "terminal_report_sha256": { + "artifact-manifest.json": "754e08a762beed0fc6ee37f9c8e69b580fdeddc3a0d3d96ad1cd0e7364fcb7f6", + "events.jsonl": "748b46de74791af1b01735be8c0636ce54d365806684d83b6dcafc38ed5faec9", + "receipt.json": "4e214880b8f5a6248060bbaf2b62dc46d4d05cafda21404d0ad6fd4bcad956ff", + "receipt.md": "db6a6576653f5f2c2ce0a3f680c4d83ca2b884445f04bffb56810e59c29560da", + "result-receipt.json": "4e214880b8f5a6248060bbaf2b62dc46d4d05cafda21404d0ad6fd4bcad956ff" + } + }, + "codex_live_receipt": { + "successful_jsonl_sha256": "81bd8e9e14451afd93ebfc467c4a5dfc9919b1551699cdb1d3cc66bd6454e893", + "thread_id": "01a0cbdf-b599-71b1-81b9-4eb1df5cd92c", + "model": "gpt-5.5", + "effort": "low", + "permission": "read-only sandbox plus one invocation-scoped approval for squad_status only", + "tool": "squad_status", + "tool_result": { + "run_id": "c88c025e-9dc8-4b5a-ac3b-1d1d0c4c9c20", + "state": "awaiting_host", + "version": 14, + "next_action": "claim_handoff" + }, + "native_usage": { + "input_tokens": 56371, + "cached_input_tokens": 9344, + "output_tokens": 296, + "reasoning_output_tokens": 161 + }, + "failed_approval_probe": { + "jsonl_sha256": "e91644f7d61292c58366682fd3a1f427ed3363eeccfc4bd051daf5de87149c86", + "outcome": "Codex loaded DevSquad and selected squad_status, but non-interactive approval policy blocked the unannotated tool before execution", + "native_usage": { + "input_tokens": 56693, + "cached_input_tokens": 38016, + "output_tokens": 165, + "reasoning_output_tokens": 20 + }, + "closure": "aa0fef5 publishes read-only/idempotent/open-world/destructive annotations for all nine tools; the successful live retry additionally used a one-invocation squad_status approval override" + } + }, + "blocking_gate": "M4 cannot be marked complete until a normally authenticated Claude Code app/CLI performs the required real handoff operation against the saved runtime; SDK-labeled client evidence is retained but is not substituted for that live host proof", + "next_independent_work": "Proceed to M5, which depends on accepted M3 rather than completion of the externally blocked M4 live gate" +} diff --git a/docs/plans/engineering-team/evidence/M6-D2-jev-pilot-2026-10-01.json b/docs/plans/engineering-team/evidence/M6-D2-jev-pilot-2026-10-01.json new file mode 100644 index 0000000..cac3c82 --- /dev/null +++ b/docs/plans/engineering-team/evidence/M6-D2-jev-pilot-2026-10-01.json @@ -0,0 +1,77 @@ +{ + "schema_version": 1, + "work_package": "M6-D2", + "status": "complete_smoke_only", + "executed_at": "2026-10-01T20:35:46.511565+00:00", + "implementation_revision": "0d37764", + "experiment_id": "jev-devsquad-routing-pilot-v1", + "spec_sha256": "473aed2c431bd1ce8218124a6fe3fd3139fdf869468a72cbc8cf184c428d9822", + "probe_sha256": "cf665296d40d2c7b556efbfd8e8cc619e583e92f70715ee2c8eccfbecd80765b", + "request_sha256": "ae5d7f39d64cd35cd8e6c11ad78a8b00f4264755d8fe1cd557c64d80980e32e7", + "command": "python3 test/core/probes/jev_decision_eval.py --env-file .env --execute --output /Users/Dikshant/.devsquad/private-probes/jev-pilot-v1-20261001T203515Z.json", + "result": { + "exit_code": 0, + "client_request_count": 1, + "retries": 0, + "requested_model": "jev-1.13.0", + "observed_model": "jev-1.13.0", + "elapsed_ms": 187, + "cases": 8, + "questions": 24, + "input_tokens": 5373, + "output_tokens": 1762, + "task_family": {"correct": 8, "total": 8}, + "specialist_skill": {"correct": 8, "total": 8}, + "execution_tier": {"correct": 6, "total": 8} + }, + "pricing": { + "rechecked_on": "2026-10-01", + "usd_per_million_input_tokens": 0.042, + "output_tokens_charge": "free per current published pricing", + "documented_max_request_cost_usd": 0.002688, + "estimated_cost_usd_from_reported_usage": 0.00022567, + "authorized_ceiling_usd": 0.01, + "invoice_verified": false, + "source": "https://docs.typesafe.ai/models" + }, + "api_conformance": { + "source": "https://docs.typesafe.ai/api", + "rechecked_on": "2026-10-01", + "validated": "Pinned model, exact answer IDs, typed choices, finite probability distributions, confidence bounds and reported usage" + }, + "tier_disagreements": [ + { + "case_id": "T07", + "expected": "frontier_analysis", + "choice": "economy_read", + "confidence": 0.92 + }, + { + "case_id": "T08", + "expected": "frontier_write", + "choice": "standard_write", + "confidence": 0.23 + } + ], + "private_receipt": { + "path": "/Users/Dikshant/.devsquad/private-probes/jev-pilot-v1-20261001T203515Z.json", + "sha256": "fcf9e71f16e93f0b96cbf162d23ed92909c52cb3792e4b10a0c1c2a1f72e31c6", + "permissions": "600", + "contents": "Redacted scored synthetic result; no API key or task text" + }, + "checkpoint_verification": { + "bash": "11 files, 227 assertions passed", + "json_generated_reference_whitespace": "passed", + "credential_protection": "Local .env remains ignored and untracked; no key stored in this evidence", + "source_changes": "Documentation/evidence only; no full Python rerun. The probe's unchanged source retains its 14 passing focused tests; R3b.1 full integration remains pending." + }, + "decision": { + "runtime_mode": "off", + "active_policy_changed": false, + "scope": "One authenticated typed synthetic smoke, not production quality or speedup proof", + "cost_or_access_laya_trigger": false, + "quality_gate": "The smoke has no frozen numeric adoption threshold; two tier disagreements do not establish a passed production gate or automatically authorize a new comparison", + "additional_hosted_requests_authorized": 0, + "next_action": "Finish R3-R6 runtime repairs; predeclare corpus, held-out thresholds and budget before separately authorized shadow/adoption work. Do not repeat this pilot or install Laya without its declared trigger." + } +} diff --git a/docs/plans/engineering-team/evidence/M7-installed-runtime-2026-09-29.json b/docs/plans/engineering-team/evidence/M7-installed-runtime-2026-09-29.json new file mode 100644 index 0000000..5c03abf --- /dev/null +++ b/docs/plans/engineering-team/evidence/M7-installed-runtime-2026-09-29.json @@ -0,0 +1,116 @@ +{ + "schema_version": 1, + "milestone": "M7", + "status": "in_progress", + "implementation_revision": "ccba125e342041ff2ef72fd7e0c9a9b44c356e7d", + "recorded_at": "2026-09-29T10:52:08Z", + "offline_gate": { + "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core", + "result": "300 tests passed with 2 optional-SDK skips", + "bash_command": "bash test/run.sh", + "bash_result": "11 test files and 227 assertions passed", + "installer_focus": "8 fresh-install tests passed, including offline synthetic wheels, one-line JSON and pip check", + "installed_sdk_focus": "22 tests passed against the official mcp==2.2.0 environment" + }, + "standalone_install": { + "launcher": "/Users/Dikshant/.local/bin/squad", + "immutable_release": "/Users/Dikshant/.devsquad/releases/0.1.0-py31214-4c5f034c67bf-mcp-a26bc88afbef", + "release_manifest_sha256": "513cb11ff54fae31cb994c2e8931cd7cd86476642c92d92bcc1cbf3d0ed5e5f8", + "source_digest": "4c5f034c67bf1f83a185df0a8910058f38bfdfc7b38e4430923ab1167d08a048", + "mcp_environment_digest": "a26bc88afbef7b5a1ed23014ebece205902a8c43c047fec7afb893da945e66b2", + "python": "3.12.14", + "mcp_sdk": "2.2.0", + "pip_check": "passed", + "second_install": "changed=false with source/plugin/installed drift all false", + "doctor": "ready; four installed local apps and four matching registrations", + "second_setup": "all four hosts unchanged" + }, + "packaging_and_migration": { + "fresh_without_claude": "passed", + "legacy_plugin_contents": "passed", + "legacy_launcher_migration": "known DevSquad release symlink adopted; arbitrary launcher refused", + "active_run_update_survival": "passed with the old immutable release retained", + "duplicate_registration": "second setup changed no host", + "json_contract": "optional MCP install emitted exactly one JSON line", + "dependency_integrity": "installer now runs pip check before publishing an MCP release" + }, + "fresh_home_walkthrough": { + "home": "private temporary directory outside the repository", + "base_install": "documented composite --core-only command succeeded without Claude and squad --version returned 0.1.0", + "mcp_update": "documented offline wheelhouse command selected a second immutable release", + "dry_run": "all four installed hosts returned would_add without mutation", + "setup": "all four hosts returned added and matching inside the temporary home", + "doctor": "ready with four installed/configured hosts and exact mcp==2.2.0", + "setup_dry_run_sha256": "3d282aa80d94e3f7b8acbf32ee94bca21bc37aa6c8ffca4431cfb3a68a7ec2ef", + "setup_sha256": "104aa74642aa8b523b3e14d7a37d66fd57897d1ad4c9ced62b1748df05866ea1", + "doctor_sha256": "15697608ea6d6a1cf9be21017de7d698cfbf1e496947fa20c7da75b2d1cbe936", + "finding": "the required normal squad review/fix/council entry commands are absent; M7 remains in progress while the shared task-entry layer is implemented" + }, + "codex_compatibility": { + "binary": "/Applications/ChatGPT.app/Contents/Resources/codex", + "version": "codex-cli 0.155.0-alpha.9.2", + "native_probe": "initialize and complete model/list succeeded without a model generation; seven models, one default, reasoning-effort metadata present", + "resolver": "verified bundled binary wins over unverified PATH codex-cli 0.135.0", + "status": "supported" + }, + "cross_surface_run": { + "run_id": "f121e96b-502a-4986-a3c3-7ba8ab418bb3", + "terminal_start_state": "awaiting_host", + "terminal_start_version": 14, + "observed_release": "0.1.0-py31214-6a705bf7d65d-mcp-a26bc88afbef", + "terminal_cancel_state": "cancelled", + "terminal_cancel_version": 20, + "handoff_status": "cancelled", + "private_probe_state_sha256": "b44f0efa04b1a3a52f316b530d24bb7b15ffff77785ad98f9a3e46e907eb3c85", + "same_run_id_across_surfaces": true + }, + "antigravity_live_receipt": { + "version": "1.2.13", + "model": "gemini-3.8-flash-low", + "permission": "project-scoped mcp(devsquad/squad_status) only", + "tool": "squad_status", + "tool_result": { + "run_id": "f121e96b-502a-4986-a3c3-7ba8ab418bb3", + "state": "awaiting_host", + "version": 14, + "next_action": "claim_handoff" + }, + "duration_seconds": 42.290051, + "usage": { + "input_tokens": 29629, + "cache_read_tokens": 20388, + "output_tokens": 170, + "thinking_tokens": 0 + }, + "private_result_sha256": "6cab24e6905c5e4e35c2218320b548efc4dfc4835d4e96e21f8720b4d2cb8fec" + }, + "codex_live_receipt": { + "version": "codex-cli 0.155.0-alpha.9.2", + "model": "gpt-6-luna", + "effort": "low", + "session": "ephemeral; user configuration and history ignored; only the DevSquad MCP server configured", + "permission": "read-only sandbox with approval policy never", + "tool": "squad_status", + "tool_result": { + "run_id": "f121e96b-502a-4986-a3c3-7ba8ab418bb3", + "state": "awaiting_host", + "version": 14, + "next_action": "claim_handoff" + }, + "usage": { + "input_tokens": 56834, + "cached_input_tokens": 47360, + "output_tokens": 148, + "reasoning_output_tokens": 0 + }, + "private_events_sha256": "67fb9ba8061f8fe5a8845f53c1330e4dc3ae64e9cd25d62cad6dcf7ed9bb9d9b", + "private_last_message_sha256": "739df4cb90c4b2d1d3604b2456e15ae0527cebd9081d2584929292b95abb3f3a" + }, + "remaining_live_gates": { + "normal_task_entry": "squad review and squad fix are not yet implemented; squad council belongs to the separately gated C1 extension", + "claude_code": "M4 handoff and M5 implementation-to-Codex delivery require normal Claude login; claude auth status reports loggedIn=false", + "grok_build": "registration matches, but provider authentication is expired", + "installed_delivery": "the required genuine installed Claude implementation followed by a different-model Codex review/check/disposition remains blocked with M5", + "jev": "the optional capped M6-D2 measurement remains blocked on TYPESAFE_API_KEY and is not required for normal routing" + } +} diff --git a/docs/plans/engineering-team/evidence/M7-live-codex-review-2026-09-29.json b/docs/plans/engineering-team/evidence/M7-live-codex-review-2026-09-29.json new file mode 100644 index 0000000..48fec8c --- /dev/null +++ b/docs/plans/engineering-team/evidence/M7-live-codex-review-2026-09-29.json @@ -0,0 +1,128 @@ +{ + "schema_version": 1, + "milestone": "M7", + "status": "blocked_external", + "implementation_revision": "b1d52adae6bf11c547acb34f3b78c3d2853211be", + "recorded_at": "2026-09-29T12:13:54Z", + "purpose": "Prove the installed normal-entry Codex review path, including trusted checks in the detached worker environment and an artifact-bound host disposition.", + "repairs": [ + { + "revision": "1553768", + "outcome": "A terminal turn-items fallback recovers completed Codex output when delta delivery is incomplete." + }, + { + "revision": "09e435c", + "outcome": "Empty and malformed output failures retain redacted structural protocol diagnostics without model text." + }, + { + "revision": "9d70888", + "outcome": "The last completed agent message is authoritative, preventing draft and final structured messages from being concatenated." + }, + { + "revision": "b1d52ad", + "outcome": "Every trusted check receives a fresh temporary HOME for compatibility and credential isolation; the home is removed after the check." + } + ], + "offline_gate": { + "python_command": "PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core", + "python_result": "317 tests passed with 2 optional-SDK skips; one interpreter-finalization SQLite ResourceWarning was printed while the process still completed successfully", + "focused_result": "26 review-runtime tests passed, including distinct isolated HOME directories, inherited-marker isolation and cleanup", + "bash_command": "bash test/run.sh", + "bash_result": "11 test files and 227 assertions passed", + "generated_reference": "current" + }, + "installed_runtime": { + "launcher": "/Users/Dikshant/.local/bin/squad", + "immutable_release": "/Users/Dikshant/.devsquad/releases/0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", + "release_manifest_sha256": "a277b2e5f0a7b76610c9da20fe00ac05503c3d7d88d879818a7484bcd01039b9", + "source_digest": "9e5cdea2aa99be819f722f82f6e75dcf4f0ef9cf2936df017ec3c7e66bd562a1", + "mcp_environment_digest": "a26bc88afbef7b5a1ed23014ebece205902a8c43c047fec7afb893da945e66b2", + "python": "3.12.14", + "mcp_sdk": "2.2.0", + "pip_check": "passed", + "second_install": "changed=false with source/plugin/installed drift all false", + "setup": "all four hosts unchanged and matching", + "doctor": "ready" + }, + "live_managed_review": { + "run_id": "c611ad4c-5473-4f05-a870-a136f24464d3", + "exact_range": { + "base_oid": "9d70888f8e3a3510e41cb83127670d984946649b", + "target_oid": "b1d52adae6bf11c547acb34f3b78c3d2853211be", + "candidate_sha256": "a48ea624535452112cb1a71cd6c9d8452578752601fe1b9911485c6b3e18edb9" + }, + "observed_identity": { + "harness": "codex", + "harness_version": "codex-cli 0.155.0-alpha.9.2", + "model_provider": "openai", + "model_id": "gpt-5.5", + "effort": "low", + "permission_policy": "read_only", + "verification": "verified" + }, + "review": { + "verdict": "clean", + "finding_count": 0, + "artifact_id": "ce9edb9d-3596-46cb-b282-eaba707b0552", + "sha256": "5171aacdf618edb0b972be5ea752639548abc298584a478c19f635bfddc25795" + }, + "checks": [ + { + "id": "candidate-diff-check", + "status": "passed", + "returncode": 0, + "duration_ms": 12 + }, + { + "id": "detected-tests", + "command": "bash test/run.sh", + "status": "passed", + "returncode": 0, + "duration_ms": 17162, + "result": "11 test files and 227 assertions passed under the isolated trusted-check HOME" + } + ], + "checks_artifact": { + "artifact_id": "f7a49e89-5c5a-48ea-8de7-8ac6545cd71c", + "sha256": "8155b124099979e60f3a02cd3e56eb4d31c099157e8388a71272d7045ba69ae8" + }, + "evaluation_artifact": { + "artifact_id": "87c5669c-7f32-4422-8108-9dc65ca1aacd", + "sha256": "47433aa1c55053d36fe2fca0a49e93fbb7e7780c2cbc8dd33b7ba41729a51602", + "accept_allowed": true, + "required_checks_passed": true + }, + "attempt_artifact": { + "artifact_id": "82277739-0922-4cde-8cfe-6585aa7f96b6", + "sha256": "888961f482d368ee125b40f7841a8af2673d72b07099682a5be2d45838a50b6d" + }, + "native_usage": { + "input_tokens": 119440, + "output_tokens": 2786, + "total_tokens": 122226, + "source": "native_reported" + }, + "handoff": { + "packet_sha256": "0810249a84da5d0537f40e84ee532fd812bfeda231c870c3becb764d41d18d9b", + "disposition": "accept", + "submission_id": "m7-review-b1d52ad-accept", + "submission_hash": "cc3dde0d760d4e1ab140ea1722ab3a9baf3c22c3e822d2fa15858c5ba4a78185" + }, + "terminal": { + "state": "succeeded", + "version": 22, + "receipt_sha256": "0103db19a528cff916ebef80c6ec94b682ed4801fc22db020cdbc21796040f70" + } + }, + "pre_fix_live_evidence": { + "run_id": "228db373-ff4f-4a1d-a692-7dff255036b5", + "receipt_sha256": "2431590ad0bbe4b4bff840f6f673f73671164236cb3933513838e0f65465bce6", + "outcome": "The review itself was clean, but the detected Bash suite failed because the detached trusted-check environment omitted HOME. The host rejected that evidence before b1d52ad." + }, + "remaining_live_gates": { + "claude_code": "Normal Claude login is still required for the M4 handoff and genuine M5 Claude implementation followed by different-model Codex review/check/disposition.", + "grok_build": "Registration matches, but a supported live operation requires renewed authentication.", + "jev": "M6-D2 remains blocked on TYPESAFE_API_KEY; Laya runs only if the declared Jev fallback trigger fires." + }, + "redaction": "Raw provider messages, private diagnostics and credentials remain outside tracked evidence." +} diff --git a/docs/plans/engineering-team/evidence/M7-normal-entry-2026-09-29.json b/docs/plans/engineering-team/evidence/M7-normal-entry-2026-09-29.json new file mode 100644 index 0000000..67a92db --- /dev/null +++ b/docs/plans/engineering-team/evidence/M7-normal-entry-2026-09-29.json @@ -0,0 +1,51 @@ +{ + "schema_version": 1, + "milestone": "M7", + "status": "blocked_external", + "implementation_revision": "85aa378cf990687a9889409674bfa530fe4890a7", + "recorded_at": "2026-09-29T11:17:22Z", + "normal_entry": { + "review": "resolves exact base/target commits, uses read-only Codex review and report-only checks, and requires no task/profile/policy JSON", + "fix": "uses one isolated Claude writer, different-harness Codex review, required checks, bounded revisions and host lead disposition", + "routing": "strict embedded profile registry and policy are canonicalized, hash-frozen and retained in the saved run", + "catalog": "Codex model and effort are selected from a complete native model/list without a model generation", + "output": "returns role/profile selection, selection reason, scope, checks, run ID/state and a next command without exposing model IDs in the normal response", + "wait_behavior": "fix --wait resumes the saved candidate-review phase once and returns at the host handoff" + }, + "offline_gate": { + "python_command": "PYTHONWARNINGS=error::ResourceWarning PYTHONDONTWRITEBYTECODE=1 PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core -v", + "python_result": "311 tests passed with 2 optional-SDK skips", + "bash_command": "bash test/run.sh", + "bash_result": "11 test files and 227 assertions passed", + "managed_delivery": "one isolated implementation writer, frozen candidate, independent review, required checks, unchanged source checkout and host handoff passed with offline fixtures", + "generated_reference": "current" + }, + "installed_runtime": { + "launcher": "/Users/Dikshant/.local/bin/squad", + "immutable_release": "/Users/Dikshant/.devsquad/releases/0.1.0-py31214-129b9107ed59-mcp-a26bc88afbef", + "release_manifest_sha256": "99debfc9f15c27dc31c13178a91ca4fa05e7aefd75648d3aa519d35ffffa4f63", + "source_digest": "129b9107ed59141d1da8462ce47f16a70f478d49de88a2822a187864816aec74", + "mcp_environment_digest": "a26bc88afbef7b5a1ed23014ebece205902a8c43c047fec7afb893da945e66b2", + "python": "3.12.14", + "mcp_sdk": "2.2.0", + "pip_check": "passed", + "second_install": "changed=false with source/plugin/installed drift all false", + "setup": "all four hosts unchanged and matching", + "doctor": "ready" + }, + "installed_entry_checks": { + "review": { + "command_shape": "squad review --base HEAD^ --target HEAD --model gpt-6-luna --effort low --dry-run --json", + "result": "exact commits, read-only reviewer, detected checks and deterministic idempotency key returned; no run or generation created" + }, + "fix": { + "command_shape": "squad fix ISSUE --write-path PATH --check COMMAND --review-model gpt-6-luna --review-effort low --dry-run --json", + "result": "bounded write scope, Claude implementer, different-harness Codex reviewer, detected/user checks and deterministic idempotency key returned; no run or generation created" + } + }, + "remaining_live_gates": { + "claude_code": "M4 handoff and M5 installed implementation-to-Codex delivery require normal Claude login", + "grok_build": "registration matches, but provider authentication is expired", + "jev": "M6-D2 remains blocked on TYPESAFE_API_KEY; Laya runs only if the declared Jev fallback trigger fires" + } +} diff --git a/docs/plans/engineering-team/evidence/R1-candidate-integrity-2026-09-29.json b/docs/plans/engineering-team/evidence/R1-candidate-integrity-2026-09-29.json new file mode 100644 index 0000000..2c4846f --- /dev/null +++ b/docs/plans/engineering-team/evidence/R1-candidate-integrity-2026-09-29.json @@ -0,0 +1,43 @@ +{ + "schema_version": 1, + "work_package": "R1", + "status": "source_verified", + "baseline_revision": "399d93d", + "recorded_on": "2026-09-29", + "scope": "Shared branch-review and issue-delivery candidate integrity through trusted checks", + "baseline_reproduction": { + "command": "python3 -m unittest discover -s test/core -p test_check_integrity_runtime.py -v", + "result": "Four mutation regressions failed on unmodified production; two permitted-build-output positive cases passed. Both host and headless paths could continue checking altered source." + }, + "acceptance_mapping": [ + {"requirement": "Public review/delivery and host/headless cannot accept source-changing report-only checks", "test": "test_check_integrity_runtime.PublicCheckIntegrityTest"}, + {"requirement": "Undeclared new source is rejected; approved untracked build artifacts remain usable", "test": "test_check_integrity_runtime.PublicCheckIntegrityTest"}, + {"requirement": "Later checks never consume a contaminated tree; exact candidate and original checkout remain unchanged", "test": "test_check_integrity_runtime.PublicCheckIntegrityTest"}, + {"requirement": "Restart/replay preserves invalidation and terminal receipt evidence", "test": "test_check_integrity_runtime.PublicCheckIntegrityTest"}, + {"requirement": "Delete, executable mode, staged/index-only edits, assume-unchanged, HEAD checkout/attachment, symlinks, ignored source and review-tree mutation are detected", "test": "test_workspaces.ReviewWorkspaceTest.test_check_integrity_covers_tracked_modes_index_head_and_hidden_changes"}, + {"requirement": "Legacy evidence stays readable but cannot authorize a new acceptance; completed submissions remain replayable", "test": "test_review_workflow.BranchReviewWorkflowTest.test_legacy_checks_remain_readable_but_cannot_be_imported_or_accepted_anew and test_check_integrity_runtime.PublicCheckIntegrityTest.test_terminal_completion_replay_preserves_historical_acceptance"}, + {"requirement": "Output declarations and v2 integrity witnesses cannot widen frozen checks", "test": "test_validation.AdversarialValidationTest.test_check_outputs_are_explicit_bounded_relative_paths and test_review_workflow.BranchReviewWorkflowTest.test_integrity_is_non_overridable_and_strictly_bound_to_frozen_outputs"}, + {"requirement": "Per-check temporary HOME remains isolated", "test": "test_review_runtime.DurableBranchReviewTest.test_each_trusted_check_gets_a_fresh_isolated_home"} + ], + "verification": { + "core": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core -v: 330 tests ran in 190.386 seconds; OK with 2 optional-SDK skips. The existing SQLite cleanup warning remains recorded.", + "bash": "bash test/run.sh: 11 files, 227 assertions passed before the R1 checkpoint; no real provider CLIs required.", + "generated_reference": "Current after packaged task/check-result schema updates", + "independent_review": "Bounded leaf review identified undeclared source and legacy replay gaps, both repaired; final read-only R1 audit reported no further actionable issues.", + "earlier_runs": "An intermediate 71-test run overlapped schema edits and is not verification. The first stable 330-test gate found two path-assertion mistakes and one existing fixture needing an explicit generated-output declaration; none is counted as a pass." + }, + "source_sha256": { + "plugin/core/src/devsquad/review_worker.py": "883717bb8e72ed714e5f3362bb9649b549b6ab3ea9568c47a10087278a63ca55", + "plugin/core/src/devsquad/workspaces.py": "5b045ee74935ee69eb27f93c959a6cb8a5b3e366a1ea146bd75b3fb65eeedfe0", + "plugin/core/src/devsquad/workflows.py": "449237aa230e6a778be42f21f931106beeb15326f1a4fd5054b99fce5e47b7dc", + "plugin/core/src/devsquad/service.py": "1f019f4603c14098a93984db1a7d5c48083081687b0461936bdbed9aa5b2a3ae", + "plugin/core/src/devsquad/reports.py": "11299daaf9a33e20fbb2bbda1983ed730702b67d1292be330d5e8f0b33228ec4", + "plugin/core/src/devsquad/validation.py": "0ed79e40c762d3ab609b5553296135620529fd874f5cc8a5c8c83ef8e62cd5b2" + }, + "limitations": [ + "No real provider generation or installed runtime replacement occurred; refreshed installed/live proof belongs to R8.", + "Integrity is checked at trusted command boundaries, not as a sandbox against deliberate mutate-and-restore inside one invocation.", + "Git submodule candidate inputs fail closed pending explicit support.", + "The existing unclosed SQLite ResourceWarning remains assigned to R5." + ] +} diff --git a/docs/plans/engineering-team/evidence/R2-identity-red-baseline-2026-09-29.json b/docs/plans/engineering-team/evidence/R2-identity-red-baseline-2026-09-29.json new file mode 100644 index 0000000..b8d9f37 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R2-identity-red-baseline-2026-09-29.json @@ -0,0 +1,39 @@ +{ + "schema_version": 1, + "work_package": "R2", + "recorded_on": "2026-09-29", + "status": "red_baseline_only", + "production_revision": "6848f1156bee10e1ef2617402dfc7eaebdc58076", + "production_changed": false, + "provider_calls": 0, + "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core -p test_claude_identity.py -v", + "result": { + "exit_code": 1, + "test_methods_run": 16, + "assertion_or_subtest_failures": 39, + "errors": 4, + "elapsed_seconds": 3.162, + "error_cause": "Non-object JSON reaches provider_document.get and raises AttributeError instead of ContractError" + }, + "sha256": { + "test/core/test_claude_identity.py": "c18f2f91ff547f2e3c90902bb3516550e20c39e8c9a01f7c8b8ec3ad8f2a2d99", + "plugin/core/src/devsquad/claude_delivery_worker.py": "c89bca4fa04e01541d1877c8ee8511df724e46dc00b6e172add32d03094de18c", + "plugin/core/src/devsquad/workflows.py": "449237aa230e6a778be42f21f931106beeb15326f1a4fd5054b99fce5e47b7dc" + }, + "scope": "Direct worker conformance using only a temporary fake CLI; not public delivery, installed runtime, or live-provider proof", + "handoff_checks": { + "bash_test_run": "227 assertions across 11 files passed with process inspection permitted; initial sandbox-stalled run stopped with exit 143 and is not a pass", + "generated_reference": "current", + "json_validation": "passed", + "diff_check": "passed", + "full_core_suite": "not rerun for this planning checkpoint; new focused R2 regressions are known to fail" + }, + "remaining": [ + "Repair native parser and distinguish requested settings from reported identity", + "Add strict evidence-import and actual-model independent-review public regressions", + "Retain safe failed native evidence in durable receipts and test recovery/replay", + "Run the repaired focused and full offline gates plus bounded independent patch review", + "Obtain a corrected live Claude-to-different-model-Codex receipt after normal Claude login" + ], + "completion_claim": "R2 remains in progress. The earlier 330-test R1 green gate predates this failing test file; it is not a current-tree pass." +} diff --git a/docs/plans/engineering-team/evidence/R2-observed-identity-2026-10-01.json b/docs/plans/engineering-team/evidence/R2-observed-identity-2026-10-01.json new file mode 100644 index 0000000..5fc8724 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R2-observed-identity-2026-10-01.json @@ -0,0 +1,88 @@ +{ + "schema_version": 1, + "work_package": "R2", + "status": "source_verified", + "implementation_revision": "ef988897e2dda495482dd26e7c583bc428086ba0", + "recorded_on": "2026-10-01", + "scope": "Claude native reported-model identity, actual-attempt import binding, different-model delivery acceptance and durable failed evidence; offline proof only", + "baseline_reproduction": { + "revision": "6848f11", + "artifact": "R2-identity-red-baseline-2026-09-29.json", + "result": "16 worker tests exposed 39 assertion/subtest failures and four errors before the repair. The original red artifact is preserved." + }, + "acceptance_mapping": [ + { + "requirement": "Concrete reported identity and supported family aliases are validated; missing, wrong, malformed or multiple native model identities fail closed; unreported effort/revision stays unknown", + "tests": "test/core/test_claude_identity.py: 16 tests" + }, + { + "requirement": "Strict JSON, typed native metadata and session/usage/provenance binding reject tampered evidence before candidate finalization", + "tests": "test/core/test_implementation_identity_evidence.py: 12 tests including the exact-attempt and native-byte regressions" + }, + { + "requirement": "The imported selected profile must equal the durable attempt's frozen profile, not merely another allowed fallback", + "tests": "test_implementation_identity_evidence.test_import_must_match_actual_attempt_not_another_allowed_fallback" + }, + { + "requirement": "Invalid UTF-8 and CRLF output retain their original native byte hashes; invalid stderr cannot preempt a valid result", + "tests": "test_implementation_identity_evidence.test_invalid_utf8_stdout_retains_original_native_byte_hash, test_failed_result_hash_preserves_crlf_native_bytes, test_invalid_utf8_stderr_cannot_preempt_valid_result_parsing" + }, + { + "requirement": "Exact and alias-resolved identities complete public delivery; host and headless acceptance reject the same reported model behind different labels", + "tests": "test/core/test_delivery_identity_runtime.py: exact/alias positive cases and host/headless independence cases" + }, + { + "requirement": "Invalid identity cannot publish a candidate; failed diagnostics, native usage and stream references survive fallback, cancellation and replay", + "tests": "test_delivery_identity_runtime.test_invalid_identity_never_publishes_candidate_and_retains_failed_evidence, test_native_fallback_keeps_failed_identity_and_usage_in_success_receipt, test_cancellation_keeps_prior_failed_native_identity_and_usage" + }, + { + "requirement": "Historical v1 receipts remain readable and exact completed acceptance replays; unverified old evidence cannot authorize new acceptance", + "tests": "test_delivery_identity_runtime.test_legacy_identity_cannot_authorize_new_acceptance_but_rejection_replays and test_historical_successful_legacy_receipt_reads_and_exact_acceptance_replays" + } + ], + "verification": { + "core": { + "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core -v", + "result": "OK (skipped=2)", + "tests_run": 367, + "failures": 0, + "errors": 0, + "skipped": 2, + "suite_seconds": 215.886, + "wrapper_wall_seconds": 216.021, + "wrapper_monotonic_seconds": 216.020, + "exit_code": 0, + "notes": "The command ran inside a subprocess timing wrapper. The source stayed unchanged during discovery. Two official MCP SDK conformance tests skipped because the optional SDK is absent in this interpreter." + }, + "bash": "bash test/run.sh: 11 test files, 227 assertions passed before the planning/evidence checkpoint", + "generated_reference": "python3 scripts/generate-core-reference.py --check: current", + "independent_review": "A bounded source audit found actual-attempt profile binding and exact native-byte retention gaps. Both were reproduced and repaired; a follow-up independently ran four targeted regressions successfully and reported no additional actionable finding in those fixes.", + "earlier_runs": [ + "An earlier 363-test integration run passed with two optional-SDK skips before the last two audit fixes; it is not final-source proof.", + "The first final-source 367-test run completed in 230.999 seconds with six failures and two skips. Four failures showed preflight budget expiry or approximately 15-17 minute UTC gaps, supporting a timing explanation but not establishing the cause of all six failures.", + "All six failing cases reran unchanged successfully in 18.998 seconds with per-test UTC and monotonic elapsed measurements agreeing. The subsequent complete 367-test gate above passed." + ], + "warning": "An ignored finalizer exception reported an unclosed SQLite connection during test_running_cancel_race_reaps_runner_and_worker_once. PYTHONWARNINGS=error::ResourceWarning did not turn that finalizer exception into a failing suite. Cleanup remains R5; this result is not warning-free." + }, + "source_sha256": { + "plugin/core/src/devsquad/claude_identity.py": "765c09b06d9005e88ce14ad735ff033ace88e29bd9c2c5abb77bb45c29edc8c7", + "plugin/core/src/devsquad/claude_delivery_worker.py": "62493f26927c8bd24bce2b39a9ac604290d5fb656f5eeac428eb982a24367450", + "plugin/core/src/devsquad/workflows.py": "6e0be346a78a50927913762fe78e989a3b73f56f84808fdbda3886de7edf00ef", + "plugin/core/src/devsquad/supervisor.py": "b9732da56ec43cb9235c36ed19c9b982bd81674b4c04dfddba9d76ae9af8d89f", + "plugin/core/src/devsquad/service.py": "07671078c8ac337778ac1e0ba10b1475074b5dfcb3cd0bad030b44d5d54e86c2", + "plugin/core/src/devsquad/reports.py": "df30017e12cb2fd4cd1b39576bb5a2296365ce0611b961ab3a7c437d6d71fef6" + }, + "test_sha256": { + "test/core/test_claude_identity.py": "c18f2f91ff547f2e3c90902bb3516550e20c39e8c9a01f7c8b8ec3ad8f2a2d99", + "test/core/test_implementation_identity_evidence.py": "42ea920c2c31a3df2fd10477d1521f789a69bfbd5ee381fdd20f88e2ad98515c", + "test/core/test_delivery_identity_runtime.py": "b107f3a1c5b4a25b8d9c21bcf76f42d19598ea08b631eebd12fda8b1a6ba044a" + }, + "limitations": [ + "This checkpoint records already implemented source and completed offline verification; the planning update made no production source changes.", + "No real provider generation, installation refresh, authentication reprobe or global provider-setting change occurred. The installed runtime predates R1/R2; affected installed/live proof remains R8.", + "Native modelUsage establishes the bounded reported-model scope only. Unreported effective effort/backing revision remains null; pricing canonicalModel is not serving identity authority; multiple entries do not identify a unique writer.", + "The genuine Claude implementation to different-model review receipt and real Claude host handoff remain blocked on last-observed normal Claude login. No fresh authentication check is claimed.", + "R3-R6 integration, C1/R7, other R8 gates and the SQLite cleanup warning remain open. This is not full M5 or product completion.", + "Raw diagnostics stay outside tracked evidence; this portable record contains only source hashes, test results and bounded observations." + ] +} diff --git a/docs/plans/engineering-team/evidence/R3-audit-repairs-2026-10-01.json b/docs/plans/engineering-team/evidence/R3-audit-repairs-2026-10-01.json new file mode 100644 index 0000000..96204fa --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3-audit-repairs-2026-10-01.json @@ -0,0 +1,31 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-01", + "baseline_revision": "db55d3f", + "status": "source_targeted_verified_full_gate_pending", + "findings": { + "R3-001": "Delivery imports are mandatory for successful attempts; captured evidence is strictly decoded against immutable task, candidate/patch/event fences and recorded revision decisions. Review, check, evaluation and attempt imports must equal captured evidence. Failed workers require the exact hash-verified terminal attempt receipt.", + "R3-002": "Qualification rejects submitted latency/usage ratios when no verified paired saved measurement exists. Unknown remains unknown; finite gates requiring either metric block qualification.", + "check_bytecode": "Trusted checks set invocation-local PYTHONDONTWRITEBYTECODE and redirect PYTHONPYCACHEPREFIX under the isolated temporary HOME; candidate integrity allowances are unchanged." + }, + "red": { + "delivery": "Hash-consistent malformed stdout, missing imports, candidate drift and changed imports were accepted by the baseline reader; negative regressions failed.", + "ratios": "Independent native audit confirmed invented-ratio bypass. Initial finite-gate regressions exposed submitted ratios not being checked against measurements; full integration/re-review pending.", + "bytecode": "1 test failed: importing a candidate module created __pycache__ in the candidate tree." + }, + "verified": [ + {"gate": "reader/eligibility/check targeted", "tests": 22, "seconds": 71.034, "result": "pass"}, + {"gate": "delivery/task-entry/CLI/ratios/bytecode/historical installed upgrade", "tests": 37, "seconds": 37.167, "result": "pass"}, + {"gate": "Bash", "assertions": 227, "files": 11, "result": "pass"}, + {"gate": "generated reference and whitespace", "result": "pass"}, + {"gate": "schema-13 old-package active/recoverable upgrade", "result": "pass", "scope": "Temporary install defers without selector/schema changes; old active run cancels; old queued run resumes and succeeds; subsequent update migrates to schema 15, retains old release and readable result."}, + {"gate": "normal alias promotion", "result": "pass", "scope": "Four fake-native Codex protocol runs supply public paired evaluation. Promotion affects next normal entry; queued old snapshot and explicit pins remain unchanged. Not real-model qualification."} + ], + "failed_intermediate_gates": [ + {"tests": 34, "seconds": 21.036, "errors": 1, "cause": "Test asserted a paired-case verdict for an unpaired failed arm; corrected fixture runs both arms."}, + {"tests": 19, "seconds": 20.453, "failures": 1, "cause": "Public stale-evidence rejection wrapped the lower-level failed-receipt reason; corrected assertion."}, + {"tests": 1, "seconds": 6.649, "errors": 2, "cause": "Test tearDown deleted runtime before registered cleanups; corrected cleanup ordering."} + ], + "installed_state": {"schema": 13, "nonterminal_runs": 0, "refreshed": false}, + "limitations": ["Full core integration and bounded independent repair re-review pending", "No updated-installation Claude/Grok/Gemini live proof", "No verified paired latency/usage metric; required finite metric gates remain blocked", "R4 catalog/quota, R5 public controller/outcomes, R6 terminal UX and R7 Council remain separately open"] +} diff --git a/docs/plans/engineering-team/evidence/R3-closure-2026-10-02.json b/docs/plans/engineering-team/evidence/R3-closure-2026-10-02.json new file mode 100644 index 0000000..4eae4b5 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3-closure-2026-10-02.json @@ -0,0 +1,44 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "status": "accepted_source_package", + "target_revision": "39b95f1c89ca89a73148cd88dfe5565f4d5dc377", + "independent_review": { + "run_id": "f0a4f86b-b305-4839-88a2-cb5fed8739ab", + "base_revision": "672e38305fcc764d87d7da6afb5914cbf3c5a3ce", + "state": "succeeded", + "version": 22, + "candidate_sha256": "010aa83f442fb6f275c1793b9131e5769b4481fd905ad9a1e83163ac1fc7c315", + "identity": {"harness": "codex", "harness_version": "codex-cli 0.159.2", "model_id": "gpt-6.1-sol", "effort": "low", "verification": "verified", "permission_policy": "read_only"}, + "verdict": "clean", + "findings": [], + "review_sha256": "c2a03b32bc382c2a5f9985d0a71e15e847b10fbe11bca93dc65f12b3ee413ccc", + "checks_sha256": "60567b4ef724ce176ed6bdba63164d182c5ce49e94524754edab07c08b16524f", + "receipt_sha256": "b013baa9fc4979c3dd979286753131067cc8e5d44f6f37b0f3e8d48a5cceb552", + "mandatory_checks": {"diff": "passed", "bash": "passed", "focused": {"tests": 51, "seconds": 200.684, "result": "pass"}, "candidate_integrity": "verified_unchanged_for_all_checks"}, + "usage": {"source": "native_reported", "input_tokens": 337192, "output_tokens": 2173, "total_tokens": 339365} + }, + "full_integration": { + "run_id": "360a4993-d016-406e-8bb4-ae6d57dfe6cf", + "candidate_revision": "f8c4f8c83f170870eb37d564e54eee6188fc233c", + "tests": 477, + "seconds": 455.860, + "skips": 2, + "errors": 0, + "failures": 0, + "unraisable": [], + "applicability": "git diff --exit-code confirms all six R3 source-module blobs identical between the accepted full-gate candidate and the independent audit target; no duplicate full run required" + }, + "requirements": [ + {"requirement": "immutable prelaunch and independent paired assignment", "proof": ["test_experiment_saved_runs", "test_experiment_evidence_integrity.test_trial_reservation_rejects_changed_inputs_before_any_worker", "test_experiment_v2_evaluation"], "result": "pass"}, + {"requirement": "branch-review and issue-delivery stream/input/import provenance", "proof": ["test_experiment_evidence_integrity", "test_delivery_experiment_integrity"], "result": "pass"}, + {"requirement": "honest failed, missing, fallback and cancelled arms", "proof": ["test_experiment_saved_runs", "test_delivery_experiment_integrity.test_real_failed_writer_requires_an_exact_terminal_receipt"], "result": "pass"}, + {"requirement": "unmeasured ratios cannot satisfy finite qualification gates", "proof": ["test_experiment_eligibility.test_unmeasured_ratios_cannot_bypass_finite_qualification_gates"], "result": "pass"}, + {"requirement": "correction after evaluation and qualification blocks new authority atomically", "proof": ["test_experiment_eligibility.test_correction_after_evaluation_blocks_new_qualification_atomically", "test_experiment_eligibility.test_correction_after_qualification_blocks_replay_and_new_promotion", "test_experiment_eligibility.test_correction_cannot_commit_between_qualification_validation_and_commit"], "result": "pass"}, + {"requirement": "valid promotion/rollback, stale rollback and catalog fallback gates", "proof": ["test_lifecycle.test_qualification_cas_promotion_and_rollback_are_replay_safe", "test_lifecycle.test_catalog_unavailable_incumbent_uses_only_qualified_predecessor", "test_experiment_eligibility.test_stale_regression_cannot_roll_back_until_explicit_current_review", "test_experiment_eligibility.test_stale_qualified_target_blocks_regression_and_is_skipped_by_catalog_fallback"], "result": "pass"}, + {"requirement": "append-only explicit revision, historical completed replay and duplicate-outcome upgrade read/proposal paths", "proof": ["test_experiment_eligibility.test_explicit_revision_keeps_original_bytes_and_reuses_original_assignments", "test_experiment_eligibility.test_completed_decision_replay_preserves_historical_receipt_after_correction", "test_experiment_eligibility.test_upgraded_reused_outcome_history_remains_readable_but_cannot_authorize_new_decisions"], "result": "pass"} + ], + "preserved_history": ["R3-audit-repairs-2026-10-01.json", "R3b2-eligibility-partial-2026-10-01.json", "R8-upgrade-review-2026-10-01.json", "R8-installed-workflows-2026-10-01.json"], + "separate_open_scope": ["R4 catalog/quota package acceptance", "R5 automatic final outcomes and public trial controller", "R6 terminal UX/readiness", "R7 Council", "R8 whole-delivery and desktop UI acceptance"], + "qualification_note": "Public offline fixtures establish machinery/provenance, not real-model superiority; original finite sample and budget gates remain unchanged" +} diff --git a/docs/plans/engineering-team/evidence/R3-independent-audit-2026-10-01.json b/docs/plans/engineering-team/evidence/R3-independent-audit-2026-10-01.json new file mode 100644 index 0000000..864b779 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3-independent-audit-2026-10-01.json @@ -0,0 +1,21 @@ +{ + "schema_version": 1, + "work_package": "R3", + "status": "findings_open", + "base_revision": "672e38305fcc764d87d7da6afb5914cbf3c5a3ce", + "target_revision": "933419041fc8aa415dda622810e8a17fde41aac5", + "run_id": "80fdd336-07e2-445c-8a08-5877cd6534df", + "observed_identity": {"harness": "codex", "version": "codex-cli 0.159.2", "model": "gpt-6.1-sol", "effort": "high", "permission": "read_only", "verification": "verified"}, + "review_sha256": "97c4d5ccd0874fc7eaaf3a24311cb6edbba309c5abdb8ba714b517bafc8fdba8", + "attempt_sha256": "1c2b381acf2961bb96557a81b53554fb5dba8e0c7bd7965d8cbf0eff975c720b", + "findings": [ + {"id": "R3-001", "severity": "high", "path": "plugin/core/src/devsquad/experiment_evidence.py", "summary": "Delivery exposure can be accepted without semantically validated implementation/reviewer output and matching import or terminal failure evidence", "status": "open"}, + {"id": "R3-002", "severity": "medium", "path": "plugin/core/src/devsquad/store.py", "summary": "Latency/usage qualification ratios are trusted from the caller instead of verified saved measurements", "status": "open"} + ], + "checks": {"diff": "passed", "bash": "Exited 0, but invalidated by undeclared Python bytecode in the check workspace; candidate integrity correctly blocked acceptance"}, + "host_disposition": "reject", + "terminal": {"state": "failed", "version": 22}, + "native_usage": {"input_tokens": 1642079, "output_tokens": 10733, "source": "native_reported", "worker_invocations": 1, "native_model_requests": null}, + "limitations": "Source-only audit, not an updated-installation or Claude-to-Codex delivery proof. Full integration pass precedes these findings and does not close them", + "next_action": "Reproduce and repair the two bounded findings and bytecode/check friction. Do not repeat the broad review; re-review only repaired boundaries" +} diff --git a/docs/plans/engineering-team/evidence/R3a-provenance-contract-2026-10-01.json b/docs/plans/engineering-team/evidence/R3a-provenance-contract-2026-10-01.json new file mode 100644 index 0000000..46911f2 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3a-provenance-contract-2026-10-01.json @@ -0,0 +1,61 @@ +{ + "schema_version": 1, + "work_package": "R3a", + "status": "partial_source_verified", + "implementation_revision": "672e383", + "recorded_on": "2026-10-01", + "scope": "Outcome uniqueness, v2 provenance contracts, schema-14 fenced prelaunch assignments and pure normalized-chain evaluation; not end-to-end R3 closure", + "baseline_reproductions": [ + "Five new reused-outcome tests failed before production changes: repeated pair, cross-arm reuse, cross-split reuse, evaluator acceptance and invalid persisted evaluation. All five now pass.", + "The first v2 contract run had one failure and seven errors across nine tests because the new schema/module did not exist; this is a missing-feature baseline, not eight reproduced production exploits.", + "A subsequent regression confirmed that unbound v2 outcome labels could still be evaluated after schema support; requiring saved-project and normalized provenance now rejects them.", + "An independent normalized-chain test exposed the incorrect late-correction kind check; the strict valid correction now changes the evidence digest and blocks the proposal.", + "Review exposed omitted tested-role fallback policy and native execution context. Dedicated regressions failed before the fixes; fallback settings now affect pair context and each arm explicitly predeclares its profile/native execution fingerprint." + ], + "acceptance_mapping": [ + {"requirement": "No global outcome reuse across arms/cases/splits or invalid new evaluation persistence", "tests": "test/core/test_experiment_reuse.py: 5 regressions"}, + {"requirement": "Strict v2 profile/execution/assignment contracts and distinct corpus versus controlled-pair fingerprints", "tests": "test/core/test_experiment_provenance.py: 12 contract regressions"}, + {"requirement": "Assignments precede attempts under the preparation fence; rebinding, duplicate arms, changed specs and invalid input fail atomically", "tests": "test/core/test_experiment_assignment_store.py: 8 persistence/migration regressions"}, + {"requirement": "Normalized chains reject reused runs/attempts and crossed profiles/inputs even with missing partners; failures, missingness and corrections remain visible", "tests": "test/core/test_experiment_v2_evaluation.py: 10 pure evaluator regressions"} + ], + "verification": { + "focused_command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_experiment_reuse test_experiment_provenance test_experiment_assignment_store test_experiment_v2_evaluation test_learning test_lifecycle", + "focused_result": "47 tests passed in 1.470 seconds after the final source change", + "migration_checks": "Nine independently run targeted migration tests passed, including both installed-wheel upgrade checks; existing schema-version expectations were updated without changing unrelated run-version fixtures", + "independent_review": "Bounded R3a audit found tested native execution metadata was not bound. The repaired predeclared per-arm execution hash passed an independent two-test follow-up in 0.098 seconds, with no remaining concrete finding in those changes. The deferred reader/lifecycle/public-controller paths were outside this audit.", + "bash": "11 files and 227 assertions passed before the source checkpoint", + "generated_reference": "python3 scripts/generate-core-reference.py --check: current", + "core": { + "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core -v", + "result": "OK (skipped=2)", + "tests_run": 402, + "failures": 0, + "errors": 0, + "skipped": 2, + "suite_seconds": 213.208, + "wrapper_wall_seconds": 213.376, + "wrapper_monotonic_seconds": 213.376, + "exit_code": 0, + "notes": "Source remained fixed at 672e383. Two official MCP SDK tests skipped because that optional dependency is absent in this interpreter. An ignored finalizer ResourceWarning for an unclosed SQLite connection appeared during a cross-process test; it remains R5 and the suite is not warning-free." + } + }, + "source_sha256": { + "plugin/core/src/devsquad/learning.py": "4558b53cc97ac1c00a1a6f3480f757859dc4fa6164ba3a7a257658050393f8b8", + "plugin/core/src/devsquad/experiment_provenance.py": "c0e98d3371679f50ddaa0652092382b434bac46bca6a9fdb6e5b4bace34a1934", + "plugin/core/src/devsquad/store.py": "362b815eb818054c013c33e13573595222acf4b48ff62c77ee82a72e13e8345e", + "plugin/core/src/devsquad/migrations/014_experiment_assignments.sql": "e0675f9c71df3ded92467835972792a90f78ae17a4a86e5a88335eb90262afa7" + }, + "test_sha256": { + "test/core/test_experiment_reuse.py": "8f1b9916867c0192de3d10b8ea7231a44261766b93407d2f5ae20039dbdc0dc0", + "test/core/test_experiment_provenance.py": "eece72a69ae68dda7caf20205864b0deef188307ddfca8911cfcf38601f21c91", + "test/core/test_experiment_assignment_store.py": "ca1d39f1f725e5f45eabcb9ac9d0890c09f184e6e71cba372c4891df8ff7e3ab", + "test/core/test_experiment_v2_evaluation.py": "5f1da4d399082ce6510a60126d3095481ae1d4d22718b6cb18a75b96deb5046a" + }, + "remaining_requirements": [ + "R3b must build v2 chains from immutable assignment records, actual saved attempts and outcome/correction rows. Store's current public evaluation reader does not yet supply that provenance, so v2 evaluation fails closed.", + "R3b must share current-evidence eligibility across evaluation/qualification replay, new qualification, promotion, regression rollback and catalog fallback. Existing v1 label-based positives are not proof of repaired eligibility.", + "R3c must preserve readable legacy audit/proposal evidence without authorizing new decisions; schema migration preserving bytes alone does not complete that requirement.", + "Public end-to-end realistic experiment/lifecycle fixtures and the R5 paired-trial controller are still required. Pure normalized witnesses in this artifact are deliberately fabricated unit inputs, not saved-run authority or live model proof.", + "The full R1-R8/M1-M7+C1 objective remains active. No provider call, installed runtime refresh, paid API fallback or global setting change occurred." + ] +} diff --git a/docs/plans/engineering-team/evidence/R3b1-failure-path-2026-10-01.json b/docs/plans/engineering-team/evidence/R3b1-failure-path-2026-10-01.json new file mode 100644 index 0000000..3e1602b --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3b1-failure-path-2026-10-01.json @@ -0,0 +1,29 @@ +{ + "schema_version": 1, + "work_package": "R3b.1", + "status": "source_offline_verified_independent_audit_pending", + "recorded_at": "2026-10-01T20:57:36Z", + "baseline_revision": "bd0cf5c", + "source_sha256": { + "plugin/core/src/devsquad/experiment_evidence.py": "38a11bca28bcde0835dee5f5aaaef450e5bed0e951b57edeec88063d4ee558ee", + "test/core/experiment_runtime_fixture.py": "08a8cf77c34052fc23fd782041d0d272110fbc1e3fa1db6bccaad8903005eaaf", + "test/core/test_experiment_saved_runs.py": "7652cae8b40d512c8019c58edc856537f34c27d06120c974040bd4f9953558dc" + }, + "verification": { + "red": "Real terminal failed worker regression errored in strict stdout JSON decoding; 1 test in 4.198 seconds, exit 1", + "initial_green_attempt": "Decoder repair reached evaluation, then the test itself raised KeyError on the wrong case-row key; 1 test in 4.297 seconds, exit 1, not counted as a pass", + "failure_and_negative_receipts": "2 tests passed in 5.363 seconds; valid failed exposure retained, malformed or mismatched hash-valid failure receipts rejected", + "focused_command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_experiment_saved_runs test_experiment_provenance test_learning test_lifecycle -v", + "focused_result": "34 tests passed in 42.301 seconds, exit 0", + "integrity_command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_experiment_saved_runs.ExperimentSavedRunsTest.test_success_cannot_hide_missing_review_behind_failure_metadata test_experiment_evidence_integrity -v", + "integrity_result": "6 tests passed in 9.803 seconds, exit 0", + "bash": "bash test/run.sh: 227 assertions in 11 files passed", + "generated_reference": "python3 scripts/generate-core-reference.py --check passed", + "full_core": "a4a87fd unchanged source: 427 tests in 260.966 seconds, OK with 2 optional-SDK skips, exit 0; UTC elapsed 261.146617 seconds and monotonic 261.146937 seconds", + "full_core_warning": "Known unclosed SQLite finalizer ResourceWarning remains R5; not suppressed or counted as repaired", + "independent_review": "Not completed; no independent audit claimed" + }, + "fixture_boundary": "Real offline workers and public terminal completion; test-only predeclared-assignment seam, manual outcome addition and synthetic host dispositions. Not the R5 production trial controller or native quality proof.", + "external_actions": "No provider request, installation refresh, purchase, reset, global setting change or push", + "next_action": "R3b.2 shared current-evidence eligibility and explicit append-only evaluation/review revisions; bounded independent reader audit remains outstanding" +} diff --git a/docs/plans/engineering-team/evidence/R3b1-reader-partial-2026-10-01.json b/docs/plans/engineering-team/evidence/R3b1-reader-partial-2026-10-01.json new file mode 100644 index 0000000..c4b85dc --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3b1-reader-partial-2026-10-01.json @@ -0,0 +1,61 @@ +{ + "schema_version": 1, + "work_package": "R3b.1", + "status": "partial", + "recorded_at": "2026-10-01T19:58:04Z", + "baseline_revision": "a22b949", + "implementation": [ + "Read v2 evidence from immutable preparation assignments and actual saved runs inside one ledger transaction", + "Validate preparation/event/attempt/profile/package/outcome identities and captured artifact bytes", + "Reject changed-evidence v2 replay while preserving historical evaluation bytes", + "Check controlled inputs and selected execution before reserving an assigned trial attempt" + ], + "source_sha256": { + "plugin/core/src/devsquad/store.py": "b474a0771150f4a38ba953a2fe13869fdaa7f60fb85ec15b0420e7a6d20ed847", + "plugin/core/src/devsquad/experiment_evidence.py": "d8c756fd18bdfdd77f3096385b09c872f7cf12a9d4dd66dda393f2eff3679d65", + "test/core/experiment_runtime_fixture.py": "eb653e4f60466bee52fbf33ffeb95353a79cdddc0f3c637f5b16a5186d0d32ed", + "test/core/test_experiment_saved_runs.py": "df6c0ac8f7e2b398727291945eab05a798ac1eceaacf4090890d62767fdbf507", + "test/core/test_experiment_evidence_integrity.py": "834d58e4827e96cafa1cc7053098fd11ed42fcb795fcad001f5b64cedbf77483" + }, + "verification": { + "public_reader_red": "One public evaluator test errored before integration after four real offline arms completed; missing saved-project identity; 5.527 seconds", + "imported_profile_red": "One deterministic test failed before semantic artifact binding: ContractError not raised; 1.505 seconds", + "earlier_focused_green": "59 tests passed in 43.570 seconds; extra prelaunch guard case then passed separately", + "checkpoint_focused_command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_experiment_evidence_integrity test_experiment_saved_runs test_experiment_reuse test_experiment_provenance test_experiment_assignment_store test_experiment_v2_evaluation test_learning test_lifecycle -v", + "checkpoint_focused_result": "60 tests passed in 46.206 seconds, exit 0; process access enabled for cancellation and cleanup", + "bash_command": "bash test/run.sh", + "bash_result": "11 files, 227 assertions passed, exit 0; process access enabled", + "generated_reference": "python3 scripts/generate-core-reference.py --check passed", + "full_core_gate": "Not run for this partial reader source", + "independent_reader_review": "Incomplete; a bounded review raised an execution-binding concern but its follow-up hit a usage limit. No passed independent audit is claimed.", + "unavailable_earlier_handle": "A prior 50-test handle disappeared during a model switch; its result is unavailable and is not counted as a pass" + }, + "fixture_boundary": { + "positive_runs": "Public Service start, real offline worker/check subprocesses, host claim/complete and manual outcome addition", + "predeclaration": "A test-only preparation resolver seam attaches the frozen assignment before real complete_preparation; production paired-trial entry remains R5", + "metrics": "Synthetic host accept/reject dispositions; not native model-quality evidence", + "sql_tampering": "Negative corruption cases only; positive terminal runs are not fabricated with SQL" + }, + "open_findings": [ + { + "status": "inspection_concern_not_yet_reproduced", + "description": "Branch-review stdout validation currently treats absent output_metadata.failure as success. An honest terminal failed worker may lack that key and emit non-JSON output.", + "next_action": "Add a real offline terminal failed-reviewer regression; distinguish successful imported review evidence from opaque failed output, preserving failed exposure and diagnostics" + } + ], + "remaining": [ + "Resolve the terminal-failure concern and run full core integration before accepting R3b.1", + "R3b.2 shared eligibility for replay, qualification, promotion, rollback and catalog fallback", + "Explicit append-only evaluation/review revisions after new corrections", + "R3c public historical/legacy compatibility and realistic failure/repair proof", + "R5 production paired-trial controller and automatic terminal-outcome projection", + "R8 updated install and provider/surface acceptance" + ], + "gemini_antigravity_status": { + "evidence": "M7-installed-runtime-2026-09-29.json", + "verified": "Real read-only squad_status through Antigravity observed the terminal-created saved run", + "not_verified": "Gemini implementation/review worker execution and operation against the newly repaired installed runtime", + "user_action": "No additional Gemini login action currently recorded; preserve normal signed-in session" + }, + "external_actions": "No provider generation, API request, installation refresh, global settings change, purchase, reset redemption or push during this reader slice" +} diff --git a/docs/plans/engineering-team/evidence/R3b2-eligibility-partial-2026-10-01.json b/docs/plans/engineering-team/evidence/R3b2-eligibility-partial-2026-10-01.json new file mode 100644 index 0000000..bfc69b0 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3b2-eligibility-partial-2026-10-01.json @@ -0,0 +1,34 @@ +{ + "schema_version": 1, + "work_package": "R3b.2/R3c", + "status": "partial", + "recorded_at": "2026-10-01T21:18:00Z", + "baseline_revision": "a4a87fd", + "source_sha256": { + "plugin/core/src/devsquad/experiment_eligibility.py": "3f53719032bf0fbeb160b1c1c8a82378c9c7a66d8e20b95b76fb292c8bd1691d", + "plugin/core/src/devsquad/store.py": "a7ec8170ca06a5275ad122d27c74d566b18bf8376e4abb30905f531e33ac9151", + "plugin/core/src/devsquad/learning.py": "358df4f1cb4751c9cbb739439d432f82f9b9dc88389975b2fb1ebbf14771850e", + "plugin/core/src/devsquad/migrations/015_experiment_evaluation_revisions.sql": "b5dc3cb747d4fed29dbac615e1b84c869417bdda51d88153efd5e72677b109b4", + "test/core/test_experiment_eligibility.py": "59663e05da53c52d2de483bf8663ab25f4c9086efd50b5bee196946dd357570f" + }, + "verification": { + "red": "2 tests in 10.631 seconds: stale qualification failed to raise ContractError, explicit revision errored because Service did not accept revision_id; exit 1", + "eligibility": "6 tests in 32.393 seconds passed: stale qualification/replay/promotion, exact completed decision replay, profile fingerprint/task context and append-only correction review", + "lifecycle_assignment": "14 tests in 22.413 seconds passed, including real-worker qualified promotion/regression rollback and historical schema-13 byte preservation", + "latest_command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_learning test_experiment_eligibility test_experiment_saved_runs -v", + "latest_result": "23 tests in 91.779 seconds passed", + "bash": "227 assertions in 11 files passed", + "generated_reference": "Regenerated and --check passed", + "full_core": "Not run on schema-15 source; a4a87fd's prior 427-test gate is not substituted", + "independent_audit": "Not completed" + }, + "fixture_boundary": "Positive lifecycle and learning authority now uses real offline worker/check subprocesses and public host completion. Original pair gates retained (lifecycle 1+1; learning 2+1). A test-only assignment seam and manual outcome import remain until R5; no native quality or public production trial-controller claim.", + "remaining": [ + "Stale catalog-fallback and regression-rollback negative/public tests", + "Correction-versus-qualification transaction race proof", + "Revision chain, explicit CLI and upgraded historical duplicate-outcome audit proof", + "R3c remaining workflow coverage, full integration and bounded independent audit", + "R4-R6 runtime connections, old active/recoverable-run safe schema upgrade and R8 installed/live gates" + ], + "external_actions": "No provider call, API request, installation, purchase, reset, global setting change or push" +} diff --git a/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json b/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json new file mode 100644 index 0000000..d7ef154 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json @@ -0,0 +1,37 @@ +{ + "schema_version": 1, + "work_package": "R3b.2/R3c", + "status": "source_offline_verified_independent_audit_pending", + "recorded_at": "2026-10-01T21:29:24Z", + "baseline_revision": "4e6297f", + "source_sha256": { + "plugin/core/src/devsquad/experiment_eligibility.py": "66d1c4d079e4824daf5cd349fa03a35f09b70d2b54367447b2633092d60a5baa", + "plugin/core/src/devsquad/store.py": "881398323a238d0f5538c6ce2cfeb8b9dc630a5b617ee5c2c794597c1099c8c6", + "test/core/experiment_runtime_fixture.py": "cf98bc312665b138e4d4d9a9633f888ef2f2e48ecba55a56037103402afa7646", + "test/core/test_experiment_eligibility.py": "940d6fb19cf9baa5685c633094b2b309b261504f5546785acee6c227b453f897", + "test/core/test_experiment_saved_runs.py": "68ce3265c5c342805dbf433760172d2b531694b821421e3ee1f9d1d24bba9240" + }, + "verification": { + "stale_rollback_and_fallback": "2 tests in 26.432 seconds passed: stale regression blocked, explicit revision permitted valid rollback; stale qualified target blocked rollback and was skipped by catalog fallback; proven bootstrap preserved", + "race_and_proposal": "Both targeted tests passed during a 3-test 11.138-second run; the third historical fixture test failed, so that combined run is not a pass", + "historical_failure": "Historical fixture stored /var project path instead of resolved /private/var path and proposal returned no_evaluated_experiment; corrected the fixture to match old Store normalization", + "historical_green": "Schema-13 duplicate-outcome upgrade/report/proposal/new-authority rejection test passed separately in 0.074 seconds", + "cli": "Original and explicit-revision dispatch tests passed, 2 tests in 0.006 seconds", + "delivery_fixture_failures": "New fixture first used an invalid internal implementation shape (0.555 seconds), then waited on an internal phase (20.605 seconds), then mistook pre-daemon candidate readiness for reviewed completion (1.426 seconds). These were test-helper defects, not runtime findings", + "delivery_green": "Four public paired issue-delivery arms with 8 real worker attempts, mandatory checks and source preservation passed in 7.988 seconds", + "fresh_reviewed_qualification": "1 test in 5.449 seconds passed; fresh reviewed revision qualifies and pins the new binding decision", + "revision_and_wheel": "2 tests in 8.102 seconds passed: revision arguments/predecessor checks and installed-wheel current schema-15 content/migration", + "bash": "227 assertions in 11 files passed", + "generated_reference_and_whitespace": "Both checks passed", + "full_core": "At 5de6655: 441 tests in 372.506 seconds FAILED (1 missing-receipt error, 2 optional-SDK skips); UTC elapsed 372.689586 and monotonic 372.691304 seconds. Not a passing gate", + "failed_case_recheck": "The exact headless delivery check-integrity case passed unchanged (1 test in 2.760 seconds); all 9 check-integrity cases passed in 19.194 seconds. Cause remains unproven; added status/artifact-name diagnostics", + "sqlite_cleanup": "Found installer test transaction context that did not close its SQLite connection; replaced with contextlib.closing. Active-release test passed in 6.837 seconds under ResourceWarning=error, forced collection and zero unraisable exceptions. Full 441-test rerun at dc4e110 had zero unraisable exceptions and no SQLite warning", + "full_core_rerun": "At dc4e110: 441 tests in 374.998 seconds OK (2 optional-SDK skips); UTC elapsed 375.114707 and monotonic 375.116386 seconds; forced collection, zero unraisable exceptions. Earlier failed gate retained", + "independent_audit": "Not completed" + }, + "remaining": [ + "Bounded independent R3 audit", + "R4-R6 runtime repairs and R8 safe updated installation/live Claude, Grok and Gemini proofs" + ], + "external_actions": "No provider request, installation refresh, purchase, reset, global settings change or push" +} diff --git a/docs/plans/engineering-team/evidence/R4-closure-2026-10-02.json b/docs/plans/engineering-team/evidence/R4-closure-2026-10-02.json new file mode 100644 index 0000000..722b83b --- /dev/null +++ b/docs/plans/engineering-team/evidence/R4-closure-2026-10-02.json @@ -0,0 +1,46 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "status": "accepted_source_and_installed_routing_package", + "source_revision": "913abc1267e47684fee63dbbb1d69945f67f4a32", + "full_gate": {"tests": 484, "seconds": 458.607, "skips": 2, "errors": 0, "failures": 0, "unraisable": [], "monotonic_seconds": 458.6838033749991, "utc_seconds": 458.680991, "offline_build_interpreter": "explicit_python3.12", "source_and_tests_frozen": true}, + "focused_gate": {"tests": 102, "seconds": 17.610, "result": "pass"}, + "independent_repair_review": { + "run_id": "f45c723b-3067-473f-9349-f1012f5548e7", + "base_revision": "500d1b1917ecbcf417af73fbfd21ae67accc45e7", + "candidate_sha256": "15ca9655c33f89b9be71fcad07c82c5c996c5c77b7c7ea1c67f18141ccd61076", + "state": "succeeded", "version": 22, "disposition": "accept", "verdict": "clean", "findings": [], + "identity": {"harness": "codex", "harness_version": "codex-cli 0.159.2", "model_id": "gpt-6.1-sol", "effort": "low", "verification": "verified", "permission_policy": "read_only"}, + "review_sha256": "a3e3d56f5cc91b56737952eb53912d818dd914d4d7a08abf69ce6660ba9feba7", + "checks_sha256": "1bca1c29edd2337634a9a14d0d1d701ad1b6a6abbbbd2102bf95a13c6651160c", + "receipt_sha256": "7f3a50bf9c5a23064278b78cda4f8773ab40ab153d238ca88294778f188a5c76", + "checks": {"diff_and_bash": "pass", "focused_tests": 30, "seconds": 0.993, "optional_installed_wheel_environment_skips": 1, "integrity": "verified_unchanged"}, + "usage": {"source": "native_reported", "input_tokens": 95191, "output_tokens": 567, "total_tokens": 95758} + }, + "installation": { + "release": "0.1.0-py31214-3631f1737bc9-mcp-a26bc88afbef", + "source_digest": "3631f1737bc97132dcb51596c65b929c13b1a0904f2f9776a586f106e6c53b9d", + "python": "3.12.14", "mcp": "2.2.0", "ledger_schema": 15, + "nonterminal_runs_before": 0, "first_changed": true, "reinstall_changed": false, + "network_downloads": false, "all_payload_drift": false, "manifest_matches": true, + "pip_check": "pass", "previous_releases_retained": true, + "installed_sdk_transport": {"tests": 9, "seconds": 2.743, "skips": 0, "result": "pass", "import_origin": "resolved installed release, not source"}, + "normal_cli_discovery": {"workflow": "branch-review", "dry_run": true, "model": "gpt-6.1-sol", "effort": "low", "selection_mode": "bounded_trial", "run_created": false, "task_sha256": "d026a58b367879fb61d754a0fd252e798a10a5e091acd71346bb206f9833a4e2"}, + "host_registrations": {"matching": 4, "writes": 0}, + "doctor_note": "Current doctor confirms installation/registration only, not authenticated workflow readiness; that distinction is R6. Antigravity externally changed to 1.2.14 and remains unverified; prior 1.2.13 live receipt is historical" + }, + "requirements": [ + {"requirement": "approved aliases survive provider-default hints; reviewed promotion changes only new runs; exact pins and frozen runs preserved", "proof": "test_task_entry public promotion and pin fixtures plus test_native_catalog default-hint fingerprint test", "result": "pass"}, + {"requirement": "scoped complete last-good catalog, TTL, one refresh owner, interruption, initial and prior-snapshot failure backoff", "proof": "test_native_catalog and production normal CLI discovery connection", "result": "pass"}, + {"requirement": "affected model/account/config/version changes block normal approved roles for requalification, without alias mutation or promotion from discovery", "proof": "test_task_entry native context/catalog mismatch and scoped discovery tests", "result": "pass"}, + {"requirement": "two projects honor weekly exhaustion despite available primary quota before worker launch", "proof": "test_task_entry actual normal CLI preparation over two committed projects; daemon launch asserted absent", "result": "pass"}, + {"requirement": "same account across discovery configurations shares one reservation fence and retains fresh exhausted quota", "proof": "test_task_entry cross-config public discovery and Store reservation fence regression; independent audit follow-up", "result": "pass"}, + {"requirement": "incompatible accounts do not reuse capacity; unsupported/null/stale observations remain unknown; manual observations retained", "proof": "test_native_catalog, test_capacity and cross-account normal discovery fixture", "result": "pass"}, + {"requirement": "no billing, permission, login, reset-credit or paid-API expansion", "proof": "read-only RPC allowlist and fixed normal role permissions; all discovery calls are non-generating", "result": "pass"} + ], + "official_native_protocol_reference": "https://learn.chatgpt.com/docs/app-server", + "preserved_failed_and_partial_history": "R4-native-scoped-partial-2026-10-02.json", + "operator_probe_note": "First installed-origin assertion compared a symlink path string instead of resolved path; corrected resolved-origin assertion passed the actual installed SDK gate. Expected negative squad_events argument rejection is not a failed test", + "separate_open_scope": ["R5 automatic terminal outcomes/public trials", "R6 terminal UX/auth readiness/generated-reference check", "R7 Council", "R8 final delivery/desktop UI/new Antigravity version acceptance"], + "adoption_note": "No real-model superiority or automatic policy promotion is claimed. Normal unqualified selection remains a bounded trial; Jev stays off" +} diff --git a/docs/plans/engineering-team/evidence/R4-native-compatibility-2026-10-01.json b/docs/plans/engineering-team/evidence/R4-native-compatibility-2026-10-01.json new file mode 100644 index 0000000..f3b0c1f --- /dev/null +++ b/docs/plans/engineering-team/evidence/R4-native-compatibility-2026-10-01.json @@ -0,0 +1,24 @@ +{ + "schema_version": 1, + "work_package": "R4/R8", + "status": "discovery_compatibility_verified_operation_pending", + "observations": { + "path_cli": "codex-cli 0.135.0 is unverified and is not selected for native work", + "bundled_binary": "/Applications/ChatGPT.app/Contents/Resources/codex-cli/bin/codex", + "bundled_version": "codex-cli 0.159.2", + "non_generating_protocol": "initialize, complete model/list pagination, account/read refreshToken=false and account/rateLimits/read all succeeded", + "model_catalog": "8 models with native effort metadata, one default", + "auth": "chatgpt; no credentials or account identifier retained here", + "protocol_schema": "Generated by the installed bundled binary; official account/rateLimits documentation also inspected" + }, + "verification": { + "red": "New bundled layout/version test failed on missing manifest path before the repair", + "green": "61 M1/MCP/task-entry tests passed in 5.886 seconds, with two optional-SDK skips", + "bash": "227 assertions in 11 files passed", + "native_generation": "Not run for this metadata slice", + "installation": "Not updated" + }, + "documentation": "https://learn.chatgpt.com/docs/app-server", + "remaining": ["Independent R3 audit", "R4 stable aliases/scoped catalog/quota", "R8 safe upgrade and installed provider proofs"], + "external_mutations": "No global settings, authentication changes, purchases, credit resets, paid APIs, pushes or installation refresh" +} diff --git a/docs/plans/engineering-team/evidence/R4-native-scoped-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R4-native-scoped-partial-2026-10-02.json new file mode 100644 index 0000000..70cfa44 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R4-native-scoped-partial-2026-10-02.json @@ -0,0 +1,54 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "baseline_revision": "39b95f1", + "status": "source_focused_verified_full_and_independent_gates_pending", + "review_finding_and_repair": { + "review_run_id": "9a12f6f6-59aa-43c0-a003-9f057f42f670", + "review_target": "500d1b1917ecbcf417af73fbfd21ae67accc45e7", + "state": "failed", + "version": 22, + "disposition": "reject", + "finding_id": "r4-account-pool-isolation", + "severity": "high", + "description": "Config/binary/version-scoped capacity pools partitioned a single subscription and bypassed max_concurrency=1", + "repair": "Use account-only opaque pool identity for shared observations/reservations, but keep separate account/config/binary/version native-scope and catalog-fingerprint qualification evidence", + "targeted_tests": {"tests": 30, "seconds": 2.712, "result": "pass"}, + "combined_affected_tests": {"tests": 102, "seconds": 17.610, "result": "pass", "installed_wheel_build_interpreter": "explicit_offline_python3.12", "resource_warnings": "errors"}, + "review_checks": {"focused_tests": 79, "seconds": 16.390, "result": "pass", "diff_and_bash": "pass", "integrity": "verified_unchanged"} + }, + "initial_full_gate": { + "target_revision": "500d1b1", + "result": "interrupted_not_passed", + "exit_code": 130, + "known_failures": "Two CLI mock call expectations omitted the newly required runtime argument; corrected without weakening preparation checks", + "reason": "Stopped exact full-test PID with SIGINT after known failures and independent high-severity finding, before source repair", + "unraisable_note": "Interruption during test setUp produced an implicit TemporaryDirectory ResourceWarning; no clean full-gate result is claimed" + }, + "scope": "Normal CLI native subscription discovery, private account/config/binary/version-scoped last-good catalog, 24-hour TTL, single OS-lock refresh owner, bounded pagination deadline, two-minute failure backoff, native typed quota ingestion and affected alias requalification block", + "verified": [ + {"gate": "native cache, normal entry, protocol and capacity", "tests": 79, "seconds": 14.696, "result": "pass"}, + {"gate": "two-project public normal review entry", "result": "pass", "scope": "Actual CLI preparation ingests quota and terminalizes both projects before any worker launch when the shared weekly window is exhausted despite available primary capacity"}, + {"gate": "real non-generating native discovery", "result": "pass", "scope": "Installed verified Codex binary, native account/config/model/quota read RPCs, exact gpt-6.1-sol/low, scoped pool and catalog fingerprint; two saved native windows yielded unknown capacity, not guessed availability"}, + {"gate": "privacy and isolation", "result": "pass", "scope": "Only hashes of account and effective configuration persist; incompatible account/config/binary/version scope cannot reuse catalogs. Failed quota query preserves still-fresh known exhaustion; a new incompatible account remains unknown"}, + {"gate": "provider default hint", "result": "pass", "scope": "Default-hint-only changes do not invalidate approved capability fingerprints or alter alias policy"} + ], + "intermediate_failures": [ + "Initial red test could not import the not-yet-implemented native_catalog module", + "Initial quota assertion lacked the required harness target selector; test corrected to query the actual Codex target", + "Initial cross-project fixture created Git repositories inside a blanket Popen mock; corrected selective native-process mock and created committed fixtures before mocking", + "Initial first-refresh backoff check accidentally caught its own ContractError (a ValueError subclass); separated the backoff decision from JSON validation", + "One overlapping focused invocation saw changing paired input bytes; final focused invocation used frozen source and passed. Full gate must always freeze source/tests" + ], + "constraints": { + "generating_calls": 0, + "login_or_config_mutations": 0, + "reset_credit_or_purchase_calls": 0, + "paid_api_fallback": false, + "manual_provider_observations_preserved": true, + "existing_aliases_mutated": false, + "explicit_pins_preserved": true + }, + "open_gates": ["R3c independent closure audit", "one full source integration gate", "bounded independent R4 review", "fresh installed normal-command recheck"], + "r3_review": {"run_id": "f0a4f86b-b305-4839-88a2-cb5fed8739ab", "target_revision": "39b95f1", "status": "succeeded", "version": 22, "closure_evidence": "R3-closure-2026-10-02.json"} +} diff --git a/docs/plans/engineering-team/evidence/R5-closure-2026-10-02.json b/docs/plans/engineering-team/evidence/R5-closure-2026-10-02.json new file mode 100644 index 0000000..927cf66 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R5-closure-2026-10-02.json @@ -0,0 +1,41 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "status": "accepted_source_and_installed_public_outcomes_trials", + "source_revision": "dfe9976224e9ff2166a016b802e074aa1f763c4b", + "reviewed_source_revision": "f87060b7980a9fbe95314bb0986f84e424eccd3e", + "source_scope_equivalence": "git diff --exit-code f87060b dfe997 -- plugin/core passed; later changes only test expectations/observation and checkpoint documents", + "full_gate": {"command": "env DEVSQUAD_BUILD_PYTHON= PYTHONPATH=plugin/core/src:test/core PYTHONWARNINGS=error::ResourceWarning python3 scripts/run-core-tests.py --failfast", "tests": 504, "seconds": 511.241, "failures": 0, "errors": 0, "skips": 2, "unraisable": [], "monotonic_seconds": 511.3239737499989, "utc_seconds": 511.325922, "source_and_tests_frozen": true}, + "focused_gates": [{"tests": 54, "seconds": 48.572, "scope": "review repairs/outcomes/trials/store/process ownership", "result": "pass"}, {"tests": 66, "seconds": 125.322, "scope": "integrity/outcomes/review/delivery", "result": "pass"}], + "independent_review": { + "initial_run": "5a70f3db-91a8-4606-9809-c0b993b988b1", "initial_disposition": "reject", "findings": ["R5-001", "R5-002"], + "repair_run": "e046a174-55ce-4496-973e-2fc97c68a493", "state": "succeeded", "version": 22, "disposition": "accept", "verdict": "clean", "remaining_findings": [], + "identity": {"harness": "codex", "harness_version": "codex-cli 0.159.2", "model_id": "gpt-6.1-sol", "effort": "low", "verification": "verified", "permission_policy": "read_only"}, + "candidate_sha256": "564e2e937dc01d1127378537b45b7a9170b91c8a90545e7d3775321025640aa2", + "review_sha256": "a448ab152287e163ac4610850bee574939f97fb822863d31e8d8050fdf09ac6f", + "checks_sha256": "a1a74d34c4fda3960d880ea018fbf30f67e38f92d026cfe63f40e56b207c86de", + "receipt_sha256": "58eea09f033104b98b0dd5ef086ecbaf65faa050802d39bb4da4bf08d8c5b254", + "checks": {"diff_bash_reference": "pass", "public_tests": 13, "seconds": 10.729, "integrity": "verified_unchanged"} + }, + "requirements": [ + {"requirement": "one objective final per new public run from every terminal origin; replay after terminal/projection crash", "proof": "test_objective_outcomes public preparation failure/cancel, worker failure, host/headless completion, timeout and replay tests", "result": "pass"}, + {"requirement": "failed/fallback/revision/finding/lead-repair attribution; no invented independent success or criterion-level test proof", "proof": "test_objective_outcomes, learning/lifecycle and full gate", "result": "pass"}, + {"requirement": "append-only late corrections; legacy history unchanged; scoped pending projection repair and consistent proposal read", "proof": "public report/proposal replay, manual final replacement rejection, corruption isolation and native R5-001 follow-up", "result": "pass"}, + {"requirement": "public bounded trials freeze predeclared spec/assignment before real attempts, with one shared all-attempt cap", "proof": "test_public_trials concurrency/fallback/declaration replay; fixture now calls trial_start without preparation monkeypatch or manual final import", "result": "pass"}, + {"requirement": "experiment wall deadline stops already launched work with owned-child cleanup", "proof": "public actual delayed worker regression, runner ownership tests and native R5-002 follow-up", "result": "pass"}, + {"requirement": "full public pair/evaluate/qualify/reviewed promotion/new-run/held-out regression/rollback chain", "proof": "test_public_trials.PublicTrialTest.test_public_outcomes_evaluate_qualify_promote_new_run_and_roll_back", "result": "pass_offline_actual_workers"}, + {"requirement": "safe schema-changing upgrade and installed SDK transport", "proof": "full installer old schema13 active/recoverable deferral, cancellation/resume and migration-to16 tests; installed SDK nine tests", "result": "pass"} + ], + "installation": { + "release": "0.1.0-py31214-90f1e87fb9ac-mcp-a26bc88afbef", "source_digest": "90f1e87fb9acf7756dad46780672de9b84fe75e3631322857a310f089280345e", + "python": "3.12.14", "mcp": "2.2.0", "ledger_schema": 16, "nonterminal_runs_before": 0, "nonterminal_runs_after": 0, + "legacy_projection_jobs_after_migration": 0, "first_changed": true, "reinstall_changed": false, "payload_drift": false, "manifest_matches": true, "pip_check": "pass", "network_downloads": false, "previous_releases_retained": true, + "installed_sdk": {"tests": 9, "seconds": 2.674, "skips": 0, "failures": 0, "errors": 0, "origin": "resolved installed release, not source"}, + "normal_review_dry_run": {"model": "gpt-6.1-sol", "effort": "low", "selection_mode": "bounded_trial", "run_created": false, "task_sha256": "2f71ea5247b55aa4335e737f069b0d84d4ef13e7afc18d0dc9406e4f5185e1ab"}, + "matching_host_registrations": 4, + "doctor_note": "R5 doctor still describes installation/registration, not full authenticated or operation-verified readiness; R6 repairs this distinction" + }, + "other_gates": {"bash_assertions": 227, "generated_reference": "current", "git_diff_check": "pass"}, + "failure_history": "R5-public-integration-partial-2026-10-02.json preserves rejected native findings, red tests, interrupted and failed full invocations", + "limits": "Actual offline workers prove runtime/provenance, not model superiority or subscription savings. Automatic experimentation/promotion and Jev remain off. R6 usability, R7/C1 Council and R8 final desktop/live acceptance remain separate." +} diff --git a/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json new file mode 100644 index 0000000..38d0b44 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json @@ -0,0 +1,65 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "status": "source_partial_not_accepted_or_installed", + "base_revision": "d3f7c0b", + "source_schema": 16, + "installed_schema": 15, + "requirements": [ + {"requirement": "idempotent objective final projection from preparation failure/cancel, worker failure, host/headless completion and crash replay", "proof": "test_objective_outcomes public Service transitions", "result": "focused_pass"}, + {"requirement": "append-only late corrections, no manual public replacement final, scoped corruption isolation and public report replay", "proof": "test_objective_outcomes and test_service", "result": "focused_pass"}, + {"requirement": "public predeclared assignments and automatic finals, not preparation monkeypatch or manual import", "proof": "ExperimentRuntimeFixture now calls Service.trial_start; public reviewer and implementer pair tests", "result": "focused_pass"}, + {"requirement": "one shared experiment reservation cap across concurrent arms and fallback", "proof": "test_public_trials actual concurrent public resumes and fallback failure", "result": "focused_pass"}, + {"requirement": "evaluate, qualify, reviewed promotion for a new run, held-out regression and rollback", "proof": "test_public_trials complete Service API chain with actual offline worker processes", "result": "focused_pass"}, + {"requirement": "declaration-time experiment wall deadline and precise headless lead-repair attribution", "proof": "test_public_trials deadline regression and final affected learning/lifecycle run", "result": "behavioral_pass_in_71_test_gate_with_two_stale_schema_assertions"} + ], + "passed_gates": [ + {"command": "python3 -m unittest test_check_integrity_runtime test_objective_outcomes test_review_runtime test_delivery_workflow", "tests": 66, "seconds": 125.322, "resource_warnings": "promoted_to_errors", "result": "pass_after_headless_test_transition_fix"}, + {"command": "python3 -m unittest test_objective_outcomes test_public_trials test_store test_supervisor test_m2_supervisor_gate", "tests": 54, "seconds": 48.572, "resource_warnings": "promoted_to_errors", "result": "pass_after_independent_findings_repair"}, + {"command": "python3 -m unittest test_objective_outcomes test_public_trials test_cli test_store test_service", "tests": 88, "seconds": 61.236, "resource_warnings": "promoted_to_errors", "offline_build_interpreter": "explicit_python3.12", "result": "pass_before_final_deadline_and_lead_attribution_refinement"}, + {"command": "python3 -m unittest test_public_trials.PublicTrialTest.test_public_outcomes_evaluate_qualify_promote_new_run_and_roll_back", "tests": 1, "seconds": 12.300, "result": "pass"} + ], + "failure_history": [ + "Initial seven public terminal-origin tests failed because no automatic finals existed (7 failures/9.322s).", + "First projection implementation had wrong criterion-status and incomplete worker-failure mapping (1 failure/4 errors in 7 tests/9.356s).", + "Headless receipt attempts do not include IDs on successful semantic records; keyed failure lookup was corrected (1 error in 7 tests/9.472s).", + "Existing generic fake-runner native exit receipt lacks managed run/state fields; explicitly fixture-gated normalization now preserves that original receipt (1 error in 3 tests/15.093s).", + "New whole-chain test used an invalid invented template mode human_reviewed; corrected to the existing reviewed contract, not a source policy relaxation (1 error in 6 tests/6.586s).", + "An exploratory fixture diagnostic requested a nonexistent exit_code column and failed before printing; the corrected diagnostic used finally cleanup. This is not a passing test or a production fault.", + "The 71-test frozen affected learning/lifecycle gate completed in 233.881s with two stale schema-15 assertion failures (test_learning and test_lifecycle). No behavioral errors; the assertions were updated to schema 16, not removed. The two direct checks and another 22 migration/capacity/decision tests (1.208s) then passed.", + "Independent native audit rejected exact 14b3a391 for pending-projection nested transaction and active-worker experiment deadline gaps, despite passing required diff/Bash/16 public tests/reference checks.", + "Controlled 3s experiment / 6s worker diagnostic reached resume_candidate_review after 7.274s, confirming the deadline gap. Temporary diagnostic run was cancelled and cleaned.", + "Root full gate on that rejected candidate was intentionally SIGINT-stopped at verified PID 77765, exit 130; not a timeout or pass. unittest interruption skipped cleanup and emitted TemporaryDirectory and SQLite finalizer ResourceWarnings; no scoped remaining process was found.", + "Three public red regressions reproduced proposal nested BEGIN, generic TIMEOUT receipt normalization and active-worker deadline overrun (1 failure/2 errors/14.842s). Fixed source passes the 54-test affected gate.", + "The next complete fail-fast core invocation stopped after 17 tests/10.494s (one failure, no errors/skips/unraisable): the integrity test inspected a terminal receipt during the transient headless awaiting_host-to-queued lead transition. The isolated test passed in 2.719s; the test now waits for terminal state only in headless mode, retaining every integrity/source-preservation assertion. All 66 integrity/outcome/review/delivery tests pass in 125.322s. This is not a quota timeout or a passing full gate.", + "Frozen bf3de086 fail-fast gate stopped after 188 tests/274.410s, one stale test_handoff_store schema-four upgrade expectation (15 vs actual 16), no errors/skips/unraisable; UTC/monotonic both 274.50s. Corrected only the current-schema expected value; all schema_migrations expectations were searched to distinguish preserved historical schemas. Direct migration regression passes in 0.090s. This is not a runtime migration defect, quota timeout or passing full gate." + ], + "rejected_independent_review": { + "run_id": "5a70f3db-91a8-4606-9809-c0b993b988b1", "state": "failed", "version": 22, "host_disposition": "reject", + "base_revision": "d3f7c0b3cc1c704dee32f96a87500a3284572d9b", "target_revision": "14b3a3910972484074ed52d99f70123944e56fd0", + "candidate_sha256": "2aa50bf50541ebabe337ce7d9f1df99ce7151c0a888f08a195efd6dc0239c6f9", + "identity": {"harness": "codex", "harness_version": "codex-cli 0.159.2", "model_id": "gpt-6.1-sol", "effort": "low", "verification": "verified", "permission_policy": "read_only"}, + "findings": [{"id": "R5-001", "severity": "medium", "title": "Pending projection breaks transactional policy proposal reads", "status": "source_repaired_focused_pass_followup_pending"}, {"id": "R5-002", "severity": "high", "title": "Experiment wall deadline only fences reservations", "status": "source_repaired_focused_pass_followup_pending"}], + "review_sha256": "912c79e25ebef6cd51ed00fb118aa761b80a6301f19937ca948298d570e52529", + "checks_sha256": "42204bc469b2298e66602e3864f70c8a9e7e70f9d4131da37b322c72c343c9ca", + "receipt_sha256": "c63483b4835fc5eb27afa30dd657436a9b5e595166d8ba765ead35ead934e3e5", + "checks": {"diff_bash_reference": "pass", "public_tests": 16, "seconds": 37.184, "integrity": "verified_unchanged"}, + "usage": {"source": "native_reported", "input_tokens": 196084, "output_tokens": 1316, "total_tokens": 197400} + }, + "accepted_independent_repair_review": { + "run_id": "e046a174-55ce-4496-973e-2fc97c68a493", "state": "succeeded", "version": 22, "host_disposition": "accept", + "base_revision": "14b3a3910972484074ed52d99f70123944e56fd0", "target_revision": "f87060b7980a9fbe95314bb0986f84e424eccd3e", + "candidate_sha256": "564e2e937dc01d1127378537b45b7a9170b91c8a90545e7d3775321025640aa2", + "verdict": "clean", "findings": [], "closed_findings": ["R5-001", "R5-002"], + "identity": {"harness": "codex", "harness_version": "codex-cli 0.159.2", "model_id": "gpt-6.1-sol", "effort": "low", "verification": "verified", "permission_policy": "read_only"}, + "review_sha256": "a448ab152287e163ac4610850bee574939f97fb822863d31e8d8050fdf09ac6f", + "checks_sha256": "a1a74d34c4fda3960d880ea018fbf30f67e38f92d026cfe63f40e56b207c86de", + "receipt_sha256": "58eea09f033104b98b0dd5ef086ecbaf65faa050802d39bb4da4bf08d8c5b254", + "checks": {"diff_bash_reference": "pass", "public_tests": 13, "seconds": 10.729, "integrity": "verified_unchanged"}, + "usage": {"source": "native_reported", "input_tokens": 131692, "output_tokens": 1078, "total_tokens": 132770}, + "scope": "Only the declared repair diff is accepted; whole R5 still requires the full and installation gates. Subsequent change is test-only headless observation timing." + }, + "open_gates": ["full core suite", "schema-16 safe upgrade and installed transport proof", "R6/R7/R8 separate acceptance"], + "policy": {"automatic_experiments": "off", "automatic_promotion": "not_enabled", "bounded_native_review_runs": 2, "paid_api_fallback": false, "legacy_outcomes_rewritten": false}, + "limits": "Fixture outcomes demonstrate runtime/provenance behavior, not real model superiority or subscription cost savings. Installed runtime remains the accepted R4 release until acceptance gates pass." +} diff --git a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json new file mode 100644 index 0000000..12bb398 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json @@ -0,0 +1,183 @@ +{ + "schema_version": 1, + "work_package": "R6", + "recorded_on": "2026-10-02", + "status": "ownership_and_eof_repair_integrated_final_acceptance_pending", + "base_revision": "8f2c1c6a7d9edfe7cc3d78ecbc59ce105be636b2", + "integrated_checkpoints": [ + "dd709804e70e0791614319a9ae5c2597768d7318", + "79f9b767984f2eb64dca430941f405ef952ecf68", + "b785d04013042960b25b47292c3c3a5ead528c87", + "d0eec55c9dbede205e3c8831b7c224d3d77114d3", + "4a6bc15d81017aa996be9a21e6c81bc02ebd17a6", + "03124c79ba53cb1b12296bb554024605e4cadbc7", + "22a4ed8ac85b95861f4ba9725ca729388b7a9f49", + "31a8a68b59db3d0af136c5c18cb67fe5088c02b0", + "e7e9042cbf4d1416209cf39e0269f7fa85479f4c", + "093ed0f397289bffe9ea249c10796cb436027f00", + "4f01456189b38252a678938e7197ac2f12a6838e" + ], + "release_candidate_preflight": { + "target_revision": "fb64e43a28e9b8eab6670be3d7da8e4fd0fcaa97", + "trusted_core": "immutable accepted R5 review_worker._run_check, prepare_check_workspace, candidate_input_state imported using installed Python -P/-B", + "argv": ["env","PYTHONDONTWRITEBYTECODE=1","DEVSQUAD_BUILD_PYTHON=/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3.12","PYTHONPATH=plugin/core/src:test/core","PYTHONWARNINGS=error::ResourceWarning","python3","-m","unittest","test_diagnostics","test_task_entry"], + "tests":60,"seconds":33.182,"trusted_seconds":33.854,"returncode":0,"failures":0,"errors":0, + "input_fingerprint_before_and_after":"131755f0ae8bfe75831607ffd840e1346172aa4f7452f3b2394cdd58614d5216", + "tracked_payload_fingerprint":"3a0baeb9f87bcc8e1f1776a3b64b07cdeac1fce554ee535a87738d78f2c167bd", + "full_payload_fingerprint":"a59b2bd8e47995a93786019cc454cbf3373bf2844518f04290c7bebc5de9ddfa", + "tracked_files":333,"ignored_untracked_additions":0,"bytecode_before_and_after":0,"unchanged":true, + "fingerprint_limitation":"Input fingerprint only, NOT branch-review candidate identity", + "validated_owned_worktree_removed":"nonforce, returncode0", + "first_final_full": {"tests":45,"seconds":27.541,"failures":1,"errors":0,"unraisable":[],"reason":"ordinary finish mock/call contract received new optionalNone Council kwargs","acceptance":"failed"}, + "narrow_cli_repair": {"behavior":"Preserve3arg ordinary finish; forward any explicit Council field without inference; unchanged omittedrun resolver","root_affected":{"tests":32,"seconds":25.365,"failures":0,"errors":0},"independent_review":"clean narrow diff","independent_tests":2,"independent_seconds":0.012,"cli_blob":"ea9b622952478c423aa79d1c5d6a0f89159931ec","test_cli_blob":"698d0dda97803bc356b44084c68a5856d4c4f293","bash_assertions":227,"generated_reference":"current","diff":"passed"}, + "preflight_source_equivalence":"The five affected EOF/diagnostic/fixture paths are unchanged by the subsequent isolated CLI forwarding repair" + }, + "eof_followup": { + "run_id": "d2355cd0-603d-4e6a-b1a2-b05b0e99d53a", + "target_revision": "8242eff68b24538aacc1b40b67b2e88e96b04d52", + "candidate_sha256": "349e55d96ae9ba541973b9f8eef7cfd2f1c7ccfb3f0f3e8ec2525a42edb203c9", + "state": "failed", "version": 22, "host_disposition": "reject", + "finding": "R6-EOF-exit-status", "severity": "medium", + "identity": "verified codex-cli0.159.2/gpt-6.1-sol/low/read_only", + "diff_and_bash": "passed_verified", + "affected_check": {"tests":56,"seconds":27.584,"runner_result":"OK","acceptance":"invalidated_by_undeclared_bytecode"}, + "bytecode_cause": "Controlled Python provider fixture children intentionally use a minimal environment and drop Python flags before importing core; parent _run_check already disabled bytecode. Repair uses -B only in controlled fixtures, with no native environment or integrity relaxation.", + "artifact_hashes_verified": [ + "23f319dbae10d44740863fed2ff6b806d22fedb5984edcc870c546ecf87d560e", + "53f2cb30cd9c51dfe737506dccde3f6cb77fabafd38ac95f237233a96a2148c8", + "ab3081b7057c2f398136ec0e4b489ccb6e0cca2caf1e5375b4459c17b4d48abd", + "5da3c924eac152a7a381ab259f7422688af34feb391834a4df4fb2be641e7b50" + ], + "repair_revision": "4f01456189b38252a678938e7197ac2f12a6838e", + "repair": "Nonreaping natural completion under original deadline before owned cleanup; preserve true version/auth exit status and retained ownership anchor; hung EOF stays bounded and unknown", + "red": {"tests":3,"failures":4,"seconds":0.745}, + "agent_gates": [{"python":"3.12","tests":40,"seconds":12.894},{"python":"3.14","tests":40,"seconds":13.398}], + "root_diagnostics": {"tests":34,"seconds":11.159,"failures":0,"errors":0,"resource_warnings":0}, + "root_council_sdk_reconciliation": {"tests":8,"seconds":21.651,"failures":0,"errors":0}, + "packaging_regression": {"red":{"tests":1,"seconds":2.657,"failures":1,"reason":"Council schema omitted from wheel"},"green":{"tests":1,"seconds":2.831,"failures":0,"errors":0},"scope":"actual plain installed wheel packages every source schema"}, + "final_native_full_install": "pending" + }, + "contract": "SOL-REVIEW-FOLLOWUP R6/G3/G4; unchanged low-level MCP envelopes, fenced host disposition, Bash 3.2 and optional jq", + "implemented": [ + "Readable normal terminal commands; --json retains the versioned automation envelope", + "Guided finish binds the exact current packet/artifact hashes and existing independent-review/check/claim fences; rejects replay and app-owned prior host claims", + "Finish intent and claim commit atomically; only an identical canonical decision whose latest acquisition event matches the current packet/owner/fence/expiry recovers its live or expired guided claim. App claims with the same owner name and later renewals cannot reuse old terminal authority", + "Saved submissions continue through resume; human status prints a copyable, shell-quoted --reason= exact intent retry command with separate guidance, including negative-leading reasons", + "A prior authoritative expired_claim rejection can recover only with exact latest guided intent and a fresh live matching claim; the complete rejected row and digest are atomically preserved in append-only history before the operational projection becomes recorded. Original rejection and recovery/submission consecutive event versions remain visible; no schema or decision-ID rewrite", + "Omitted IDs require exactly one canonical current-project run; zero and multiple choices are explicit; unrelated runs never selected", + "Doctor distinguishes installed, registered, exact supported version, subscription authentication, workflow readiness and unknown operation proof", + "Bounded non-generating Claude auth and Codex account/read with no API keys or alternate credential-home overrides", + "Committed target check discovery includes Bash, actual Python core tests and generated reference; exact argv dedup and byte/NUL/regular-blob safety retained", + "Fresh source install, actual installed normal review/fix discovery and guided receipt path with only native provider binaries replaced by offline fixtures" + ], + "agent_gates": [ + {"revision": "dd709804", "tests": 37, "seconds": 7.852, "optional_sdk_skips": 2, "bash_assertions": 227}, + {"revision": "79f9b767", "tests": 41, "seconds": 10.391, "optional_sdk_skips": 2, "bash_assertions": 227}, + {"revision": "b785d040", "tests": 103, "seconds": 89.115, "skips": 0, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass"}, + {"revision": "d0eec55c", "tests": 1, "seconds": 11.955, "skips": 0, "failures": 0, "errors": 0, "resource_warnings": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass"}, + {"revision": "4a6bc15d", "tests": 41, "seconds": 11.356, "optional_sdk_skips": 2, "bash_assertions": 227}, + {"revision": "03124c79", "tests": 106, "seconds": 95.408, "skips": 0, "failures": 0, "errors": 0, "resource_warnings": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass"}, + {"revision": "22a4ed8a", "tests": 150, "seconds": 180.750, "failures": 0, "errors": 0, "scope": "Affected repair gate before final command-copyability edit"}, + {"revision": "22a4ed8a", "tests": 45, "seconds": 48.316, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Final CLI/terminal repair checkpoint"}, + {"revision": "31a8a68b", "tests": 156, "seconds": 350.005, "skips": 0, "failures": 0, "errors": 0, "resource_warnings": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Expiry rejection and command round-trip repair, before final original-rejection-event crosscheck"}, + {"revision": "e7e9042c", "tests": 19, "seconds": 19.207, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Final original-rejection-event crosscheck; 13 negative audit-corruption subcases"}, + {"revision":"093ed0f397289bffe9ea249c10796cb436027f00","tests":36,"seconds":11.250,"python":"accepted immutable3.12","resource_warnings":0,"failures":0,"errors":0,"bash_assertions":227,"generated_reference":"pass","diff_check":"pass","scope":"Unavailable capture/current ownership, kernel direct-child-only cleanup, retained exited zombie and TERM-ignoring descendant/EOF cleanup; exclusive Popen-reaping contract"}, + {"revision":"093ed0f397289bffe9ea249c10796cb436027f00","tests":36,"seconds":11.364,"python":"3.14","resource_warnings":0,"failures":0,"errors":0,"scope":"Same affected ownership and native-discovery regression set"} + ], + "root_affected_gates": [ + {"command": "python3 -m unittest test_diagnostics test_terminal_ux test_installed_usability test_cli test_task_entry test_mcp test_handoff_store", "tests": 114, "seconds": 65.066, "optional_sdk_skips": 2, "failures": 0, "errors": 0, "scope": "Integrated first five checkpoints, before final catalog cleanup reuse"}, + {"command": "python3 -m unittest test_delivery_workflow test_delivery_identity_runtime test_handoff_service", "tests": 32, "seconds": 66.877, "failures": 0, "errors": 0}, + {"command": "python3 -m unittest test_task_entry test_installed_usability", "tests": 27, "seconds": 25.389, "skips": 0, "failures": 0, "errors": 0, "scope": "All six integrated checkpoints including final catalog cleanup reuse"}, + {"command": "python3 -m unittest test_terminal_ux test_cli test_handoff_store test_handoff_service", "tests": 61, "seconds": 55.963, "failures": 0, "errors": 0, "scope": "Integrated initial interrupted-finish repair and current-schema assertion; ResourceWarning strict"}, + {"command": "python3 -m unittest test_handoff_store test_terminal_ux.TerminalUxTest.test_finish_retries_authoritative_expiry_rejection_with_exact_fresh_claim test_terminal_ux.TerminalUxTest.test_pending_finish_negative_leading_reason_round_trips_as_actual_cli", "tests": 19, "seconds": 7.183, "failures": 0, "errors": 0, "scope": "All recovery follow-ups including final exact original rejection audit; ResourceWarning strict"}, + {"tests":7,"seconds":1.259,"python":"accepted immutable3.12","resource_warnings":0,"failures":0,"errors":0,"source_blobs_equivalent_to":"093ed0f397289bffe9ea249c10796cb436027f00","scope":"Root independent capture/currentNone, no protocol/false readiness, direct-child absence, bounded zombie authority and EOF descendant checks"} + ], + "failure_history": [ + "Readiness red baseline: nine tests, three failures and 17 subtest errors before implementation", + "Terminal red baseline: six tests, four failures and two errors before implementation", + "Root review reproduced a surviving SIGTERM-ignoring descendant after the diagnostic parent exited; repaired by 79f9b767 with process-group absence/ownership checks and bounded escalation", + "Catalog discovery independently reproduced the same surviving-descendant defect (one failed test/0.133s); shared helper reuse in 03124c79 passes three focused tests/1.004s plus 106 affected tests. Root final gate still required", + "Fresh installed fixture seeded check wrote original-checkout bytecode; PYTHONDONTWRITEBYTECODE=1 restored the declared pristine fixture contract without weakening source gates", + "Root integrated affected command ran 116 tests in 69.606s, two optional SDK skips and two loader errors from nonexistent test_review_service/test_delivery_service module names; not a passing command gate and not a runtime defect", + "A separate follow-up invocation mistakenly included nonexistent test_delivery_runtime; corrected existing-module/full gates are required", + "Independent bounded terminal review reproduced a P1: interruption after guided finish claim commit before completion strands the run without saved claim; same finish retry rejects both live and expired own claim. Durable exact-intent recovery is integrated at 22a4ed8a; final root/full/native acceptance remains pending", + "Repair red reproduction: three selected tests had one failure and two errors before durable intent implementation", + "First expanded repair-agent gate failed a stale isolated bf3 current-schema expectation of 15; root already expected 16 after dfe9976. Changed only the current-schema assertion to SUPPORTED_SCHEMA_VERSION and retained the historical schema-four fixture checks; not a current root runtime failure", + "Independent 22a4ed8a follow-up: ten existing repair tests pass in 26.811s, but a lease expiring after claim acquisition and before submission creates a durable expired_claim rejection. A fresh own retry advances fence 2 to 3/version 16 to 17 yet the stable decision replays that old rejection indefinitely. Follow-up must preserve complete immutable rejection history while recovering only exact fresh guided authority", + "Independent 22a4ed8a command-copyability edge: saved reason --deferred is printed as --reason --deferred and argparse rejects it. Fixed by 31a8a68b attached shell-quoted --reason= and actual command round-trip", + "Follow-up red reproduction: two tests/3.845s, one failure and one error. Initial audit implementation reused a run_version for two events and failed uniqueness; consecutive atomic recovery/submitted versions repaired it before green checkpoint gates", + "Exact native audit44d0eecb found a high process-ownership defect: start_identity=None/current_identity=None/returncode=None permits group signaling; diagnostic callers use output/protocol without confirmed captured identity. Candidate explicitly rejected, failed/version22; repair093ed0f now integrated, exact follow-up pending", + "Native required check ran66 tests/89.977s/OK but created nine undeclared cpython-314 bytecode files. Integrity correctly invalidated the check; subsequent reference not run. Corrected private invocation sets PYTHONDONTWRITEBYTECODE=1 without relaxing candidate integrity", + "Ownership repair red: six regressions/1.216s expose baseline signal/read/send/false-authentication behavior. First conservative macOS3.12 repair refused known exited-parent descendant cleanup because waitid is unavailable; fixture child was removed. Exact bounded retained zombie-child anchor and EOF-before-reap contract restored this required cleanup without unsafe missing-capture signals" + ], + "baseline_full_gate": { + "target_revision": "7130ee7689f8cc14efacb606bcbc4c05f373298c", + "frozen_root_head": "32b94d66e57ca227deb1cfda37411570af92c055", + "source_tests_scripts_equivalent": true, + "exec_session": 86382, + "command": "env DEVSQUAD_BUILD_PYTHON=/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3.12 PYTHONPATH=plugin/core/src:test/core PYTHONWARNINGS=error::ResourceWarning python3 scripts/run-core-tests.py --failfast", + "tests": 554, + "seconds": 645.034, + "monotonic_seconds": 645.1674028339985, + "utc_seconds": 645.167055, + "optional_sdk_skips": 2, + "failures": 0, + "errors": 0, + "unraisable": [], + "scope": "Baseline only; does not prove the pending ownership repair or Council integration" + }, + "native_audit": { + "run_id": "44d0eecb-b5da-4c09-b1e1-c38b0d1d09ca", + "target_revision": "7130ee7689f8cc14efacb606bcbc4c05f373298c", + "candidate_sha256": "2751f6b40a2125a83ee10bc6047beee378ec617bff0817c0e26666fc76d00ee6", + "state": "failed", + "version": 22, + "host_disposition": "reject", + "accept_allowed": false, + "review_verdict": "findings", + "finding": "r6-probe-ownership-unavailable", + "severity": "high", + "identity": {"harness":"codex","harness_version":"codex-cli 0.159.2","model_id":"gpt-6.1-sol","effort":"low","verification":"verified","permission_policy":"read_only"}, + "usage": {"source":"native_reported","input_tokens":276520,"output_tokens":2246,"total_tokens":278766}, + "usage_limitation": "Not a subscription invoice or an estimate of Plus five-hour windows", + "artifact_hashes_verified": [ + "c82d6c3d20b6b93ba93a3e0d1f6cac22bb6d5a80d453e47432dbd40120c0dd79", + "cdfe7b8dd8023747ac76539dd835f5b5b0fe3e71f833c31593287dc77b78e39c", + "8c911a4fa0be95aeeba341a75978dcb109f977b7bab91462d4e9150150e4e721", + "3ed7a9a766d3f19c0ff73c956dbb196ee550c15d86f58bf38827cb75820eca51" + ], + "checks": [ + {"id":"candidate-diff-check","status":"passed","integrity":"verified","duration_ms":20}, + {"id":"detected-tests","status":"passed","integrity":"verified","duration_ms":16912}, + {"id":"user-check-1","status":"invalidated","integrity":"violated","duration_ms":90564,"tests_ran":66,"test_seconds":89.977,"test_runner_result":"OK","reason":"check:undeclared_inputs_changed"}, + {"id":"user-check-2","status":"not_run","reason":"prior_check_invalidated_candidate"} + ] + }, + "independent_followup": {"revision": "31a8a68b", "result": "clean_bounded_scope", "tests": 6, "seconds": 9.120, "source_blobs_before_after": "unchanged", "full_five_file_diff_sha256": "b4c742fa400932d216d698e483af40abaa733c741325aebd3eaaf56776accf40", "final_original_event_hunk_review": {"revision": "e7e9042c", "result": "clean_bounded_scope", "tests": 1, "negative_subcases": 13, "seconds": 0.208, "source_before_after_committed": "equivalent", "two_file_diff_sha256": "39ba13f6e5b0cf06da5c55d619d7bd0b2a3684780129d16f8493640736b8cc65"}}, + "fresh_install_scope": { + "providers": "offline native CLI protocol fixtures only", + "installer": "actual source installer with no-index and temporary HOME/install/runtime/bin", + "internal_fixture_injection": false, + "manual_task_json": false, + "manual_decision_json": false, + "discovery_stub": false, + "verified_distinct_writer_reviewer": true, + "seeded_python_check": "fails original and passes candidate", + "original_head_source_porcelain": "unchanged", + "receipt_artifact_hashes": "verified", + "provider_generation_or_production_update": false + }, + "desktop_scope": { + "control": "available for initial scoped inspection; last IDE cancel attempt returned TCC denial, not bypassed", + "claude": "Local Code tab selected DevSquad on codex/engineering-team; empty prompt; no proof call sent", + "antigravity": "User clarified Antigravity CLI, not IDE. No IDE trust/settings/MCP edit was performed; folder chooser only inspected, cancel closure unverified after TCC denial. Updated agy CLI requires final-release status proof", + "unrelated_servers_or_settings": "not changed" + }, + "remaining_gates": [ + "Fail-closed unavailable process ownership regression/repair and independent exact follow-up", + "Final frozen integration full gate; baseline554 pass is not repaired-source acceptance", + "Independent exact R6 acceptance and immutable checkpoint", + "Safe installed update, non-generating live readiness, idempotence/drift and actual installed SDK tests", + "Separate Claude Code-tab if accessible, updated Antigravity CLI and R7/R8 gates; Antigravity IDE not required by user clarification" + ] +} diff --git a/docs/plans/engineering-team/evidence/R7-NATIVE-BOUNDARY-BLOCKER.md b/docs/plans/engineering-team/evidence/R7-NATIVE-BOUNDARY-BLOCKER.md new file mode 100644 index 0000000..43d5eaf --- /dev/null +++ b/docs/plans/engineering-team/evidence/R7-NATIVE-BOUNDARY-BLOCKER.md @@ -0,0 +1,51 @@ +# Native Council boundary remains unavailable + +The October 2 isolated source gate did not make a generating native request. +MacOS 27.0 (26A5378n), `/usr/bin/sandbox-exec`, and the exact ChatGPT bundled +Codex CLI 0.159.2 were used. The last bounded metadata probe at 22:35 UTC sent +app-server `initialize`, `initialized`, and `model/list` over stdio, with copied +private subscription auth in a canonical private scratch HOME/CODEX_HOME. +All processes were owned-group cleaned and streams closed. The scratch/auth +copy was removed; no provider diagnostics or credentials are tracked. + +The exact default-deny profile's own role evidence and scratch were readable; +peer directories, ledger, private logs and artifacts were not. Binary/system +dependencies were limited to the exact executable, `/usr/lib` and +`/System/Library`, plus literal ancestors. Narrow managed-config metadata (not +config bytes), `/dev/urandom`, CFPreferences daemon/agent Mach services, exact +UID/daemon CFPreferences shared-memory names, and outbound TCP port 443 were +added. Initialization worked and model/list returned eight catalog entries. +Those entries do not prove a successful backend request. + +The last diagnostic variant also allowed these facilities, none committed as +native support: Mach services `com.apple.system.opendirectoryd.libinfo`, +`com.apple.mDNSResponder`, `com.apple.SystemConfiguration.configd`, +`com.apple.networkd`; outbound UDP port 53; `system-info net.link.addr`; exact +`/private/etc/hosts` read. It still reported +`failed to refresh available models: Connection failed: error sending request`. +There was no successful HTTPS response or HTTP status. Error classification: +network/backend attestation unavailable, underlying cause **unknown**. + +Earlier local sandbox logs demonstrated denied `system-info net.link.addr` +and `/private/etc/hosts` reads. Allowing them was not sufficient. DNS-related +Mach/UDP additions were diagnostic hypotheses, not a proven root cause. +Other denied notification/logging/user-preference/Info.plist/networkd-plist +operations were not shown necessary and were not broadly granted. No broad +filesystem, managed configuration, MCP or Mach-lookup exception was added. + +`verify_native_network` now fails capability-unavailable before any native role +launch. Doctor reports implemented partial mechanics but native-ready false. +Next: root-owned bounded non-generating backend attestation under the exact +frozen boundary (such as a genuinely successful native account/rate-limits +response), with the same peer/runtime/log/artifact/symlink/child denial probes. +Initialization, cached/fallback model lists, network error suppression, fixture +quality or a requested-only model ID must not lift this gate. + +The R6/Council reconciliation delegates version and bootstrap cleanup to the +shared bounded probe helper, refusing missing captured ownership before any +bootstrap RPC. Shared ownership repair +`093ed0f397289bffe9ea249c10796cb436027f00` is a separate explicit acceptance +dependency, not imported into this isolated Council checkpoint. Earlier local +cleanup observations above are not acceptance of the helper's ownership logic. +Root must accept and integrate that repair before new native probes; exact +backend attestation remains independently unavailable afterwards. diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json new file mode 100644 index 0000000..971bb7f --- /dev/null +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -0,0 +1,45 @@ +{ + "schema_version": 1, + "status": "requested_runtime_update_workflows_regressions_and_final_gemini_cli_recheck_passed", + "final_installation": {"source_checkpoint": "6d2e0ba", "release": "0.1.0-py31214-01fad439adea-mcp-a26bc88afbef", "source_digest": "01fad439adea52330b9bcaa7bc61f294ebda6d7a7ccd8ab8e540eefd252767cf", "python": "3.12.14", "mcp": "2.2.0", "network_downloads": false, "nonterminal_runs_before": 0, "first_changed": true, "reinstall_changed": false, "all_payload_drift": false, "manifest_matches": true, "pip_check": "pass", "previous_releases_retained": true, "setup": {"ready": true, "hosts": 4, "unchanged": 4}, "doctor_ready": true, "installed_sdk": {"tests": 9, "seconds": 2.803, "result": "pass", "skips": 0}, "sdk_note": "Printed squad_events argument rejection is the expected negative-contract test, not a failed test"}, + "final_gemini_recheck": {"status": "pass_cli_mcp", "release": "0.1.0-py31214-01fad439adea-mcp-a26bc88afbef", "version": "1.2.13", "model": "gemini-3.8-flash-low", "tool": "devsquad/squad_status", "tool_calls": 1, "tool_response_ok": true, "observed": {"run_id": "360a4993-d016-406e-8bb4-ae6d57dfe6cf", "state": "succeeded", "version": 31}, "returncode": 0, "timed_out": false, "seconds": 12.117595374999837, "stdout_sha256": "32636cfdf65c0695ae389d392da3855ce1d3df35c404d559f4a61b99eae5627f", "stderr_sha256": "3a7f8550106d86abb322e8ed0fb30a0296c0e97c6be03d5a8033b99d16737459", "usage": {"input_tokens": 35090, "output_tokens": 213, "thinking_tokens": 0, "cache_read_tokens": 44857, "total_tokens": 35303}, "permissions": "existing project-only status grant, plan mode, sandbox; no bypass", "schema_read": "only DevSquad squad_status tool schema", "ide_ui": "unverified_permission_denied_do_not_bypass"}, + "fresh_byte_correction": {"run_id":"360a4993-d016-406e-8bb4-ae6d57dfe6cf","state_at_checkpoint":"running","version":4,"target":"af994e948eb38764884d27fd61665efa776cb47b","scope":"two writable files, no revisions, two workers, 1800s wall budget; original frozen task unchanged","mandatory_checks":"diff, Bash and full spawn-safe core suite with explicit offline build interpreter","operator_red":{"case":"real Git index/tree non-UTF-8 documentation and test filenames","result":"both raise UnicodeDecodeError on af994","fixture_note":"Git plumbing inserts byte names into the index; macOS APFS rejects such working-tree names"},"next":"finish candidate/native review/full gates, then narrow full-combined review, integration, offline install and one Gemini recheck","terminal":{"state":"succeeded","version":31,"host":"accept"},"candidate":{"commit":"f8c4f8c83f170870eb37d564e54eee6188fc233c","sha256":"3f0fa0e86edd8a4c7b04bc1a2447d70b136e596cafe3624826f1f46c2fcee568"},"writer":{"model":"claude-sonnet-5","verification":"verified","harness":"2.1.220 (Claude Code)","input_tokens":12,"output_tokens":12079,"total_tokens":12091},"review":{"model":"gpt-6.1-sol","effort":"low","verification":"verified","permission_policy":"read_only","verdict":"clean","input_tokens":90119,"output_tokens":481,"total_tokens":90600},"gates":{"checks":"all_three_passed_unchanged_verified_integrity","bash_assertions":227,"full_core":{"tests":477,"unittest_seconds":455.86,"monotonic_seconds":456.2232173750008,"utc_seconds":456.21989,"errors":0,"failures":0,"skips":2,"unraisable":[],"result":"pass"},"checks_artifact_sha256":"ee7d7c903872364b6c3172d2b68084f8e352578ef666564e58dcbd1a63985820"},"operator_recheck":"All five README/symlink/Unicode/byte-documentation/byte-test cases pass"}, + "byte_fix_combined_review": {"run_id":"a6ecd887-5122-4281-b988-4d344a662e20","state_at_checkpoint":"running","version":4,"base":"1781b89b1228dfca2ca9148df437af3ac04b971f","target":"f8c4f8c83f170870eb37d564e54eee6188fc233c","scope":"five changed files","checks":"mandatory diff, Bash and 27 affected tests; no duplicate full gate","terminal":{"state":"succeeded","version":22,"host":"accept"},"review":{"model":"gpt-6.1-sol","effort":"low","verification":"verified","verdict":"clean","input_tokens":76865,"output_tokens":716,"total_tokens":77581},"gates":{"checks":"all_three_passed_with_unchanged_verified_integrity","tests":27,"seconds":24.173,"checks_artifact_sha256":"4bbc33c831fdfeed25f3329dd51ed5e0487648bfdf5ada047d23e3fdd692b30e"}}, + "source_integration": {"candidate":"f8c4f8c83f170870eb37d564e54eee6188fc233c","exact_blob_matches":[{"path":"plugin/core/src/devsquad/task_entry.py","blob":"d350e8409fcb3a47a9d1d10a81dbbc2be184b308","matched":true},{"path":"test/core/test_task_entry.py","blob":"907596e10915f3935c4dbe636f75e044021c6ae5","matched":true},{"path":"test/core/test_cli.py","blob":"8acb256d06a6b0ca8c1cfd8b7be85f7dde6baf82","matched":true},{"path":"test/core/test_handoff_store.py","blob":"2293bfd9ad57c5749b539438191ee62ae4624bff","matched":true},{"path":"test/core/test_mcp.py","blob":"dccf29017f77580c3f7500ff321c88e254052676","matched":true}],"affected":{"tests":27,"seconds":18.94,"result":"pass"},"bash_assertions":227,"reference":"current","diff_check":"pass","full_gate":"accepted exact-candidate 477-test run; not unnecessarily repeated"}, + "source_checkpoint": "c7ccc02", + "installation": { + "previous_release": "0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", + "release": "0.1.0-py31214-674889018f98-mcp-a26bc88afbef", + "source_digest": "674889018f980ce648c6ac349d46253a723698206f8c6994454aa947e6efd4ab", + "source_plugin_installed_drift": false, + "python": "3.12.14", "mcp": "2.2.0", "network_downloads": false, + "first_install_changed": true, "reinstall_changed": false, + "pip_check": "pass", "previous_release_retained": true, + "ledger_backup": "/Users/Dikshant/.devsquad/backups/runtime-before-schema15-20261001.sqlite3", + "schema_before": 13, "schema_after_first_access": 15, + "nonterminal_runs_before": 0, + "old_result_readback": {"run_id": "c611ad4c-5473-4f05-a870-a136f24464d3", "state": "succeeded", "version": 22}, + "setup": {"hosts": ["antigravity", "claude-code", "codex", "grok"], "matching": 4, "changed": 0}, + "sdk_tests": {"tests": 22, "seconds": 5.585, "skips": 0, "result": "pass"} + }, + "delivery": {"status": "initial_attempt_failed_before_implementation", "run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13, "worker_invocations": 1, "reported_tokens": 0, "cause": "Claude variadic --tools consumes the final positional prompt without an explicit -- separator", "scope": "G4 exact-target check discovery and inclusion of Python core regressions"}, + "second_installation": {"source_checkpoint": "6f51db9", "release": "0.1.0-py31214-7d7e408303b3-mcp-a26bc88afbef", "source_digest": "7d7e408303b3f17043c0f413d2d9828feb390a73dacff0bce358921fe1802365", "payload_drift": false, "pip_check": "pass", "prior_releases_retained": true}, + "auth_repaired_installation": {"source_checkpoint": "1781b89", "release": "0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef", "source_digest": "68d542f6e8ea69c9bdefdfd4625af45ea46a09fb7926964397cbad3d96c83f28", "payload_drift": false, "pip_check": "pass", "previous_releases_retained": true}, + "auth_repaired_delivery": {"run_id": "45667697-1aa2-4a49-9737-2c4e637ebd26", "status": "cancelled", "version": 17, "candidate_produced": true, "candidate_commit": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "candidate_patch_sha256": "bb47b4b5522826609cabc335e0984b99ffd8311ad620a4e7cfe04c05a6906e26", "writer": {"model": "claude-sonnet-5", "verification": "verified", "model_source": "claude.stream.assistant.message.model", "reported_usage": {"input_tokens": 36, "output_tokens": 29404, "total_tokens": 29440}}, "review_and_checks": "no_completed_evidence", "reviewer_failure": "runner and child died without an exit receipt; exact cause unknown; zero capture/log bytes; dead ownership safely cancelled", "scope": "Genuine installed Claude implementation passed, but independent review/test acceptance did not complete"}, + "candidate_correction": {"run_id":"288ee6f9-f503-421a-b81a-d50ee0cf44d8","status":"failed","requested_base":"1781b89b1228dfca2ca9148df437af3ac04b971f","target":"5cb9eab81bf4f7ba01861ffb05193a8a67521200","candidate_commit":"6375955f66ce87a541fdfbf60db6eb0fcd30a55e","candidate_sha256":"77d16f59fcf8594615b25f774b3f106f4883565ff83a613453469c045be91b8a","operator_red_cases":["README-only tests tree incorrectly selects Python discovery","tests symlink incorrectly selects Python discovery"],"operator_recheck":"both pass on corrective candidate","remaining_operator_finding":"Git-quoted Unicode test filename test_π.py is missed; reproduced offline, requires fenced revision","actual_review_base":"5cb9eab81bf4f7ba01861ffb05193a8a67521200","review_scope_note":"Frozen delivery review uses implementer baseline, not original requested task base; full combined branch review remains required","mandatory_checks":"Bash plus spawn-safe full Python core runner, coordinator-owned","version":31}, + "corrective_gate": {"run_id": "288ee6f9-f503-421a-b81a-d50ee0cf44d8", "review": "verified native gpt-6.1-sol/low, clean; summary explicitly says full G4 diff inspected, frozen comparison 5cb9eab to 6375955", "bash_diff": "passed_with_verified_integrity", "full_core": {"tests": 471, "seconds": 473.7623341669996, "utc_seconds": 473.758578, "errors": 3, "failures": 0, "skips": 2, "unraisable": [], "result": "failed"}, "cause": "Three installed-wheel build_python helpers raise FileNotFoundError for nonexistent HOME-relative cached interpreter under deliberately isolated HOME instead of probing next candidate", "focused_reproduction": {"tests": 2, "errors": 2}, "host_disposition": "reject", "terminal": {"state": "failed", "version": 31}}, + "final_runtime_repair": {"run_id":"33c64397-3a57-42e7-82d5-773d50f466ee","status":"failed","version":37,"target":"6375955f66ce87a541fdfbf60db6eb0fcd30a55e","candidate":"af994e948eb38764884d27fd61665efa776cb47b","candidate_sha256":"efb502a38972a97aba89cce371b398e2ac7e7dd1b6dd9b0a8fa83a51d98d9f5c","writer":"verified native claude-sonnet-5","reviewer":"verified native gpt-6.1-sol/low, read_only","review":"clean","checks":"all_three_passed_with_verified_unchanged_integrity","full_core":{"tests":475,"unittest_seconds":460.153,"monotonic_seconds":460.51592204100007,"utc_seconds":460.515864,"errors":0,"failures":0,"skips":2,"unraisable":[],"result":"pass"},"accept_allowed":null,"checks_artifact_sha256":"5b543be7cb05ed8dbf7e66c60aee12fba16e8401332da5815f14569112fe156e","operator_cases":"README and symlink negatives plus Unicode positive pass","scope":"NUL-safe filename discovery plus three unavailable-interpreter wheel-test helpers and regressions","required_full_gate":"Explicit existing offline DEVSQUAD_BUILD_PYTHON, isolated HOME and strict ResourceWarning; installed-wheel gates execute","gate_scope":"First iteration only; later fenced revision failed, no host acceptance"}, + "final_combined_review": {"run_id":"ba14b743-9473-46fe-9e1f-d5b79fb032d4","status":"failed","base":"1781b89b1228dfca2ca9148df437af3ac04b971f","target":"af994e948eb38764884d27fd61665efa776cb47b","scope":"five_changed_files","checks":"mandatory diff, Bash and affected task-entry/wheel tests; no duplicate full gate","version":22}, + "combined_review_finding": {"run_id": "ba14b743-9473-46fe-9e1f-d5b79fb032d4", "verdict": "findings", "finding": {"id": "g4-filename-decoding", "severity": "medium", "cause": "Strict text=True decoding rejects valid Git non-UTF-8 filenames before test filtering"}, "required_checks": "all_passed_with_verified_integrity", "affected_tests": {"tests": 25, "seconds": 24.390, "result": "pass"}, "host": "reject", "terminal": {"state": "failed", "version": 22}}, + "byte_filename_revision": {"run_id":"33c64397-3a57-42e7-82d5-773d50f466ee","host":"revise","state_at_submission":"queued","version":26,"repair_requested":"Read NUL Git output as bytes with UTF-8 surrogateescape; add real non-UTF-8 documentation plus valid-test regression","scope_and_budget":"unchanged_frozen_contract","remaining_gates":"new candidate native review/full suite plus final combined scope review, integration and final install","terminal":{"state":"failed","version":37,"error":"WORKFLOW_OUTPUT_INVALID","message":"delivery candidate contains no changes"},"stdout_sha256":"767af9da8716b5eda3922a9a299108fc99e67ad0d77c0efb015e714e79affffc","diagnosis":"Saved revision reason present and normalized prompt hash validated; verified native writer summary incorrectly says original task already satisfied. Worktree unchanged at af994. Not a quota timeout."}, + "delivery_retry": {"run_id":"8c116b7d-3416-4f80-a990-d610e2a5a558","state":"failed","version":13,"worker_invocations":1,"reason":"native_result_invalid","output_bytes":6183,"output_sha256":"0886c28213a28db6617d2fa2e75a2cb0eb772007ba3006bdcfce598537fe9485","candidate_produced":false,"review_and_checks":"not_started","cause":"Reproduced missing HOME/USER detached saved-login boundary; fixed at 1781b89, subsequent genuine writer succeeds."}, + "detached_auth_diagnosis": {"result": "reproduced_and_source_repaired", "original_environment": "PATH and frozen PYTHONPATH, without HOME/USER", "native_failure": "Synthetic assistant model, is_error true and not-logged-in terminal; no model usage", "home_only": "still_failed", "home_plus_user": "pass_one_Read_with_verified_claude_sonnet_5", "failure_stream_sha256": "f7f7a53de76012c54c4a0757aa346124553366bf6b1f0c120cb7288ce1491ab7", "home_user_stream_sha256": "1061f1ca7688fa178b4add47934c5341287d02c66a4ba468d91343398fa830d3", "red_regressions": {"tests": 2, "errors": 2}, "repair": "Allowlist HOME and USER in detached launch; classify synthetic native error terminals without claiming writer identity. API keys and provider overrides remain excluded.", "focused_gate": {"tests": 50, "seconds": 69.606, "result": "pass"}, "auth_priority_rerun": {"tests": 22, "seconds": 3.652, "result": "pass"}, "verification": "source_targeted_pass; managed_delivery_still_pending"}, + "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, + "claude_full_gate": {"revision": "3555a91", "tests": 458, "seconds": 467.018, "errors": 0, "failures": 0, "skips": 2, "unraisable": [], "result": "pass"}, + "claude_provider_label_gate": {"tests": 63, "seconds": 70.391, "result": "pass", "scope": "Version-bound firstParty label normalization added after the 458-test gate; real saved native stream decodes to verified Sonnet 5 with auxiliary usage retained. The final managed candidate's mandatory full suite must include this small mapping patch."}, + "claude_independent_review": {"revision": "3555a91", "run_id": "b9501b55-1c65-4b32-a03f-1087cca8fefc", "model": "gpt-6.1-sol", "effort": "low", "verdict": "clean", "declared_checks": 3, "check_result": "all_passed", "terminal_state": "succeeded", "terminal_version": 23, "scope": "Argument framing and correlated native stream evidence; subsequent version-bound provider label patch is separately targeted and will be inspected in delivery review."}, + "claude_handoff": {"status": "pass", "run_id": "b9501b55-1c65-4b32-a03f-1087cca8fefc", "harness": "Claude Code CLI", "version": "2.1.220", "writer_model": "claude-sonnet-5", "calls": ["ToolSearch", "squad_status", "squad_handoff_claim", "squad_handoff_complete", "squad_status"], "terminal": {"state": "succeeded", "version": 23}, "permission_denials": 0, "stdout_sha256": "2b3aa171838e222249de7663c8d5ead499fc726975a516377bca32d7691308cb", "seconds": 38.61665333295241, "prior_attempts": ["Zero-tool probe ended with MCP still pending and did not invoke any tool", "ToolSearch-enabled probe claimed but falsely reported an artifact hash mismatch; deterministic exact comparison proved all refs equal. Its claim was retained and renewed before actual completion."], "usage": {"input_tokens": 12, "output_tokens": 2347, "cache_read_input_tokens": 77771, "cache_creation_input_tokens": 16166}, "scope": "Actual portable CLI/MCP handoff and acceptance, not Claude desktop local Code-tab UI proof"}, + "grok_operation": {"status": "pass", "old_version_blocker": "HTTP 426: 0.2.111 rejected, requires 1.0.13 or later", "user_authorized_update": true, "updated_version": "1.0.46 (2765805b9442) [stable]", "old_executable_backup": "private_probe_directory", "native_model": "grok-4.7-build", "tool": "devsquad__squad_status", "observed": {"run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13}, "stdout_sha256": "bc031b922e8c36b3cdfff5c7c717916a27019a19d650f67cd7402b832544bfdd", "stderr_sha256": "1be68163f727c0712fefcd45b80d5f1d383ee995ff8f61f7da601fe82ee4a9b3", "usage": {"input_tokens": 34088, "cache_read_input_tokens": 69888, "output_tokens": 434, "reasoning_tokens": 285, "total_tokens": 104410, "turns": 3}, "extra_bounded_smoke": {"result": "pass", "stdout_sha256": "cbc772baa3dabf93b833cf9843c5a873bb72cbebf9fb8ddf7d9b5ae7f12a8f43", "total_tokens": 103523}, "scope": "native Grok Build MCP operation, not a verified implementation/reviewer adapter"}, + "gemini_antigravity": {"status": "pass_cli_mcp", "version": "1.2.13", "native_model": "gemini-3.8-flash-low", "tool": "devsquad/squad_status", "permission": "existing project-scoped status grant, plan mode and sandbox", "schema_discovery": "read only DevSquad squad_status tool schema", "observed": {"run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13}, "seconds": 10.62698745902162, "stdout_sha256": "8781e538501b3553c100ca754d2880c8f45a011062eddd5fd3b108cf0878181d", "usage": {"input_tokens": 33060, "output_tokens": 220, "thinking_tokens": 0, "cache_read_tokens": 44852, "total_tokens": 33280}, "initial_probe": "Argument parser rejected unitless print timeout before generation; corrected to 120s", "ide_ui": "computer_use_permission_denied_do_not_bypass"}, + "limitations": ["R4 catalog/quota, R5 public trials/outcomes, remaining R6 UX and R7 Council are separate open scope", "Jev runtime remains off and its one-request allowance is spent", "No paid API fallback, purchases, resets, global AI settings, push or external messages"] +} diff --git a/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json new file mode 100644 index 0000000..01a72ba --- /dev/null +++ b/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json @@ -0,0 +1,35 @@ +{ + "schema_version": 1, + "reviewed_revision": "e8cdf07", + "status": "upgrade_follow_up_and_full_gate_pass_installed_proofs_pending", + "run_id": "fcc030af-26ec-423c-b462-b6fc0f8b55d1", + "runtime": "/Users/Dikshant/.devsquad/private-probes/r3-bounded-repair-review-20261001", + "review_sha256": "13b77ee0812c9beefda98df697af4189c132c7e311df995c7b78e9f6282b8644", + "review_attempt_sha256": "5517920d44aa8eaa66bffbf1d147fe8e64bc8acc16be3d62f2c8c38ddcd6bb74", + "checks_sha256": "1e94a349e0bc098ce23f3b24332ef36aed099c23eb713463f07e1965e774805d", + "reviewer": {"harness": "codex", "version": "0.159.2", "model": "gpt-6.1-sol", "effort": "high", "permission": "read_only", "verification": "verified"}, + "finding": {"id": "upgrade-activation-window", "severity": "high", "summary": "Migration committed while the old release selector remained active; pre-opened old clients could admit work after the pending-run check."}, + "host_disposition": "reject", + "terminal": {"state": "failed", "version": 22}, + "check": {"command": "bash test/run.sh", "returncode": 0, "assertions": 227, "integrity": "verified", "changed_inputs": []}, + "usage": {"input_tokens": 390975, "output_tokens": 4422, "total_tokens": 395397, "source": "native_reported", "worker_invocations": 1, "native_model_requests": null}, + "repair": "Readiness/selector swap under one ledger lock without advancing schema. New-release lazy migration transactionally installs connection-version write guards. Old active/recoverable runs defer; late preopened old clients fail before writing. Previous releases and receipts retained.", + "targeted": {"tests": 12, "seconds": 9.109, "result": "pass"}, + "evidence_process_regressions": {"tests": 32, "seconds": 137.768, "result": "pass"}, + "bash": {"assertions": 227, "files": 11, "result": "pass"}, + "clock_fixture_gate": {"tests": 20, "seconds": 128.937, "result": "pass", "production_clock_skew_validation": "unchanged"}, + "full_gate_after_fixture_repair": {"revision": "dc68944", "tests": 455, "seconds": 485.168, "errors": 0, "failures": 0, "skips": 2, "unraisable": [], "monotonic_seconds": 485.3602193329716, "utc_seconds": 485.364056, "result": "pass", "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 scripts/run-core-tests.py --failfast"}, + "full_gate": {"tests": 228, "seconds": 313.702, "errors": 1, "failures": 0, "unraisable": [], "result": "failed_failfast", "cause": "Lifecycle test injected module-import time more than five minutes before real outcome observations; reproduced with ten-minute-old clock. Production skew fence unchanged."}, + "upgrade_follow_up": { + "revision": "6874de7", "run_id": "3c89bb19-64ec-4600-9080-436df89edcfb", + "runtime": "/Users/Dikshant/.devsquad/private-probes/r8-upgrade-follow-up-20261001", + "reviewer": {"harness": "codex", "version": "0.159.2", "model": "gpt-6.1-sol", "effort": "low", "permission": "read_only", "verification": "verified"}, + "verdict": "clean", "required_checks": 2, "checks": "passed", "integrity": "verified_unchanged", + "host_disposition": "accept", "terminal": {"state": "succeeded", "version": 22}, + "review_sha256": "68d90b7b05a382b665df7862af528c97457451dc388f5140d39033d5c272fa2f", + "checks_sha256": "77de92c3ae4eec05188f956a995586d273ab10abd49eeaf693f4cee76fe495a3", + "review_attempt_sha256": "059d4286fbc991ec9cd169f7869a7ed68c2d3c2405ae247eaf18ac832d005f37", + "usage": {"input_tokens": 81225, "output_tokens": 675, "total_tokens": 81900, "worker_invocations": 1, "native_model_requests": null} + }, + "installed_refresh": false +} diff --git a/docs/plans/engineering-team/evidence/legacy-watchdog-release-repair.json b/docs/plans/engineering-team/evidence/legacy-watchdog-release-repair.json new file mode 100644 index 0000000..012903f --- /dev/null +++ b/docs/plans/engineering-team/evidence/legacy-watchdog-release-repair.json @@ -0,0 +1,111 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "Isolated portable legacy watchdog and narrow offline CI plumbing repair; not release publication or whole-plan acceptance.", + "baseline_revision": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", + "branch": "codex/release-watchdog", + "preserved_branch": "codex/council-integration", + "implementation_blobs": { + "plugin/lib/adapter.sh": "1847357c19677f276b06886386e2ac93e586b203", + "test/test_m1_legacy.sh": "8cc3fe9f9ee9df77b2fcb2d75c76c236a1f8d13a", + ".github/workflows/offline.yml": "2d231113b575a2568d3c51844f4551f0f30d7c70" + }, + "watchdog_diff_sha256": "51fd52c51c51b870d18d98d4d573b3031e0bd7414d9ee85f9434286c703b8cc7", + "cause": "The original portable watchdog decremented timeout_secs * 20 polls. Each external ps/tr probe and sleep consumed additional elapsed time, so configured deadlines drifted on macOS Bash 3.2.", + "repair": { + "deadline": "Background Bash builtin read -t on a private duplex FIFO has a real deadline independent of process-probe/polling cost. The parent observes timer completion before reading its fully published timeout/error marker.", + "cancellation": "A command-scoped parent FD9 writes cancel and remains duplex-open through wait. Cancellation before the timer opens remains buffered; cancellation after natural timer exit cannot block. No numeric timer signals or external sleep timer are used.", + "ownership": "Bash's owned job table observes direct-child completion. Only the existing CLI/descendant timeout cleanup receives TERM/KILL; no CLI wrapper or new process-authority contract was introduced.", + "resources": "mktemp creates the private mode700 directory; FIFO mode is600. Timer stdio is /dev/null, its inherited EXIT trap is reset, and the private FIFO is not inherited by the CLI or descendants. Command-local cancellation redirection restores caller FD9.", + "exit_scope": "The original EXIT trap is captured with builtin trap -p. Active-scope interruption still cancels/reaps the timer and cleans owned paths. Normal completion reads and removes capture files, then restores the shell-generated original trap before any classification return; caller globals cannot become cleanup targets.", + "compatibility": "Stock macOS Bash3.2, optional jq, no GNU timeout or Python runtime dependency added. Actual CLI nonzero status remains visible in CLI_ERROR; actual deadline retains TIMEOUT and normalized wrapper exit1. Existing descendant escalation and timing thresholds are unchanged." + }, + "failure_history": [ + { + "kind": "reported_CI_release_blocker", + "reported_by": "root", + "run_id": "37082130929", + "job_id": "111084739785", + "observation": "macOS legacy gate reported 4s for a1s deadline and6.123s/8.644s for a3s TERM-ignoring root; existing bounds were <4s/<5s.", + "agent_log_fetch": "gh run view --log-failed reported logs unavailable while the job was still running; no direct CI pass or complete-log inspection is claimed." + }, + { + "kind": "controlled_original_deadline_red", + "probe_delay_seconds": 0.2, + "deadline_seconds": 1, + "observed_elapsed_seconds": 5.848, + "initial_legacy_gate": "21 passed/2 failed: delayed-probe deadline plus an initial timer-observation assertion incompatible with the old poll-only implementation. The latter is instrumentation history, not another product defect." + }, + { + "kind": "rejected_uncommitted_timer_designs", + "observation": "TERM+wait on a sleep timer inherited ignored TERM and took2.096s on the fast path (independent helper2.025s). A jobs snapshot followed by numeric timer kill still had a stale-PID race; the independent signal observer recorded a former PID with kernel_live=false without sending a real unowned signal. These designs were replaced, not accepted." + }, + { + "kind": "controlled_EXIT_scope_red", + "prior_adapter_blob": "0789aae95e52272c5eb37e65745fdbb114486342", + "legacy_gate": "44 passed/6 failed", + "observation": "After invocation locals unwound, the new EXIT trap deleted same-named caller timer paths and lost the prior caller EXIT hook on portable success, GNU success and portable natural failure. All six regressions now pass." + }, + { + "kind": "test_only_scheduling_failure", + "prior_test_blob": "e6534aebed4fd16fd54b738b95e4c8e862650ab7", + "bash_gate": "1/11 files failed; legacy49 passed/1 failed", + "observation": "A fixed .2s fixture delay did not always force cancellation before FIFO open under combined-gate scheduling. Product timing bounds passed. The fixture now positively gates timer entry until the actual cancellation write succeeds, with parent FD9 retained; no timing assertion was widened." + } + ], + "verified_gates": { + "legacy_focused": { + "command": "/bin/bash test/test_m1_legacy.sh", + "bash": "3.2.57(1)-release arm64-apple-darwin26", + "assertions": 50, + "failures": 0, + "coverage": ["fast completion <1s", "inherited ignored TERM <1s", "actual buffered cancellation before timer FIFO open", "already-exited/reaped timer cancellation without numeric signals", "private directory/FIFO permissions and cleanup", "caller FD9 preservation", "provider/descendant without private FIFO FD", "CLI exit7 retained in CLI_ERROR", "caller EXIT hook and sentinel paths preserved on both branches", "active-scope exit17 cancels/reaps timer", "TERM-resistant root and descendants removed", "deliberately slow ps does not extend1s deadline"] + }, + "bash_all": { + "command": "/bin/bash test/run.sh", + "files": 11, + "assertions": 259, + "failures": 0, + "timings_seconds": {"fast": 0.273, "ignored_TERM": 0.197, "buffered_pre_open_cancel": 0.209, "TERM_resistant_3s": 3.246, "delayed_ps_1s": 1.243} + }, + "offline_python_affected": { + "command": "DEVSQUAD_BUILD_PYTHON=/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3 PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core /Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3 -m unittest test_m1 test_m1_gate_review test_adapters -q", + "tests": 45, + "seconds": 0.711, + "failures": 0, + "errors": 0, + "skips": 0 + }, + "other": ["/bin/bash -n plugin/lib/adapter.sh test/test_m1_legacy.sh", "scripts/generate-core-reference.py --check", "git diff --check"] + }, + "independent_review": { + "reviewer": "r6_readiness", + "scope": "Frozen adapter and legacy test only", + "result": "clean; all scoped findings resolved", + "legacy_assertions": 50, + "failures": 0, + "timings_seconds": {"fast": 0.194, "ignored_TERM": 0.202, "buffered_pre_open_cancel": 0.216, "TERM_resistant_3s": 3.236, "delayed_ps_1s": 1.244}, + "before_after_hashes_unchanged": true, + "confirmed": "The original private caller-path deletion reproduction now preserves its markers after shell EXIT. The final test-only gate releases FIFO open strictly after a successful cancellation write." + }, + "CI_plumbing": { + "core_checkout": "actions/checkout@v4 with fetch-depth:0 supplies historical archived installer fixtures.", + "core_runner": "python scripts/run-core-tests.py --failfast retains complete discovery, spawn-safe main, unraisable diagnostics and fail-fast behavior.", + "controlled_archive_proof": { + "clone": "Unique private local file:// --depth1 --single-branch clone of codex/release-watchdog; exact HEAD b83dda7, is-shallow=true; no network.", + "archives": ["f4fa657:plugin/core", "bf3de0867484552d354e6e8b6ba835f31a93aeb3:plugin/core"], + "shallow_exit_codes": [128, 128], + "shallow_error": "fatal: not a valid object name", + "original_full_history_exit_codes": [0, 0] + }, + "remote_CI_rerun": "Pending root integration/publication; no remote CI pass claimed." + }, + "unchanged_core": { + "plugin_core_tree": "62eea7fa31153ef732fd5e9dfd97951d95b367d8", + "python_tests_tree": "669eaf8487a7f2f983305162da33c1bda53116b5", + "core_runner_blob": "011416767e4e145fce6a49a0a38444cad49a68b7", + "prior_root_full_gate": "Root reported604 tests/793.462s, two optional SDK skips, zero failures/errors/unraisable on frozen b83dda7. This agent did not repeat the full suite." + }, + "boundaries": "No root source/ledger/installation changes, live providers, native audits, full core suite, publication or global AI settings. RESUME/backlog remain root-owned. Private controlled fixtures are outside Git.", + "next_action": "Root reviews CI config and integrates this checkpoint, records authoritative RESUME, rebuilds only changed final source/plugin assets, and owns remaining release/CI gates." +} diff --git a/docs/plans/engineering-team/evidence/public-release-0.11.0.json b/docs/plans/engineering-team/evidence/public-release-0.11.0.json new file mode 100644 index 0000000..7884de0 --- /dev/null +++ b/docs/plans/engineering-team/evidence/public-release-0.11.0.json @@ -0,0 +1,367 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "status": "delivery_phase_corrected_root_gate_passed_next_public_ci_pending", + "authorization": "User chose public GitHub release in existing joshidikshant/devsquad; no package registry or hosted service", + "pull_request": "https://github.com/joshidikshant/devsquad/pull/1", + "frozen_core_candidate": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", + "frozen_core_tree": "62eea7fa31153ef732fd5e9dfd97951d95b367d8", + "native_eof_followup": { + "run_id": "cb68f3ae-bea0-4744-90c4-96f25418fb8c", + "base": "8242eff68b24538aacc1b40b67b2e88e96b04d52", + "target": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", + "candidate_sha256": "f3243afeae0f011321d69fca1dc2f2225e4651f2a35e3452c1266cfa3da58791", + "verdict": "clean", + "findings": [], + "state": "succeeded", + "version": 22, + "host_disposition": "accept", + "identity": { + "harness": "codex", + "harness_version": "codex-cli0.159.2", + "model_id": "gpt-6.1-sol", + "effort": "low", + "verification": "verified", + "permission_policy": "read_only" + }, + "checks": { + "diff": "passed", + "bash": "passed", + "affected_tests": 60, + "seconds": 36.416, + "reference": "passed", + "all_required": true, + "all_integrity": "verified_unchanged" + }, + "artifact_hashes_verified": [ + "0dcb6d16b34c4f9c6e3eed6aa7e0161bf1d46c5f7f61370a7f8b80008dae5b69", + "46b3f1b3c8c1d8a7e824e4fef4502eb88841f27e1deaddf1b166359937f97283", + "d0f0b5b3e2169e34ed4ae63eacc7cdedce82b68fd9de4a904d01323f6910253f", + "182acd24afa59bc37c539f6fa8f0e6b1dd01275cfcd784970c2ffd8e21f00e32" + ], + "usage": { + "source": "native_reported", + "input_tokens": 95453, + "output_tokens": 517, + "total_tokens": 95970 + }, + "limits": "Narrow EOF/fixture audit only, not whole-plan or native Council acceptance; usage is not invoice/Plus-window estimate" + }, + "first_ci": { + "push_run": 37082086201, + "pr_run": 37082130929, + "status": "cancelled_superseded_after_confirmed_legacy_failure", + "legacy": "Actual Bash3.2 deadline failure: poll-count timing stretched1s to4s and3s to6.123s/8.644s", + "optional_mcp": "passed_both_runs", + "core": "cancelled_not_accepted; shallow checkout cannot provide immutable f4fa657/bf3 migration fixtures", + "repair": "Integrated3fdc6bd FIFO real-deadline watchdog, unchanged timing limits/caller EXIT restoration; full-history checkout and tracked spawn-safe failfast runner. Independent final source review clean.", + "no_passing_failure_claim": true + }, + "fresh_install_b83": { + "actual_core_only": true, + "version": "squad0.1.0", + "changed": true, + "reinstall_changed": false, + "status_changed": false, + "all_drift": false, + "manifest_matches": true, + "digest": "303a0e0a6c87ff472d1d7b52cb63ea56798a11f345f483736a18669c5f4d2a52", + "wheel_schemas": 11, + "wheel_migrations": 17, + "all_packaged_bytes_match_source": true, + "proof_sha256": "a0c7914ad5228276834ac242602fd510601f65562e406dd3dd75b31d107931ed", + "initial_failed_gate": "Whole-production filesystem equality invalidated by separately authorized concurrent root native ledger activity; perfield attribution unavailable, no unchanged-ledger claim", + "subsequent_validation": "Stable selector/manifest/installstate/launcher observed unchanged; unique private install/runtime/bin only", + "provider_or_network_calls": 0, + "production_update": false + }, + "final_full": { + "target": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", + "exec_session": 12430, + "tests": 604, + "seconds": 793.462, + "monotonic_seconds": 793.8633149590023, + "utc_seconds": 793.881373, + "failures": 0, + "errors": 0, + "optional_sdk_skips": 2, + "unraisable": [], + "status": "passed", + "source_tests_scripts_frozen": true + }, + "installation": { + "release": "0.1.0-py31214-303a0e0a6c87-mcp-a26bc88afbef", + "source_digest": "303a0e0a6c87ff472d1d7b52cb63ea56798a11f345f483736a18669c5f4d2a52", + "python": "3.12.14", + "mcp": "2.2.0", + "ledger_schema": 17, + "nonterminal_runs_before_and_after": 0, + "private_pre16_online_backup": "retained0600_integrity_ok", + "first_changed": true, + "reinstall_changed": false, + "all_drift": false, + "manifest_matches": true, + "pip_check": "passed", + "previous_releases_retained": true, + "downloads": 0, + "installed_sdk": { + "tests": 9, + "seconds": 3.126, + "failures": 0, + "errors": 0, + "skips": 0, + "origin": "selected immutable release/core/src, not primary checkout" + }, + "sdk_harness_first_assertion": "Initial helper assumed site-packages origin and aborted before tests; installer intentionally imports its immutable copied core/src. Corrected origin assertion retains release-bound/no-primary-source check; nine tests then passed.", + "doctor": { + "ready": true, + "branch_review_ready": true, + "issue_delivery_ready": true, + "matching_registrations": 4, + "claude_subscription_authenticated": true, + "codex_subscription_authenticated": true, + "council_native_ready": false, + "grok_antigravity_worker_readiness": "unverified/unknown, not implied by MCP registration" + }, + "saved_run_continuity": { + "run_id": "cb68f3ae-bea0-4744-90c4-96f25418fb8c", + "state": "succeeded", + "version": 22 + } + }, + "updated_antigravity_cli": { + "version": "1.2.14", + "observed_init_model": "gemini-3.8-flash-low", + "mode": "plan/sandbox; native request-review", + "operation": "one devsquad/squad_status call", + "run_id": "cb68f3ae-bea0-4744-90c4-96f25418fb8c", + "observed_state": "succeeded", + "observed_version": 22, + "seconds": 18.65435924999838, + "returncode": 0, + "timed_out": false, + "native_result": "SUCCESS", + "stdout_sha256": "cc6045878de2124b845adab33401ecd1f52317dd1f6e654b2b439041b2da1863", + "stderr_sha256": "3a7f8550106d86abb322e8ed0fb30a0296c0e97c6be03d5a8033b99d16737459", + "actual_mcp_output_sha256": "d22dccedbcf001999c59d6167820e04a833484f70a2d738e2e7f5c708b9a98f8", + "native_file_reads": "Own DevSquad MCP schema and provider-generated tool result file only; not zero file reads. Actual tool-result envelope independently parsed and matched saved run; no project files, other servers or mutations.", + "limits": "CLI MCP operation proof only; not Antigravity IDE, automatic writer/reviewer, cost savings or whole-plan acceptance" + }, + "codex_app_existing_connection": { + "operation": "squad_status cb68", + "result": "SCHEMA_UNSUPPORTED", + "reason": "existing long-running server is bound to old package; reconnect only DevSquad connection to select new release", + "new_cli_and_agy_same_run": "succeeded22", + "user_action": "async reconnect request pending; no global restart/settings changes" + }, + "limitations": { + "council": "native-unavailable, automaticOFF, fixture comparison inconclusive; no more network attempts", + "jev": "OFF; authorized single pilot spent", + "claude_desktop_code_tab": "unverified/TCC, not substituted by CLI proof", + "antigravity": "agyCLI1.2.14 updated-release MCP status verified; automatic worker roles unverified and IDE out of scope", + "whole_plan": "not complete" + }, + "remaining": "Final Bash/JSON/reference/diff checks passed. Checkpoint and push delivery-test-corrected candidate. Require all final exact-head public CI jobs green; merge PR1, publish exact-merge verified v0.11.0 assets, scoped local Claude plugin update. Preserve unconfirmed diagnostic concurrency follow-up plus original Council/desktop/MCP reconnect residuals separately.", + "integrated_legacy_gate": { + "source_checkpoint": "3fdc6bd065c577a9ed59566573fdce399e9b585a", + "adapter_blob": "1847357c19677f276b06886386e2ac93e586b203", + "test_blob": "8cc3fe9f9ee9df77b2fcb2d75c76c236a1f8d13a", + "independent_review": "clean; committed bytes equal independently reviewed/tested candidate", + "root_bash": { + "assertions": 259, + "files": 11, + "failures": 0, + "timings_seconds": { + "fast": 0.267, + "ignored_TERM": 0.211, + "pre_open_cancel": 0.207, + "TERM_resistant": 3.259, + "slow_ps": 1.291 + } + }, + "root_affected_python": { + "tests": 45, + "seconds": 0.927, + "failures": 0, + "errors": 0 + }, + "reference": "passed", + "core_tree_unchanged": true, + "failure_history": "evidence/legacy-watchdog-release-repair.json" + }, + "final_ci_attempt": { + "run_id": 37084805784, + "url": "https://github.com/joshidikshant/devsquad/actions/runs/37084805784", + "head": "5a576d39469c198a3cc93474960886b875363419", + "legacy": "passed259/11 files;50 focused", + "legacy_timings_seconds": { + "fast": 0.334, + "ignored_TERM": 0.335, + "pre_open_cancel": 0.221, + "TERM_resistant": 3.609, + "slow_ps": 1.518 + }, + "optional_mcp": "passed", + "python314": { + "version": "3.14.7", + "tests": 486, + "seconds": 679.063, + "failures": 1, + "errors": 0, + "skips": 2, + "unraisable": [], + "case": "test_coordinator_crash_imports_runner_receipt_once", + "assertion": "live not in succeeded/already_finalized", + "location": "test/core/test_service.py:1108", + "status": "failed_not_accepted" + }, + "python311": { + "tests": 604, + "seconds": 1013.419, + "failures": 0, + "errors": 0, + "skips": 2, + "unraisable": [], + "status": "passed_on5a_prior_to_test_synchronization_repair" + }, + "duplicate_push_run": 37084802403, + "duplicate_push_status": "cancelled_to_avoid_duplicate_CI", + "diagnosis": "Scheduling-only defect proven on exact5a controlled original1.2s fake child: original method failed same live assertion on3.12.14/3.14.6; synchronized test passes same controlled case. Product live ownership fence remains unchanged. Independent frozen patch review clean; affected module/Bash checkpoint pending.", + "status": "completed_failure;3.14 recovery-test timing assumption failed, other three jobs passed" + }, + "root_service_integration_attempt": { + "source_checkpoint": "edfcf65b92eeee97323b20db6d252b18e2fbc99a", + "test_blob": "e19495c811c5e52e3d66e5588d18f325e306a8ad", + "bash": "passed259/11files", + "service": { + "tests": 33, + "seconds": 32.435, + "failures": 1, + "case": "test_crash_after_runner_identity_before_gate_requeues_without_execution", + "location": "test/core/test_service.py:598", + "assertion": "ambiguous != dead", + "status": "failed_not_accepted" + }, + "concurrency": "Bash and service module ran concurrently on same host; no causal attribution asserted from concurrency alone", + "diagnosis": "Existing neighboring waiter stops at not-live yet assertsdead. Transient process identity ambiguity during exit is conservatively valid; bounded exactdead synchronization repair in isolated test only, independent review pending.", + "runtime_core_unchanged": true + }, + "corrected_service_gate": { + "receipt_checkpoint": "edfcf65b92eeee97323b20db6d252b18e2fbc99a", + "runner_exit_checkpoint": "e7b63b0d52808fdac109707abf9851ae9f3dabde", + "test_blob": "5456e21d10f7e65b06445a266ac0b177d8bd2066", + "core_runtime_unchanged": true, + "independent_reviews": "clean; both committed source-equivalent to reviewed candidates", + "agent_service": { + "python312_tests": 33, + "python312_seconds": 32.621, + "python314_tests": 33, + "python314_seconds": 35.47, + "python314_version": "3.14.6_not_CI3.14.7", + "failures": 0, + "errors": 0, + "skips": 0 + }, + "root_service": { + "tests": 33, + "seconds": 29.82, + "failures": 0, + "errors": 0, + "skips": 0 + }, + "root_bash": { + "assertions": 259, + "files": 11, + "failures": 0 + }, + "evidence": [ + "evidence/release-receipt-synchronization.json", + "evidence/release-runner-exit-synchronization.json" + ], + "bounds": "Original .3s work and5s positive boundaries unchanged; confirmed child-start/receipt/original runner dead, no live/ambiguous authority relaxation; exactonce retained" + }, + "corrected_public_ci": { + "run_id": 37086958301, + "url": "https://github.com/joshidikshant/devsquad/actions/runs/37086958301", + "head": "f40c618caefd100637c4686660673e159899ecb7", + "status": "completed_failed_python311;python314_passed604", + "duplicate_push_run": 37086954389, + "duplicate_push_status": "cancelled", + "core_runtime_and_plugin_payloads_unchanged": true, + "legacy": "passed", + "optional_mcp": "passed", + "python311": { + "version": "3.11.9", + "job_id": 111099154149, + "tests": 137, + "seconds": 191.313, + "failures": 1, + "errors": 0, + "skips": 0, + "unraisable": [], + "case": "test_killed_repair_supervisor_never_launches_a_duplicate_writer", + "location": "test/core/test_delivery_workflow.py:1001", + "assertion": "ownership_ambiguous != retain_ownership", + "status": "failed_not_accepted" + }, + "diagnosis": "Test kills attempt_runner, not live detached coordinator. Fixed.1s sleep does not establish coordinator's blocked/ownership_ambiguous projection. Running resume correctly imports/blocks and returns ownership_ambiguous; typed retain applies only once blocked. Positive phase/dead-original-runner/same-attempt synchronization repair in isolated test only; independent review underway. No runtime/source edit from this observation.", + "python314": { + "version": "3.14.7", + "job_id": 111099154216, + "tests": 604, + "seconds": 697.362, + "failures": 0, + "errors": 0, + "skips": 2, + "unraisable": [], + "status": "passed_on_prior_f40_not_next_candidate_acceptance" + } + }, + "delivery_phase_correction": { + "source_checkpoint": "893490a2dbd8fa4f48624d1ae0fe33ca106f63fe", + "test_blob": "7723f6ce3b23f715a4a20a38fe5bb59c1ce23bee", + "independent_review": "clean; committed source exactly reviewed, no product edits", + "agent_combined_modules": { + "tests": 52, + "python312_seconds": 82.966, + "python314_seconds": 86.973, + "failures": 0, + "errors": 0, + "skips": 0 + }, + "root_delivery": { + "tests": 19, + "seconds": 48.911, + "failures": 0, + "errors": 0, + "skips": 0 + }, + "unchanged_service_root_gate": "33/29.820s passed on exact same5456e21 test blob", + "core_unchanged": true, + "evidence": "evidence/release-repair-ownership-synchronization.json", + "root_final_bash": { + "assertions": 259, + "files": 11, + "focused_legacy_assertions": 50, + "failures": 0, + "exec_session": 89228, + "timings_seconds": { + "fast": 0.299, + "ignored_TERM": 0.228, + "pre_open_cancel": 0.225, + "TERM_resistant": 3.296, + "slow_ps": 1.323 + } + }, + "reference_json_staged_unstaged_diff": "passed", + "next_public_ci": "pending" + }, + "diagnostic_cleanup_overlap_triage": { + "source": "independent read-only code/output triage; no new probes", + "observation": "Rejected diagnostic RELEASE plus immediate cancellation produced ConflictError recovery cancellation is not active", + "classification": "fail-closed stale-phase completion fence; not evidence of duplicate writers or false cancellation success", + "possible_interleaving": "Concurrent importer can reset cancelling attempt to blocked while cancel_orphan completion requires cancelling/recovery_cleanup; no exact ordering claimed", + "missing_evidence": "No error-time DB/events snapshot; final rows or a lost-intent liveness defect cannot be established", + "disposition": "retain diagnostic concurrency-conflict / possible liveness follow-up, not confirmed unsafe product finding or passing gate; corrected helper awaits import DONE before fixture cleanup" + } +} diff --git a/docs/plans/engineering-team/evidence/r7-council-reconciliation-partial.json b/docs/plans/engineering-team/evidence/r7-council-reconciliation-partial.json new file mode 100644 index 0000000..21cda5c --- /dev/null +++ b/docs/plans/engineering-team/evidence/r7-council-reconciliation-partial.json @@ -0,0 +1,103 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "isolated_council_r6_source_reconciliation_offline_only", + "baseline_commit": "32b94d6", + "preserved_council_slices": ["42979ba", "72ae413", "eee6db7"], + "supported_schema_version": 17, + "automatic_enabled": false, + "native_ready": false, + "source_blob_sha256": { + "plugin/core/src/devsquad/diagnostics.py": "3ba04d754cc17ec6bf7cd32ded1b41d1972afba3d334952c5b87473c2b8bdfb4", + "plugin/core/src/devsquad/service.py": "b1d2ccb334f0c981e01e2a3fc5d8292ee7887f4c53d387ca6eb9c036707ba5c9", + "plugin/core/src/devsquad/store.py": "e94514dd50f398df26b4b4c9261eea9a4c4108d0da8ac40883ee55bb466fa88e", + "plugin/core/src/devsquad/cli.py": "e06d4a22a48a7c65bc53b16bf6e4b10b16d7318827788a8502608ea6a44875e5", + "plugin/core/src/devsquad/task_entry.py": "da29929711ea7eec268205855ead98cac9c368bac98e9296185f360a4574e72b", + "plugin/core/src/devsquad/council_isolation.py": "e5156c6d918b31fcdcc78b5b05dcdfa1a181392c54aba02ccbf92e53448a196e" + }, + "verification": { + "affected_gate": { + "command": "PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_council_contract test_council_entry test_council_runtime test_council_comparison test_council_epoch test_council_install_epoch test_council_isolation test_council_reconciliation test_terminal_ux test_diagnostics test_mcp test_handoff_store test_handoff_service test_task_entry test_router test_validation test_supervisor -q", + "python_version": "3.14.6", + "tests": 168, + "monotonic_seconds_reported_by_unittest": 148.693, + "errors": 0, + "failures": 0, + "skips": 2, + "skip_reason": "optional MCP SDK absent in this interpreter", + "status": "passed" + }, + "actual_official_sdk": { + "command": "PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core ACCEPTED_R5_PYTHON -m unittest test_mcp.OfficialSDKConformanceTest -q", + "python_version": "3.12.14", + "mcp_version": "2.2.0", + "tests": 2, + "monotonic_seconds_reported_by_unittest": 2.964, + "errors": 0, + "failures": 0, + "skips": 0, + "all_nine_tool_names_and_schema_envelopes_preserved": true, + "actual_stdio_disconnect_does_not_cancel_worker": true, + "status": "passed" + }, + "bash": {"command": "bash test/run.sh", "test_files": 11, "assertions": 227, "status": "passed"}, + "generated_reference": {"command": "python3 scripts/generate-core-reference.py --check", "status": "passed"}, + "diff": {"commands": ["git diff --check", "git diff --cached --check"], "status": "passed"} + }, + "retained_failed_gates": [ + { + "scope": "initial_expanded_headless_reconciliation_test", + "errors": 4, + "cause": "test wrongly expected inactive completed claims in the public active-claim view", + "repair": "assert exact authoritative claim rows; corrected live and expired crash tests pass" + }, + { + "scope": "first_final_affected_invocation", + "reported_tests": 169, + "monotonic_seconds_reported_by_unittest": 145.486, + "errors": 1, + "failures": 0, + "skips": 2, + "cause": "nonexistent test_reservation_crash module name produced unittest loader error", + "status": "failed_invocation_not_acceptance", + "repair": "corrected exact 168-test command above passed unchanged source" + } + ], + "shared_authority": { + "store_claim_parameters": ["initial_only", "terminal_decision"], + "marker_schema_version": 1, + "marker_reader": "Store.terminal_finish_decision", + "exact_live_intent_reuses_claim": true, + "exact_expired_intent_gets_fresh_fence": true, + "same_name_app_claim_not_adopted": true, + "guided_expiry_rejection_audited_and_recovered": true, + "submitted_crash_recovery_needs_no_new_claim": true, + "headless_expiry_retains_rejection_and_completes_saved_lead_without_new_worker": true, + "explicit_council_choice_inputs_round_trip_actual_cli": true, + "no_automatic_human_choice": true + }, + "preserved_evidence": [ + "r7-public-fixture-workflow-predeclaration.json", + "r7-public-fixture-workflow-comparison.json", + "r7-temporary-install-epoch-proof.json", + "R7-NATIVE-BOUNDARY-BLOCKER.md" + ], + "acceptance_dependencies": [ + { + "kind": "shared_owned_probe_cleanup", + "isolated_repair_commit": "093ed0f397289bffe9ea249c10796cb436027f00", + "root_integrated_commit": "8242eff68b24538aacc1b40b67b2e88e96b04d52", + "not_imported_into_this_council_checkpoint": true, + "independent_acceptance": "root_owned_pending" + }, + {"kind": "exact_default_deny_native_backend_attestation", "status": "unavailable_fail_closed_before_role_generation"}, + {"kind": "frozen_combined_full_native_and_installed_acceptance", "status": "root_owned_pending"} + ], + "limitations": [ + "No production/root source or ledger changes, network probes, native generations or full suite in this isolated reconciliation", + "Public fixtures establish mechanics only; native quality, escaped defects, rework and allowance remain unknown", + "Matched and held-out comparison remains inconclusive; no automatic use or R3 promotion authority", + "Historical temporary installer evidence names its original exact candidate; the reconciled candidate's installer regression also passed in the final affected command", + "Root resolves authoritative RESUME/backlog wording when importing this source-equivalent checkpoint" + ] +} diff --git a/docs/plans/engineering-team/evidence/r7-final-nongenerating-network-diagnostic.json b/docs/plans/engineering-team/evidence/r7-final-nongenerating-network-diagnostic.json new file mode 100644 index 0000000..be22501 --- /dev/null +++ b/docs/plans/engineering-team/evidence/r7-final-nongenerating-network-diagnostic.json @@ -0,0 +1,153 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "isolated_private_default_deny_nongenerating_diagnostic_only", + "status": "failed_backend_attestation_native_remains_unready", + "source_context_commit": "72ae4130190169096415436952f7c83498bd4d59", + "council_reconciliation_commit": "0e7dd319b56802329187007738970e26b5609b1d", + "production_profile_changed": false, + "production_native_ready": false, + "automatic_enabled": false, + "generating_requests": 0, + "paid_api_fallbacks": 0, + "native": { + "executable": "/Applications/ChatGPT.app/Contents/Resources/codex-cli/CodexCLI.app/Contents/MacOS/codex", + "executable_sha256": "50ac633af64851511f9bbc71032cdae7f1ba20b3234c189687d61ba846c354c5", + "verified_version": "codex-cli 0.159.2", + "sandbox_version_returncode": 0, + "sandbox_version_output_matches_expected": true, + "rpc_argv_suffix": ["--disable", "apps", "--disable", "plugins", "app-server", "--listen", "stdio://"], + "methods_sent": ["initialize", "initialized", "account/rateLimits/read"] + }, + "immutable_shared_helpers": { + "commit": "4f01456189b38252a678938e7197ac2f12a6838e", + "probe_process_git_blob": "f97ee3a1d19af929f096f0821df16b2cfed1182b", + "diagnostics_git_blob": "40c5849324b47b9014aee6f508116263375f155b", + "probe_process_sha256": "26668bfdd9a0e04cb98cf15859170ca0bcb9d8477377841b0e8888ff161d0e1f", + "diagnostics_sha256": "e8ef0b8315cf97c5586ee16ce1e3ec1be223991cccf54d9ee0aaea9aa6c18161", + "archive_mode": "0700", + "only_archived_source_files": ["diagnostics.py", "probe_process.py"], + "immediate_missing_identity_refusal": true, + "exclusive_popen_reaping_and_shared_cleanup": true, + "root_acceptance": "separate_root_owned_gate_not_claimed_by_this_diagnostic" + }, + "profile": { + "baseline_profile_sha256": "f363999d789c6bd75e0d579c5f9f25270a43c74e4e7c3a8a426db3b013048711", + "diagnostic_variant_profile_sha256": "70a1e58436302aec7d3f179fd8a468e0490bd175b95f1d3347bebe2745c393ed", + "actual_private_scratch_profile_sha256": "253c66503f0be900e9c35a089401c2addf317033d60464e88910d42f12305840", + "additions": [ + "(allow mach-lookup (global-name \"com.apple.system.opendirectoryd.libinfo\") (global-name \"com.apple.mDNSResponder\") (global-name \"com.apple.SystemConfiguration.configd\") (global-name \"com.apple.networkd\"))", + "(allow network-outbound (remote udp \"*:53\"))", + "(allow system-info (info-type \"net.link.addr\"))", + "(allow file-read* (literal \"/private/etc/hosts\"))", + "(allow network-outbound (literal \"/private/var/run/mDNSResponder\"))" + ], + "system_info_filter_syntax_authority": "/usr/share/sandbox/mDNSResponder.sb:151", + "cat_and_shell_execution_exceptions": "exact binaries added only to separate file-rule probes, never to native backend profile", + "broad_filesystem_or_mach_or_network_exceptions": false, + "managed_config_or_mcp_exceptions": false + }, + "actual_isolation_checks": { + "own_evidence_positive_control": {"returncode": 0, "output_bytes": 54}, + "peer_proposal": {"returncode": 1, "output_bytes": 0}, + "runtime_ledger": {"returncode": 1, "output_bytes": 0}, + "private_log": {"returncode": 1, "output_bytes": 0}, + "saved_artifact": {"returncode": 1, "output_bytes": 0}, + "own_directory_symlink_to_peer": {"returncode": 1, "output_bytes": 0}, + "inherited_child_runtime_ledger": {"returncode": 1, "output_bytes": 0}, + "unapproved_child_execution": {"returncode": 71, "output_bytes": 0}, + "all_eight_checks_preserved": true + }, + "sandbox_backend_request": { + "initialize_result": true, + "native_elapsed_seconds": 0.049, + "rpc_error_code": -32603, + "error_class": "network_request_failed", + "http_status": null, + "successful_native_response": false, + "owned_identity_captured": true, + "owned_group_cleanup_confirmed": true + }, + "single_unsandboxed_nongenerating_control": { + "reason": "separate_local_boundary_failure_from_backend_service_unavailable", + "initialize_result": true, + "native_elapsed_seconds": 0.49, + "non_error_rpc_result_received": true, + "strict_rate_limits_schema_valid": false, + "valid_window_count": 0, + "error_class": "invalid_or_empty_native_rate_limits", + "successful_native_response": false, + "owned_identity_captured": true, + "owned_group_cleanup_confirmed": true, + "schema_rejection_detail": "raw account/quota payload intentionally not retained; exact offending field is unknown, so no backend-success or service-outage conclusion" + }, + "backend_validator": { + "plain_dictionary_or_cached_catalog_is_insufficient": true, + "required_top_level_keys": ["rateLimits", "rateLimitsByLimitId"], + "allowed_snapshot_keys": ["limitId", "limitName", "primary", "secondary", "credits", "planType"], + "required_snapshot_keys": ["primary", "secondary"], + "window_fields": ["usedPercent", "windowDurationMins", "resetsAt"], + "window_requirements": "finite nonboolean usage 0..100; positive integer duration; integer future reset; at least one valid window", + "credits_requirements": "exact hasCredits/unlimited boolean and bounded nullable balance shape", + "strings_and_map_keys": "bounded nonempty strings; optional nullable fields remain nullable", + "successful_schema_validated_receipts": 0 + }, + "bounded_execution": { + "paired_backend_deadline_seconds": 12, + "sandbox_backend_attempts": 1, + "unsandboxed_backend_controls": 1, + "further_network_attempts": 0, + "python_version": "3.12.14", + "resource_warnings": "error", + "bytecode_writes": "disabled by -B and sys.dont_write_bytecode", + "native_pid_absence_independently_checked_after_cleanup": true + }, + "privacy_and_preservation": { + "only_auth_json_copied": true, + "private_scratch_mode": "0700", + "temporary_auth_copy_mode": "0600", + "temporary_auth_copy_removed": true, + "original_auth_unchanged": true, + "original_diagnostic_script_sha256_unchanged": "e6f2cb9cdc371c595cc0826a99f7614e65a1aa797ff4c8c883d9b1b183a7dc23", + "original_diagnostic_result_sha256_unchanged": "6f334467e4dc0beb3a57bfc395427db24e42fa73587dbddbf642ef5480ab8ae3", + "raw_provider_messages_auth_account_and_quota_values_tracked": false, + "final_private_script_sha256": "1480f38dd83fe45a30bf79ab92da387fa01a34f5bacb4efae9c7d811a4b08e90", + "final_private_redacted_receipt_sha256": "c125c3283e994f7c7dad8322e9e0889008f46a7c7935d3c0164e7891ce40d8c1" + }, + "retained_early_aborts": [ + { + "helper_commit": "093ed0f397289bffe9ea249c10796cb436027f00", + "stage": "sandbox_version_gate_before_file_checks_or_backend_rpc", + "returncode": null, + "cause": "exact return code not recorded; do not attribute it to the known EOF helper defect without evidence", + "private_redacted_receipt_sha256": "c8ef5f89b0fcfdd6a7831b01044c84a339bf030bf8f83cde1365a3130e83d2a9" + }, + { + "helper_commit": "4f01456189b38252a678938e7197ac2f12a6838e", + "stage": "sandbox_version_gate_before_file_checks_or_backend_rpc", + "returncode": 65, + "cause": "bare net.link.addr token in system-info rule was an unbound variable; exact no-auth/no-native compiler check proved syntax defect, corrected with OS-defined info-type filter", + "private_redacted_receipt_sha256": "fa39261aba4fbe4f56112cd39f991f7cceb1801558405190f5bffc1e8d0f27c9" + }, + { + "stage": "first_no_auth_no_native_compiler_check_invocation", + "cause": "command helper correctly refused replacing the frozen executable before profile construction; corrected invocation retained frozen profile then used a denied true executable solely for parser diagnostics", + "backend_requests": 0 + } + ], + "scoped_previous_pid_log_query": { + "pid": 63678, + "info": true, + "debug": true, + "predicate_scope": "only exact codex PID sandbox deny messages, excluding the log query process", + "matching_denials": 0, + "missing_facility_inference": "none; no specific additional OS facility has been demonstrated necessary or sufficient" + }, + "next_action": "Stop network attempts. Root may investigate the exact local boundary and validator contract separately; Council remains fail-closed before native role generation. This diagnostic is not production readiness, a minimum-boundary proof, native quality evaluation, or release acceptance.", + "verification": { + "bash": {"command": "bash test/run.sh", "status": "passed", "test_files": 11, "assertions": 227}, + "generated_reference": {"command": "python3 -B scripts/generate-core-reference.py --check", "status": "passed"}, + "diff": {"command": "git diff --check", "status": "passed"}, + "json_parse": {"command": "python3 -B -m json.tool docs/plans/engineering-team/evidence/r7-final-nongenerating-network-diagnostic.json", "status": "passed"} + } +} diff --git a/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-comparison.json b/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-comparison.json new file mode 100644 index 0000000..269f4de --- /dev/null +++ b/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-comparison.json @@ -0,0 +1,204 @@ +{ + "automatic_enabled": false, + "cases": [ + { + "arms": { + "control": { + "accepted_quality": null, + "actual_prompt_sha256": [ + "6841c7ab565bb86c35c4c40c47ca7831c2119b8122e788a78826c41eeed12982" + ], + "brief_sha256": null, + "candidate": { + "base_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc", + "sha256": "ef2f034431385621374f5e97428107e742c8f583b6bbcb0ba4174452e9ea0aaa", + "target_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc" + }, + "escaped_defects": null, + "execution_elapsed_ms": 3640, + "host_usage": null, + "identity_scope": "all_fixture", + "input_contract_sha256": "3e4ee466f5141b2cd2bcea35e14adf1354cf91448cd3743855b1c2349d756d72", + "lead_disposition": "accept", + "native_model_requests": null, + "native_quota": null, + "native_usage": [ + { + "input_tokens": null, + "output_tokens": null, + "source": "unavailable", + "total_tokens": null + } + ], + "quality_missingness": "fixture mechanics and host acceptance are not independent native quality measurements", + "receipt": { + "artifact_id": "a6a606ef-f798-4684-936b-a6dded2cbadf", + "sha256": "09275379766dc1ab52e26c69a33028d9d4f4a3e0311154e82fd85a2c56469595" + }, + "rework": null, + "run_id": "76b924c1-583a-4fe0-bb3f-f27088e1ea3e", + "runtime_package_sha256": "b456841b6c89fbbfbc0ea488c1a736d3df9056c3551857dca08432b97710cbd6", + "state": "succeeded", + "versions": { + "policy_sha256": "076ee3ae42e47cdfa86eee7a27a4ade567215ba9f6d1a978a2f3585332f5a337", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "workflows.py", + "prompt_module_sha256": "6e0be346a78a50927913762fe78e989a3b73f56f84808fdbda3886de7edf00ef", + "task_sha256": "c84e2441b70f0fed5a3c86ade31a2a5f811fb2a0ec2f5ec79c6fbad48220f6dd" + }, + "worker_invocations": 1 + }, + "council": { + "accepted_quality": null, + "actual_prompt_sha256": [ + "8b2359053fd3dba49b621c09810cb7dc56889dbcfca0faf0a7f03529e70eb6fc", + "5afd1556f9e7e539759c8cf080042e4221d5556ea0930e3d0b36c9d92bc297df", + "aae30337bfb1e911e000cb354a9968a68483f439181e16a1a264cdbcfae7a9c3" + ], + "brief_sha256": "01a67fd5ba56856a1e58ac28494af09d74c41251c3872f6065d51922ffe7dc1d", + "candidate": { + "base_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc", + "sha256": "ef2f034431385621374f5e97428107e742c8f583b6bbcb0ba4174452e9ea0aaa", + "target_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc" + }, + "escaped_defects": null, + "execution_elapsed_ms": 6087, + "host_usage": null, + "identity_scope": "all_fixture", + "input_contract_sha256": "3e4ee466f5141b2cd2bcea35e14adf1354cf91448cd3743855b1c2349d756d72", + "lead_disposition": "accept", + "native_model_requests": null, + "native_quota": null, + "native_usage": [ + null, + null, + null + ], + "quality_missingness": "fixture mechanics and host acceptance are not independent native quality measurements", + "receipt": { + "artifact_id": "9192542d-a6bf-4886-9360-bf05ad76d2f4", + "sha256": "ba7fbd67466fce66795a148cf9724fa40db06702ad858bdb84c44162390be47f" + }, + "rework": null, + "run_id": "7f01a92a-5458-44f6-875d-124aab6bda54", + "runtime_package_sha256": "b456841b6c89fbbfbc0ea488c1a736d3df9056c3551857dca08432b97710cbd6", + "state": "succeeded", + "versions": { + "policy_sha256": "4fac321b5ec269b8c59f0113771c510fd4080dcc8a102e693ba128351388a46a", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "council.py", + "prompt_module_sha256": "9646b5a6455762c766a2d54b72618ad15ce0cd662309ce61c1ae045e9be8c81f", + "task_sha256": "2d82a04e0904a7c5601f2171a90e2e9976ead07c0810a0d8880e43e9cfc0d1ce" + }, + "worker_invocations": 3 + } + }, + "id": "retry-key", + "split": "matched" + }, + { + "arms": { + "control": { + "accepted_quality": null, + "actual_prompt_sha256": [ + "b5d07a1bff9c3b3c0e28c364b959d454b1fac07463754a1d12439e593807f61f" + ], + "brief_sha256": null, + "candidate": { + "base_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc", + "sha256": "ef2f034431385621374f5e97428107e742c8f583b6bbcb0ba4174452e9ea0aaa", + "target_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc" + }, + "escaped_defects": null, + "execution_elapsed_ms": 3531, + "host_usage": null, + "identity_scope": "all_fixture", + "input_contract_sha256": "c6bc137c2c1288edda333f7a8ba262e7774723efc60c4b0989ec3ef33a10e3ae", + "lead_disposition": "accept", + "native_model_requests": null, + "native_quota": null, + "native_usage": [ + { + "input_tokens": null, + "output_tokens": null, + "source": "unavailable", + "total_tokens": null + } + ], + "quality_missingness": "fixture mechanics and host acceptance are not independent native quality measurements", + "receipt": { + "artifact_id": "d3cb199a-f927-4e11-a9a2-915ba5c4eda4", + "sha256": "57901f4791141741a75d9bf3b6079f49f44ed782cd828eb8f15466767d08038a" + }, + "rework": null, + "run_id": "e5342195-90e6-47d7-b055-529e4013ef41", + "runtime_package_sha256": "b456841b6c89fbbfbc0ea488c1a736d3df9056c3551857dca08432b97710cbd6", + "state": "succeeded", + "versions": { + "policy_sha256": "076ee3ae42e47cdfa86eee7a27a4ade567215ba9f6d1a978a2f3585332f5a337", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "workflows.py", + "prompt_module_sha256": "6e0be346a78a50927913762fe78e989a3b73f56f84808fdbda3886de7edf00ef", + "task_sha256": "fd5adc9eb398a381be9cb14900b3606a95c5f2962176aa541d3b873db38175a2" + }, + "worker_invocations": 1 + }, + "council": { + "accepted_quality": null, + "actual_prompt_sha256": [ + "1b5b9192d21b912bd78ad05e0ee0c368df2edb2c4926f561c485c2a8cd9e5259", + "add9d2c081540e13c95a0db62c41e4f016060c509ea6b06607cada2b1aea8e01", + "ebb71d34afc939968cdfd51948e4bfb0b3bbe482834e787fd6066ce0a895b354" + ], + "brief_sha256": "632758d5386d411ebe7021ce1d4cdf78efa1fde57a69f12737b63b781630aacd", + "candidate": { + "base_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc", + "sha256": "ef2f034431385621374f5e97428107e742c8f583b6bbcb0ba4174452e9ea0aaa", + "target_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc" + }, + "escaped_defects": null, + "execution_elapsed_ms": 4528, + "host_usage": null, + "identity_scope": "all_fixture", + "input_contract_sha256": "c6bc137c2c1288edda333f7a8ba262e7774723efc60c4b0989ec3ef33a10e3ae", + "lead_disposition": "accept", + "native_model_requests": null, + "native_quota": null, + "native_usage": [ + null, + null, + null + ], + "quality_missingness": "fixture mechanics and host acceptance are not independent native quality measurements", + "receipt": { + "artifact_id": "b0978f78-a739-4605-aa52-d24e5ca52eee", + "sha256": "27b43840c0fcdb10777f0bd970b7d9015cecb07fa8fd07331e36178816a6d463" + }, + "rework": null, + "run_id": "fd52dd7d-5e4d-45c5-9b0c-e0977c9113de", + "runtime_package_sha256": "b456841b6c89fbbfbc0ea488c1a736d3df9056c3551857dca08432b97710cbd6", + "state": "succeeded", + "versions": { + "policy_sha256": "4fac321b5ec269b8c59f0113771c510fd4080dcc8a102e693ba128351388a46a", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "council.py", + "prompt_module_sha256": "9646b5a6455762c766a2d54b72618ad15ce0cd662309ce61c1ae045e9be8c81f", + "task_sha256": "6a1527d0de63509dced266a249dfb3472fccf851de82bd51e71cc44fd89bbefd" + }, + "worker_invocations": 3 + } + }, + "id": "retry-expiry", + "split": "heldout" + } + ], + "conclusion": "inconclusive", + "limitations": [ + "Controlled process mechanics only", + "No independent native accepted-quality/escaped-defect/rework observations", + "Workflow prompts and invocation count differ; this is not R3 single-binding eligibility" + ], + "predeclaration_sha256": "5f96d3946d2bd32a0aed16298533998d308da19e68e600daf300806ee2b7a43f", + "quality_benefit_supported": false, + "schema_version": 1 +} diff --git a/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-predeclaration.json b/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-predeclaration.json new file mode 100644 index 0000000..2b94f96 --- /dev/null +++ b/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-predeclaration.json @@ -0,0 +1,52 @@ +{ + "automatic_enabled": false, + "cases": [ + { + "control": { + "policy_sha256": "076ee3ae42e47cdfa86eee7a27a4ade567215ba9f6d1a978a2f3585332f5a337", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "workflows.py", + "prompt_module_sha256": "6e0be346a78a50927913762fe78e989a3b73f56f84808fdbda3886de7edf00ef", + "task_sha256": "c84e2441b70f0fed5a3c86ade31a2a5f811fb2a0ec2f5ec79c6fbad48220f6dd" + }, + "council": { + "policy_sha256": "4fac321b5ec269b8c59f0113771c510fd4080dcc8a102e693ba128351388a46a", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "council.py", + "prompt_module_sha256": "9646b5a6455762c766a2d54b72618ad15ce0cd662309ce61c1ae045e9be8c81f", + "task_sha256": "2d82a04e0904a7c5601f2171a90e2e9976ead07c0810a0d8880e43e9cfc0d1ce" + }, + "id": "retry-key", + "input_contract_sha256": "3e4ee466f5141b2cd2bcea35e14adf1354cf91448cd3743855b1c2349d756d72", + "split": "matched" + }, + { + "control": { + "policy_sha256": "076ee3ae42e47cdfa86eee7a27a4ade567215ba9f6d1a978a2f3585332f5a337", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "workflows.py", + "prompt_module_sha256": "6e0be346a78a50927913762fe78e989a3b73f56f84808fdbda3886de7edf00ef", + "task_sha256": "fd5adc9eb398a381be9cb14900b3606a95c5f2962176aa541d3b873db38175a2" + }, + "council": { + "policy_sha256": "4fac321b5ec269b8c59f0113771c510fd4080dcc8a102e693ba128351388a46a", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "council.py", + "prompt_module_sha256": "9646b5a6455762c766a2d54b72618ad15ce0cd662309ce61c1ae045e9be8c81f", + "task_sha256": "6a1527d0de63509dced266a249dfb3472fccf851de82bd51e71cc44fd89bbefd" + }, + "id": "retry-expiry", + "input_contract_sha256": "c6bc137c2c1288edda333f7a8ba262e7774723efc60c4b0989ec3ef33a10e3ae", + "split": "heldout" + } + ], + "created_at": "2026-10-02T23:01:18.274582+00:00", + "decision_rule": "inconclusive_without_independent_native_quality_evidence", + "quality_gate": { + "accepted_quality": "unmeasured", + "escaped_defects": "unmeasured", + "rework": "unmeasured" + }, + "schema_version": 1, + "scope": "public_fixture_mechanics" +} diff --git a/docs/plans/engineering-team/evidence/r7-temporary-install-epoch-proof.json b/docs/plans/engineering-team/evidence/r7-temporary-install-epoch-proof.json new file mode 100644 index 0000000..49b8acd --- /dev/null +++ b/docs/plans/engineering-team/evidence/r7-temporary-install-epoch-proof.json @@ -0,0 +1,30 @@ +{ + "schema_version": 1, + "scope": "actual_temporary_core_only_installer_and_public_fixture_processes", + "candidate_commit": "72ae4130190169096415436952f7c83498bd4d59", + "candidate_source_sha256": "4c6f0d92e6576abcf2f64044a0321cf9cd8c8e0c808d7923e0f9b7e97b8b4cab", + "old_commit": "bf3de0867484552d354e6e8b6ba835f31a93aeb3", + "old_source_sha256": "90f1e87fb9acf7756dad46780672de9b84fe75e3631322857a310f089280345e", + "old_package_sha256": "e03bf3a2e362b3aff6d0a5dd00bd837f15afb216d4c58ac1d9c776c4d8f09c67", + "old_schema": 16, + "new_schema": 17, + "active_deferral": true, + "recoverable_deferral": true, + "selector_and_launcher_preserved": true, + "old_active_cancelled": true, + "old_recoverable_reconciled": true, + "long_lived_old_claim_rejected": true, + "fresh_old_service_claim_rejected": true, + "council_claim_unchanged": true, + "owned_process_groups_gone": true, + "temporary_install_removed": true, + "mcp_installed_in_temporary_release": false, + "native_generation": false, + "regression_test": "test/core/test_council_install_epoch.py", + "limitations": [ + "This is a temporary core-only installation, not a production upgrade or MCP SDK lifecycle gate", + "The exact archived core matches accepted immutable R5 source and code-package digests; no version-rewritten Store was used", + "Public fixture mechanics do not prove native quality or network readiness", + "Root must reconcile R6 shared code and repeat integrated acceptance before production activation" + ] +} diff --git a/docs/plans/engineering-team/evidence/release-receipt-synchronization.json b/docs/plans/engineering-team/evidence/release-receipt-synchronization.json new file mode 100644 index 0000000..48ad16a --- /dev/null +++ b/docs/plans/engineering-team/evidence/release-receipt-synchronization.json @@ -0,0 +1,64 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "Test-only coordinator-crash receipt synchronization; no recovery/product changes or whole-suite rerun.", + "baseline_revision": "5a576d39469c198a3cc93474960886b875363419", + "branch": "codex/release-receipt-sync", + "preserved_branch": "codex/release-watchdog", + "test_blob": "e19495c811c5e52e3d66e5588d18f325e306a8ad", + "test_diff_sha256": "ef3c84d775695a832fafc6681e494262ebdfdd339be36170c4b6d44ee451facb", + "public_CI_failure": { + "run_id": "37084805784", + "head_sha": "5a576d39469c198a3cc93474960886b875363419", + "reported_by": "root", + "python": "3.14.7", + "tests_before_failfast_stop": 486, + "seconds": 679.063, + "test": "test_service.ServiceTest.test_coordinator_crash_imports_runner_receipt_once", + "assertion": "first disposition was live, expected succeeded/already_finalized at original test_service.py:1108", + "other_jobs": "Bash and optional MCP passed; Python3.11 completed604 tests/1013.419s with2 optional SDK skips, zero failures/errors/unraisable. This agent did not cancel any job." + }, + "cause": "The test assumed a .6s sleep after killing the coordinator implied durable runner completion. Actual gated Python startups, child work, drain/fsync and receipt publication need not fit that delay. Product import_durable correctly returns live before receipt parsing. Also, running is committed before the runner gate is released; child_record is the positive started-child phase boundary.", + "controlled_red": { + "method": "Run the original exact test with only its existing internal fake child's .3s work delay replaced by1.2s through a diagnostic Service.start wrapper. Capture the real resume response, original runner identity, receipt presence and attempt count. Before fixture teardown, positively wait for owned runner death and receipt publication.", + "python312": {"version": "3.12.14", "tests": 1, "failures": 1, "seconds": 1.85}, + "python314": {"version": "3.14.6", "tests": 1, "failures": 1, "seconds": 2.023}, + "same_observation_both_versions": {"disposition": "live", "runner_identity": "live", "receipt_present": false, "attempt_count": 1}, + "cleanup_both_versions": {"runner_identity": "dead", "receipt_present": true}, + "uncontrolled_baseline": "The original test also passed in isolation on3.12.14/.991s and3.14.6/1.052s, consistent with a scheduling-sensitive assumption rather than a deterministic import defect." + }, + "repair": { + "changed_code": "Only test/core/test_service.py:32 inserted/2 removed lines in the existing method.", + "started_phase": "Wait for the original attempt's durable child_record before killing the coordinator, so this is not before-gate/unstarted recovery.", + "import_phase": "Replace .6s sleep with a bounded wait for the same original receipt path and inspect_process(pid,pgid,start_id) == dead; assert both before the first import.", + "bounds": "Original .3s fixture work is unchanged. The5s positive wait boundary matches adjacent receipt/import tests. No product timeout or timing assertion was widened.", + "live_fence": "Existing test_dead_supervisor_with_live_child_never_relaunches is unchanged and included in both complete service-module gates. No new short-delay live assertion or recovery requeue/takeover was introduced.", + "exact_once": "Original successful disposition and repeated terminal resume ConflictError retained. Require one attempt, unchanged id/token/pid/pgid/start, finished status, exactly3 artifacts and exactly one attempt.output/run.succeeded event." + }, + "controlled_green": { + "method": "Repeat the identical1.2s delayed-child diagnostic against the repaired test.", + "python312": {"version": "3.12.14", "tests": 1, "failures": 0, "seconds": 1.72}, + "python314": {"version": "3.14.6", "tests": 1, "failures": 0, "seconds": 1.867}, + "same_observation_both_versions": {"disposition": "succeeded", "runner_identity": "dead", "receipt_present": true, "attempt_count": 1} + }, + "affected_service_gates": { + "environment": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core", + "command": "INTERPRETER -m unittest test_service -q", + "python312": {"interpreter": "/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3", "version": "3.12.14", "tests": 33, "seconds": 36.966, "failures": 0, "errors": 0, "skips": 0}, + "python314": {"interpreter": "/opt/homebrew/bin/python3.14", "version": "3.14.6", "tests": 33, "seconds": 39.543, "failures": 0, "errors": 0, "skips": 0}, + "coverage": "Complete service module, including live refusal, before-gate crash, orphan cancellation, importer-finalization crash, coordinator receipt recovery and concurrent exact-once receipt import." + }, + "independent_review": { + "reviewer": "r6_readiness", + "scope": "Frozen changed test and read-only recovery semantics; no edits/full/provider/install actions.", + "result": "clean; scheduling defect, live-process and exact-once fences preserved", + "focused_tests": ["test_coordinator_crash_imports_runner_receipt_once", "test_importer_crash_after_artifact_finalize_has_no_false_completion", "test_two_processes_import_one_runner_receipt_atomically"], + "python312": {"version": "3.12.14", "tests": 3, "seconds": 3.415, "failures": 0}, + "python314": {"version": "3.14.6", "tests": 3, "seconds": 3.161, "failures": 0}, + "before_after_hashes_unchanged": true + }, + "core_tree_unchanged": "62eea7fa31153ef732fd5e9dfd97951d95b367d8", + "other_gates": "Generated reference and diff checks pass; required Bash259 gate is run before the checkpoint.", + "limitations": "Local3.14 is3.14.6, not CI3.14.7. No local full604 suite, provider/native request, install, publication, global settings or root source/docs change. Final exact-head public CI remains required and root-owned.", + "next_action": "Root integrates the narrow checkpoint, records final gate/checkpoint identifiers in authoritative release evidence, verifies its bounded integration gate and pushes the next exact public CI candidate." +} diff --git a/docs/plans/engineering-team/evidence/release-repair-ownership-synchronization.json b/docs/plans/engineering-team/evidence/release-repair-ownership-synchronization.json new file mode 100644 index 0000000..f9d2746 --- /dev/null +++ b/docs/plans/engineering-team/evidence/release-repair-ownership-synchronization.json @@ -0,0 +1,88 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "Test-only positive delivery runner-death and blocked-ownership phase synchronization; no core edits or whole-suite rerun.", + "baseline_revision": "f40c618caefd100637c4686660673e159899ecb7", + "branch": "codex/release-repair-ownership-sync", + "preserved_branch": "codex/release-runner-exit-sync", + "test_blob": "7723f6ce3b23f715a4a20a38fe5bb59c1ce23bee", + "test_diff_sha256": "a3c19e9398adab6ac27526cf6d4e27a0d63fbf371cd576619c230259bf28eef5", + "public_CI_failure": { + "reported_by": "root", + "run_id": "37086958301", + "head_sha": "f40c618caefd100637c4686660673e159899ecb7", + "python": "3.11.9", + "tests_before_failfast_stop": 137, + "seconds": 191.313, + "failures": 1, + "errors": 0, + "skips": 0, + "unraisable": 0, + "test": "test_delivery_workflow.DeliveryWorkspaceTest.test_killed_repair_supervisor_never_launches_a_duplicate_writer", + "assertion": "ownership_ambiguous != retain_ownership at original line1001", + "other_jobs_at_assignment": "Bash and optional MCP passed; Python3.14 remained active. This agent did not cancel or mutate any CI job.", + "later_python314_outcome_reported_by_root": {"version": "3.14.7", "job_id": "111099154216", "tests": 604, "seconds": 697.362, "failures": 0, "errors": 0, "skips": 2, "unraisable": 0}, + "workflow_outcome": "failed because Python3.11.9 failed; Python3.14.7 success is not exact-next-head workflow acceptance" + }, + "cause": "Despite the historical test name, attempt.pid is the durable attempt_runner, NOT the detached coordinator or writer. After runner SIGKILL the surviving coordinator asynchronously reaps and imports, publishing blocked/recovery_required plus the original attempt's ownership_ambiguous status while the child may remain live. The test assumed .1s implied that phase was persisted. Service.resume called while still running first performs import_durable and returns ownership_ambiguous/launchedfalse; only the already-blocked branch validates and returns the typed retain_ownership disposition. The safety response is correct; the test's ordering assumption is not.", + "controlled_diagnostic": { + "helper_sha256": "498cebc1019a62551ff52f84318a88ca4fbcfbee14154c0e326cfe5d78431262", + "scope": "Private Popen scheduling wrapper only for frozen devsquad.detached coordinator argv. Exact frozen PYTHONPATH/package digest and worker commands remain unchanged. The spawned coordinator's in-memory import method pauses only on the original third implementer after wait_durable has reaped the killed runner, then calls its unchanged original importer.", + "synchronization": "Atomically publish exact ready JSON via temporary+replace. Parent establishes ready identity and actual original runner dead. Old test proceeds to one typed request while run is definitely running. Repaired pure status waiter first observes actual running, releases the private gate, then observes genuine coordinator blocked publication before its one typed request. Cleanup releases the gate, waits import DONE, and cancels the fixture before temporary removal. No numeric coordinator signals or classifier/state forcing.", + "original_source": "--original loads only the exact original method AST from git show f40c618caefd100637c4686660673e159899ecb7:test/core/test_delivery_workflow.py without modifying checkout files.", + "original_red": { + "python312": {"version": "3.12.14", "tests": 1, "seconds": 3.868, "failures": 1, "errors": 0}, + "python314": {"version": "3.14.6", "tests": 1, "seconds": 4.334, "failures": 1, "errors": 0}, + "observation_both": {"before_state": "running", "before_phase": null, "before_attempt_status": "running", "runner": "dead", "typed_calls": 1, "disposition": "ownership_ambiguous", "launched": false}, + "assertion": "same ownership_ambiguous != retain_ownership at original1001" + }, + "corrected_green": { + "python312": {"version": "3.12.14", "tests": 1, "seconds": 3.665, "failures": 0, "errors": 0}, + "python314": {"version": "3.14.6", "tests": 1, "seconds": 3.814, "failures": 0, "errors": 0}, + "observation_both": {"before_state": "blocked", "before_phase": "recovery_required", "before_attempt_status": "ownership_ambiguous", "runner": "dead", "typed_calls": 1, "disposition": "retain_ownership", "launched": false} + } + }, + "rejected_diagnostics_retained": [ + { + "method": "Stop coordinator before killing runner, then wait runner dead.", + "outcome": "Invalid phase seam: stopped parent cannot reap its runner zombie; inspector never reached confirmed dead. Failed1 fixture each on3.12.14/8.295s and3.14.6/8.401s, with no typed call. Cleanup resumed the owned coordinator and cancelled. Not claimed as matching CI red." + }, + { + "method": "First reaped-coordinator gate without waiting import DONE before cleanup cancellation.", + "outcome": "Matching assertion reproduced, but3.12 teardown concurrently resumed the coordinator importer and cancelled; cleanup raised ConflictError recovery cancellation is not active (1 expected failure+1 cleanup error/3.461s).3.14 had1 expected failure/no errors/3.569s. This unsuitable harness is not accepted as the clean red gate; diagnostic cleanup now positively waits DONE. The observed overlap is retained for separate root triage, without a core edit or broader product conclusion." + }, + { + "method": "Ready JSON initially used direct write_text.", + "outcome": "Independent review identified potential file-visible-before-full-JSON race; corrected to atomic temporary+replace before final repeated red/green. No product/test source change from this correction." + } + ], + "repair": { + "changed_file": "test/core/test_delivery_workflow.py only", + "phase_wait": "Replace fixed.1s sleep with bounded5s pure status/current-attempt/strong-runner-identity polling. Require dead original runner, blocked/recovery_required, same id/token/pid/pgid/start identity and ownership_ambiguous before the sole typed retain request.", + "bounds": "Original3s repair fixture and10s child-start wait are unchanged. No retry-until-success, alternative success disposition, provider delay widening or product timeout change.", + "preserved_and_strengthened": "No exit receipt; launchedfalse and exact retain_ownership; exactly implementer/reviewer/implementer attempts, original final id/token and3 worker invocations; run remains blocked until cancel; cancellation must be exactly cancelled; source checkout/index/HEAD/remotes remain unchanged. Comment distinguishes runner from coordinator without renaming the historical CI case." + }, + "affected_module_gates": { + "environment": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core", + "command": "INTERPRETER -B -m unittest test_delivery_workflow test_service -q", + "modules": {"delivery_workflow": 19, "service": 33}, + "python312": {"version": "3.12.14", "tests": 52, "seconds": 82.966, "failures": 0, "errors": 0, "skips": 0}, + "python314": {"version": "3.14.6", "tests": 52, "seconds": 86.973, "failures": 0, "errors": 0, "skips": 0} + }, + "independent_review": { + "reviewer": "r6_readiness", + "source_review": "clean; pure phase poll and all same-attempt/no-new-writer/cancel/source assertions preserved or strengthened", + "real_cases": ["test_killed_repair_supervisor_never_launches_a_duplicate_writer", "test_live_implementer_cannot_be_resumed_into_a_second_writer", "test_cancelled_repair_retains_prior_candidate_attempts_and_disposition"], + "python312": {"version": "3.12.14", "tests": 3, "seconds": 10.469, "failures": 0, "errors": 0}, + "python314": {"version": "3.14.6", "tests": 3, "seconds": 11.505, "failures": 0, "errors": 0}, + "final_helper_review_and_gate": "Current corrected atomic-ready/DONE-ordered helper red/green independently confirmed; frozen product/test diff remains clean. No wider product conclusion inferred from rejected diagnostic cleanup overlap." + }, + "core_tree_unchanged": "62eea7fa31153ef732fd5e9dfd97951d95b367d8", + "other_gates": { + "bash": {"command": "PYTHONDONTWRITEBYTECODE=1 /bin/bash test/run.sh", "test_files": 11, "assertions": 259, "failures": 0, "legacy_assertions": 50}, + "watchdog_seconds": {"fast": 0.292, "ignored_TERM": 0.216, "early_cancel": 0.223, "tree": 2, "resistant": 3.257, "slow_probe": 1.306}, + "generated_reference": "current", + "JSON_and_diff": "pass" + }, + "limitations": "Local proof uses3.12.14 and3.14.6, not public3.11.9/3.14.7. No full604 suite, provider/native request, installation, publication, global settings or root edits. Root owns integration, separate diagnostic-overlap triage and the next exact-head public CI gate." +} diff --git a/docs/plans/engineering-team/evidence/release-runner-exit-synchronization.json b/docs/plans/engineering-team/evidence/release-runner-exit-synchronization.json new file mode 100644 index 0000000..aeae57d --- /dev/null +++ b/docs/plans/engineering-team/evidence/release-runner-exit-synchronization.json @@ -0,0 +1,68 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "Test-only positive runner-death synchronization in the before-gate crash case; no core or recovery changes.", + "baseline_revision": "edfcf65b92eeee97323b20db6d252b18e2fbc99a", + "branch": "codex/release-runner-exit-sync", + "preserved_branch": "codex/release-receipt-sync", + "test_blob": "5456e21d10f7e65b06445a266ac0b177d8bd2066", + "test_diff_sha256": "e20bc04a593e35a9d5612591c19ab54081f974e3ba57256c3ba927007de0f7b0", + "root_integration_failure": { + "reported_by": "root", + "tests": 33, + "seconds": 32.435, + "failures": 1, + "test": "test_service.ServiceTest.test_crash_after_runner_identity_before_gate_requeues_without_execution", + "assertion": "ambiguous != dead at original test_service.py:598", + "context": "Root ran the complete service module while Bash was also active after integrating edfc. Bash259 passed. This agent did not repeat or edit that root gate." + }, + "cause": "The existing waiter continued only while inspect_process returned live. A dying runner can legitimately be ambiguous when its start identity was observed but getpgid then loses the PID, or the group inventory is inconclusive. Ambiguous is not confirmed dead. The test exited its waiter on that intermediate classification, then asserted dead. The product classifier correctly fails closed and is unchanged.", + "controlled_diagnostic": { + "scope": "Only the test_service.inspect_process alias is wrapped; devsquad.supervisor.inspect_process and platform calls are untouched.", + "method": "For the same captured pid/pgid/start identity, delegate real inspection first. Preserve actual live classifications. At the first genuine dead boundary, return ambiguous for two successive observations, then delegate genuine classifications. Two ambiguous observations exercise both the original loop condition and its final assertion without permitting early recovery.", + "trace_both_versions": ["live", "live", "live", "live", "ambiguous", "ambiguous"], + "identity_binding": "Every observation asserts the original pid/pgid/start tuple is unchanged. The injected observations each follow an actual dead classification, so the owned runner is already gone before any failed-test fixture teardown.", + "original_red": { + "python312": {"version": "3.12.14", "tests": 1, "failures": 1, "errors": 0, "seconds": 0.445}, + "python314": {"version": "3.14.6", "tests": 1, "failures": 1, "errors": 0, "seconds": 0.458}, + "assertion": "ambiguous != dead at598 on both versions" + }, + "corrected_green": { + "python312": {"version": "3.12.14", "tests": 1, "failures": 0, "errors": 0, "seconds": 0.856}, + "python314": {"version": "3.14.6", "tests": 1, "failures": 0, "errors": 0, "seconds": 0.946}, + "final_trace": ["dead", "dead"] + } + }, + "repair": { + "changed_code": "Only3 inserted/1 removed lines in the existing method: change while == live to while != dead and add two explanatory comments.", + "bounds": "The original5s monotonic deadline and .02s polling interval are unchanged. No sleep-only widening or product timeout change.", + "positive_boundary": "Require an actual dead classification of the original attempt identity before continuing; the final dead assertion is unchanged. Persistent ambiguous/live ownership still fails this test instead of being accepted as death.", + "preserved_assertions": "Coordinator exits24 before releasing the runner gate; original child_record and execution marker remain absent; resume launches one continuation, succeeds, never executes the blocked command, records exactly recovery_required then finished attempts and exactly one run.unstarted_attempt_recovered event. Existing live-runner refusal and receipt exact-once tests remain unchanged." + }, + "affected_service_gates": { + "environment": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core", + "command": "INTERPRETER -B -m unittest test_service -q", + "python312": {"version": "3.12.14", "tests": 33, "seconds": 32.621, "failures": 0, "errors": 0, "skips": 0}, + "python314": {"version": "3.14.6", "tests": 33, "seconds": 35.470, "failures": 0, "errors": 0, "skips": 0}, + "coverage": "Complete service module, including unstarted recovery, live-owner refusal, receipt import, orphan cancellation and concurrent exact-once import." + }, + "independent_review": { + "reviewer": "r6_readiness", + "scope": "Frozen test diff and classifier semantics; read-only offline focused checks, no core edits.", + "source_review": "clean; same bound and all final death, gate absence and exact requeue assertions retained", + "focused_tests": ["test_crash_after_runner_identity_before_gate_requeues_without_execution", "test_dead_supervisor_with_live_child_never_relaunches", "test_coordinator_crash_imports_runner_receipt_once"], + "python312": {"version": "3.12.14", "tests": 3, "seconds": 2.816, "failures": 0, "errors": 0}, + "python314": {"version": "3.14.6", "tests": 3, "seconds": 3.515, "failures": 0, "errors": 0}, + "independent_controlled_green": {"python312_seconds": 0.988, "python314_seconds": 0.947, "tests_each": 1, "failures": 0}, + "result": "clean; real ownership/refusal/receipt cases and exact alias-only diagnostic pass, with exactly two transient ambiguity observations followed by two genuine dead observations", + "before_after_hashes_unchanged": true + }, + "core_tree_unchanged": "62eea7fa31153ef732fd5e9dfd97951d95b367d8", + "other_gates": { + "bash": {"command": "PYTHONDONTWRITEBYTECODE=1 /bin/bash test/run.sh", "test_files": 11, "assertions": 259, "failures": 0, "legacy_assertions": 50}, + "watchdog_seconds": {"fast": 0.276, "ignored_TERM": 0.214, "early_cancel": 0.213, "tree": 1, "resistant": 3.258, "slow_probe": 1.26}, + "generated_reference": "current", + "JSON_and_diff": "pass" + }, + "limitations": "Local3.14 is3.14.6, not public CI3.14.7. No full604 suite, provider/native request, installation, publication, global settings or root source/docs action. Root owns integration and the next exact-head public CI candidate." +} diff --git a/docs/plans/engineering-team/examples/README.md b/docs/plans/engineering-team/examples/README.md new file mode 100644 index 0000000..139e4ab --- /dev/null +++ b/docs/plans/engineering-team/examples/README.md @@ -0,0 +1,24 @@ +# Design fixtures + +These task files show the implemented strict v1 shape. They are not directly +runnable as checked in: their repository paths, refs, routing files and test +commands deliberately name disposable fixtures. M3 and M5 exercise equivalent +tasks end to end against temporary Git repositories. + +- [branch-review.json](branch-review.json): a host lead receives the review packet; a failing report-only check becomes a finding. +- [issue-delivery.json](issue-delivery.json): a headless lead resolves a bounded implementation/review workflow; required tests must pass. + +The fixture test harness must install a `devsquad/profiles.json` and +`devsquad/policy.json` inside its temporary repository, with two fictional +model families and fake executables. The profiles file is a strict +`{schema_version, profiles, bindings}` registry; the policy declares each +account pool's allowed billing modes and local concurrency limit. Runtime +configuration uses exact locally verified models and effort settings. These +examples intentionally avoid embedding today's provider model names or +pretending that fixture profiles are proven. + +The CLI and MCP service validate equivalent task objects against the same +packaged contract. The generated command/schema reference is +[docs/generated/core-reference.md](../../../generated/core-reference.md). +Integration tests resolve paths and refs against temporary repositories; +ordinary schema tests validate shape without touching the filesystem. diff --git a/docs/plans/engineering-team/examples/branch-review.json b/docs/plans/engineering-team/examples/branch-review.json new file mode 100644 index 0000000..09cd3f4 --- /dev/null +++ b/docs/plans/engineering-team/examples/branch-review.json @@ -0,0 +1,45 @@ +{ + "schema_version": 1, + "project": { + "repo_path": "/absolute/path/to/fixture-repository", + "base_ref": "fixture-base", + "target_ref": "fixture-candidate" + }, + "workflow": "branch-review", + "goal": "Review the fixture change and report supported correctness findings.", + "task_class": "fixture-review-small", + "acceptance": [ + { + "id": "review-exact-diff", + "description": "Findings identify the resolved candidate and supporting file locations.", + "evidence_kind": "review" + }, + { + "id": "report-check-outcome", + "description": "Include the fixture check result even when it reports a failure.", + "evidence_kind": "check" + } + ], + "checks": [ + { + "id": "fixture-tests", + "argv": ["python3", "-m", "unittest", "discover", "-s", "tests"], + "cwd": ".", + "timeout_seconds": 30, + "required_to_pass": false + } + ], + "scope": {"read_paths": ["src", "tests"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json" + }, + "budget": { + "wall_seconds": 600, + "max_worker_invocations": 3, + "max_revisions": 0, + "max_fallbacks_per_step": 1 + }, + "origin": {"surface": "cli"} +} diff --git a/docs/plans/engineering-team/examples/issue-delivery.json b/docs/plans/engineering-team/examples/issue-delivery.json new file mode 100644 index 0000000..3e14491 --- /dev/null +++ b/docs/plans/engineering-team/examples/issue-delivery.json @@ -0,0 +1,45 @@ +{ + "schema_version": 1, + "project": { + "repo_path": "/absolute/path/to/fixture-repository", + "base_ref": "fixture-base", + "target_ref": "fixture-base" + }, + "workflow": "issue-delivery", + "goal": "Fix the fixture parser's handling of an empty input and add a regression test.", + "task_class": "fixture-bugfix-small", + "acceptance": [ + { + "id": "empty-input", + "description": "Empty input returns the documented empty result without an exception.", + "evidence_kind": "check" + }, + { + "id": "independent-review", + "description": "A different verified model reviews the exact candidate and resolves critical findings.", + "evidence_kind": "review" + } + ], + "checks": [ + { + "id": "fixture-tests", + "argv": ["python3", "-m", "unittest", "discover", "-s", "tests"], + "cwd": ".", + "timeout_seconds": 30, + "required_to_pass": true + } + ], + "scope": {"read_paths": ["src", "tests"], "write_paths": ["src/parser.py", "tests/test_parser.py"]}, + "lead": {"mode": "headless"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json" + }, + "budget": { + "wall_seconds": 1200, + "max_worker_invocations": 9, + "max_revisions": 2, + "max_fallbacks_per_step": 1 + }, + "origin": {"surface": "cli"} +} diff --git a/docs/plans/engineering-team/experiments/decision-helper-baseline-v1.json b/docs/plans/engineering-team/experiments/decision-helper-baseline-v1.json new file mode 100644 index 0000000..07c091b --- /dev/null +++ b/docs/plans/engineering-team/experiments/decision-helper-baseline-v1.json @@ -0,0 +1,38 @@ +{ + "schema_version": 1, + "experiment_id": "devsquad-decision-helper-baseline-v1", + "status": "frozen_offline_baseline", + "purpose": "Validate optional decision-helper mechanics and authority boundaries without claiming routing quality.", + "corpus": { + "path": "jev-pilot-v1.json", + "sha256": "473aed2c431bd1ce8218124a6fe3fd3139fdf869468a72cbc8cf184c428d9822", + "data_class": "synthetic_public_fixture", + "contains_private_content": false, + "case_ids": ["T01", "T02", "T03", "T04", "T05", "T06", "T07", "T08"] + }, + "split": { + "mechanics": ["T01", "T02", "T03", "T04", "T05", "T06"], + "held_out": ["T07", "T08"] + }, + "modes": ["off", "shadow", "advisory"], + "resource_ceiling": { + "max_calls_per_cache_key": 1, + "max_input_bytes": 4096, + "wall_seconds": 2, + "max_cost_usd": 0.01, + "network_calls_in_baseline": 0 + }, + "acceptance": { + "off_mode_route_divergences": 0, + "unauthorized_candidate_selections": 0, + "pin_overrides": 0, + "invalid_output_route_changes": 0, + "duplicate_billable_calls_on_replay": 0, + "raw_private_payloads_persisted": 0 + }, + "adoption": { + "advisory_authorized_by_baseline": false, + "runtime_default": "off", + "reason": "Synthetic mechanics and boundary tests cannot establish production routing quality or benefit." + } +} diff --git a/docs/plans/engineering-team/experiments/jev-pilot-v1.json b/docs/plans/engineering-team/experiments/jev-pilot-v1.json new file mode 100644 index 0000000..2159858 --- /dev/null +++ b/docs/plans/engineering-team/experiments/jev-pilot-v1.json @@ -0,0 +1,130 @@ +{ + "schema_version": 1, + "experiment_id": "jev-devsquad-routing-pilot-v1", + "status": "ready_for_live_run", + "model": "jev-1.13.0", + "endpoint": "https://api.typesafe.ai/v1/systemone", + "pricing": { + "checked_at": "2026-09-26", + "usd_per_million_input_tokens": 0.042, + "max_cost_usd": 0.01 + }, + "budget": { + "max_billable_requests": 1, + "retries": 0, + "timeout_seconds": 30, + "data_class": "synthetic_public_fixture" + }, + "purposes": [ + { + "id": "task_family", + "instructions": "Classify the specified engineering task by its primary purpose. Use uncertain when the task lacks enough information.", + "criteria": { + "bug_fix": "Correct a known faulty behavior or regression.", + "implementation": "Build or materially change a feature.", + "investigation": "Diagnose a cause or analyze evidence without an already-defined fix.", + "review_validation": "Review, test, audit, or verify existing work.", + "documentation": "Primarily change explanatory documentation.", + "operations": "Operate or recover development tooling, source control, packaging, or deployment state.", + "uncertain": "The request is too vague or conflicting to classify safely." + } + }, + { + "id": "execution_tier", + "instructions": "Choose the least costly sufficient execution tier for the specified task. This is advisory only; permissions and verified capabilities are filtered separately.", + "criteria": { + "economy_read": "Low-risk read-only or documentation reasoning with clear scope.", + "standard_write": "Bounded, well-specified code or documentation change with ordinary tests.", + "frontier_analysis": "Difficult diagnosis, architecture, concurrency, or security reasoning without an immediately safe write.", + "frontier_write": "High-risk or complex implementation requiring strong reasoning and mandatory independent verification.", + "lead_clarification": "The task should not be delegated until scope, requirements, or authority are clarified." + } + }, + { + "id": "specialist_skill", + "instructions": "Choose at most one specialist skill that is important for the specified task. Choose none when ordinary engineering instructions are sufficient.", + "criteria": { + "git_safety": "Recovery, branch, worktree, history, or uncommitted-work safety is central.", + "security_review": "Security boundaries, authentication, authorization, secrets, or vulnerability analysis is central.", + "browser_qa": "A browser UI must be operated or visually validated.", + "data_analysis": "Structured data quality, metrics, statistical analysis, or reporting is central.", + "none": "No listed specialist skill is central.", + "uncertain": "The request lacks enough information to select safely." + } + } + ], + "cases": [ + { + "id": "T01", + "task": "Correct two spelling mistakes in a committed README section. No commands beyond the documentation link checker are required.", + "expected": { + "task_family": "documentation", + "execution_tier": "standard_write", + "specialist_skill": "none" + } + }, + { + "id": "T02", + "task": "Fix a deterministic parser off-by-one error in src/parser.py. The failing unit test, allowed file, and acceptance result are supplied.", + "expected": { + "task_family": "bug_fix", + "execution_tier": "standard_write", + "specialist_skill": "none" + } + }, + { + "id": "T03", + "task": "Diagnose an intermittent SQLite lease race observed only across two processes. Preserve all evidence and do not change code until the cause is established.", + "expected": { + "task_family": "investigation", + "execution_tier": "frontier_analysis", + "specialist_skill": "none" + } + }, + { + "id": "T04", + "task": "Repair an authorization bypass in the worker handoff endpoint. The change touches the access-control boundary and must receive security review and mandatory tests.", + "expected": { + "task_family": "bug_fix", + "execution_tier": "frontier_write", + "specialist_skill": "security_review" + } + }, + { + "id": "T05", + "task": "Verify a browser settings workflow at desktop and mobile breakpoints, including the visible success state. Do not modify the application.", + "expected": { + "task_family": "review_validation", + "execution_tier": "economy_read", + "specialist_skill": "browser_qa" + } + }, + { + "id": "T06", + "task": "Make the project better using whichever models and tools seem best.", + "expected": { + "task_family": "uncertain", + "execution_tier": "lead_clarification", + "specialist_skill": "uncertain" + } + }, + { + "id": "T07", + "task": "Analyze a CSV of task outcomes for missing values, routing accuracy, rework rate, and confidence intervals. Produce a source-backed report only.", + "expected": { + "task_family": "investigation", + "execution_tier": "frontier_analysis", + "specialist_skill": "data_analysis" + } + }, + { + "id": "T08", + "task": "Recover valuable uncommitted changes after work continued on the wrong Git branch. Preserve every edit and do not rewrite shared history.", + "expected": { + "task_family": "operations", + "execution_tier": "frontier_write", + "specialist_skill": "git_safety" + } + } + ] +} diff --git a/install.sh b/install.sh index 0da8922..94f4db6 100755 --- a/install.sh +++ b/install.sh @@ -1,117 +1,118 @@ #!/usr/bin/env bash -# DevSquad Installer — registers marketplace, installs plugin, and wires hooks +# DevSquad composite installer: standalone core first, optional Claude plugin. set -euo pipefail -REPO_URL="https://github.com/joshidikshant/devsquad.git" +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd -P)" +REPO_URL="${DEVSQUAD_REPO_URL:-https://github.com/joshidikshant/devsquad.git}" MARKETPLACE="devsquad-marketplace" PLUGIN="devsquad@${MARKETPLACE}" -SETTINGS="$HOME/.claude/settings.json" -PLUGIN_INSTALL_DIR="$HOME/.claude/plugins/marketplaces/${MARKETPLACE}" +CLAUDE_MODE="auto" +STATUS_MODE=0 +JSON_MODE=0 +CORE_ARGS=() + +usage() { + cat <<'EOF' +Usage: ./install.sh [options] + +Installs the standalone DevSquad runtime first. Claude is not required. +When the Claude CLI is available, the legacy plugin is installed or updated +unless --core-only is supplied. + +Composite options: + --core-only Install only the standalone runtime + --with-claude Require and install/update the Claude plugin + +Standalone options are passed to scripts/install-core.sh: + --source-core PATH --install-root PATH --bin-dir PATH + --python PATH --with-mcp --mcp-wheelhouse PATH + --status --json + -h, --help +EOF +} + +fail() { + printf 'devsquad install: %s\n' "$*" >&2 + exit 1 +} + +while [ "$#" -gt 0 ]; do + case "$1" in + --core-only) CLAUDE_MODE="skip"; shift ;; + --with-claude) CLAUDE_MODE="required"; shift ;; + --source-core|--install-root|--bin-dir|--python|--mcp-wheelhouse) + [ "$#" -ge 2 ] || fail "$1 requires a value" + CORE_ARGS+=("$1" "$2"); shift 2 ;; + --with-mcp) + CORE_ARGS+=("$1"); shift ;; + --status) + STATUS_MODE=1; CORE_ARGS+=("$1"); shift ;; + --json) + JSON_MODE=1; CORE_ARGS+=("$1"); shift ;; + -h|--help) usage; exit 0 ;; + *) fail "unknown option: $1" ;; + esac +done + +if [ "$STATUS_MODE" -eq 1 ] && [ "$CLAUDE_MODE" = "required" ]; then + fail "--status cannot be combined with --with-claude" +fi +if [ "$STATUS_MODE" -eq 1 ]; then + CLAUDE_MODE="skip" +fi +if [ "$JSON_MODE" -eq 1 ] && [ "$CLAUDE_MODE" != "skip" ]; then + fail "--json requires --core-only (or --status); use scripts/install-core.sh for standalone JSON" +fi +if [ "$CLAUDE_MODE" = "required" ] && ! command -v claude >/dev/null 2>&1; then + fail "--with-claude requested, but the Claude Code CLI is unavailable" +fi -echo "=== DevSquad Installer ===" -echo +if [ "$JSON_MODE" -eq 1 ]; then + echo "=== DevSquad standalone runtime ===" >&2 +else + echo "=== DevSquad standalone runtime ===" +fi +if [ "${#CORE_ARGS[@]}" -gt 0 ]; then + "$SCRIPT_DIR/scripts/install-core.sh" "${CORE_ARGS[@]}" +else + "$SCRIPT_DIR/scripts/install-core.sh" +fi -# Check claude is available -if ! command -v claude &>/dev/null; then - echo "Error: Claude Code CLI not found. Install it first:" - echo " https://docs.anthropic.com/en/docs/claude-code" - exit 1 +if [ "$CLAUDE_MODE" = "skip" ]; then + if [ "$JSON_MODE" -eq 1 ]; then + echo "Claude plugin: skipped" >&2 + else + echo "Claude plugin: skipped" + fi + exit 0 +fi +if ! command -v claude >/dev/null 2>&1; then + echo "Claude plugin: skipped (Claude Code CLI not found)" + echo "Standalone DevSquad is ready; install the plugin later with ./install.sh --with-claude." + exit 0 fi -# Step 1: Register marketplace -echo "[1/4] Registering marketplace..." +echo +echo "=== DevSquad legacy Claude plugin ===" +echo "[1/3] Marketplace" if claude plugin marketplace list 2>/dev/null | grep -q "$MARKETPLACE"; then - echo " Marketplace already registered, updating..." claude plugin marketplace update "$MARKETPLACE" else claude plugin marketplace add "$REPO_URL" fi -# Step 2: Install plugin -echo "[2/4] Installing plugin..." +echo "[2/3] Plugin" if claude plugin list 2>/dev/null | grep -q "devsquad@"; then - echo " Plugin already installed, updating..." - claude plugin update "$PLUGIN" 2>/dev/null || true + claude plugin update "$PLUGIN" else claude plugin install "$PLUGIN" fi -# Step 3: Enable plugin -echo "[3/4] Enabling plugin..." -claude plugin enable "$PLUGIN" 2>/dev/null || true - -# Step 4: Register hooks into ~/.claude/settings.json (global) -# Hooks point at the MARKETPLACE CLONE (a git checkout that `claude plugin -# marketplace update` refreshes) — never at a versioned cache dir, which -# freezes hooks at install-time and silently drops every later fix. -# Developers hacking on DevSquad itself can point these commands at their -# source checkout instead to run hooks-at-HEAD (see docs/ARCHITECTURE.md). -# Note: per-project hook registration happens during /devsquad:setup (onboarding skill Step 3.5). -echo "[4/4] Registering hooks into global settings.json..." - -if [[ ! -f "$SETTINGS" ]]; then - echo " Creating $SETTINGS..." - echo '{"hooks":{}}' > "$SETTINGS" -fi - -if ! command -v python3 &>/dev/null; then - echo " Warning: python3 not found. Skipping hook registration." - echo " Hooks must be added to $SETTINGS manually." -else - python3 - <=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "devsquad-core" +version = "0.1.0" +requires-python = ">=3.11" +dependencies = [] + +[project.optional-dependencies] +mcp = ["mcp==2.2.0"] + +[project.scripts] +squad = "devsquad.cli:main" + +[tool.setuptools] +package-dir = {"" = "src"} + +[tool.setuptools.packages.find] +where = ["src"] + +[tool.setuptools.package-data] +"devsquad.migrations" = ["*.sql"] + +[tool.setuptools.data-files] +"share/devsquad/adapters/codex" = ["adapters/codex/adapter.json"] +"share/devsquad/adapters/antigravity" = ["adapters/antigravity/adapter.json"] +"share/devsquad/adapters/grok" = ["adapters/grok/adapter.json"] +"share/devsquad/adapters/claude" = ["adapters/claude/adapter.json"] +"share/devsquad/adapters" = ["adapters/classification-policy.conf"] +"share/devsquad/schemas" = ["schemas/adapter.schema.json", "schemas/check-result.schema.json", "schemas/council.schema.json", "schemas/execution-identity.schema.json", "schemas/launch-spec.schema.json", "schemas/normalized-result.schema.json", "schemas/policy.schema.json", "schemas/profile.schema.json", "schemas/profiles.schema.json", "schemas/review-result.schema.json", "schemas/task.schema.json"] +"share/devsquad/profiles" = ["profiles/templates.json"] +"share/devsquad/integrations/codex" = ["integrations/codex/registration.json"] +"share/devsquad/integrations/claude-code" = ["integrations/claude-code/registration.json"] +"share/devsquad/integrations/antigravity" = ["integrations/antigravity/registration.json"] +"share/devsquad/integrations/grok" = ["integrations/grok/registration.json"] diff --git a/plugin/core/requirements-mcp.lock b/plugin/core/requirements-mcp.lock new file mode 100644 index 0000000..7040407 --- /dev/null +++ b/plugin/core/requirements-mcp.lock @@ -0,0 +1,32 @@ +# DevSquad optional local MCP transport. +# Resolved from mcp==2.2.0 on 2026-09-22; install this lock only in the +# isolated MCP environment. The base devsquad-core package has no runtime +# dependencies and must remain independently installable. +annotated-types==0.8.0 +anyio==4.15.1 +attrs==26.1.0 +cffi==2.1.1 +click==8.5.0 +cryptography==50.0.1 +h11==0.16.0 +httpcore2==2.13.0 +httpx2==2.13.0 +idna==3.20 +jsonschema==4.26.0 +jsonschema-specifications==2025.9.1 +mcp==2.2.0 +mcp-types==2.2.0 +opentelemetry-api==1.44.0 +pycparser==3.0 +pydantic==2.13.5 +pydantic_core==2.46.5 +PyJWT==2.14.0 +python-multipart==0.0.32 +referencing==0.37.0 +rpds-py==2026.6.3 +sse-starlette==3.4.11 +starlette==1.6.0 +truststore==0.10.4 +typing-inspection==0.4.4 +typing_extensions==4.16.0 +uvicorn==0.53.0 diff --git a/plugin/core/schemas/adapter.schema.json b/plugin/core/schemas/adapter.schema.json new file mode 100644 index 0000000..af42568 --- /dev/null +++ b/plugin/core/schemas/adapter.schema.json @@ -0,0 +1,23 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/adapter-v1.json", + "type": "object", + "additionalProperties": true, + "required": ["schema_version", "name", "transport", "binary_candidates", "capabilities", "permission_profiles"], + "properties": { + "schema_version": {"const": 1}, + "name": {"type": "string", "minLength": 1}, + "transport": {"enum": ["cli_exec", "native_protocol"]}, + "binary_candidates": {"type": "array", "minItems": 1, "items": {"type": "string"}}, + "capabilities": {"type": "object"}, + "permission_profiles": {"type": "object"}, + "permission_tools": { + "type": "object", + "additionalProperties": { + "type": "array", + "uniqueItems": true, + "items": {"type": "string", "minLength": 1} + } + } + } +} diff --git a/plugin/core/schemas/check-result.schema.json b/plugin/core/schemas/check-result.schema.json new file mode 100644 index 0000000..7cfbab5 --- /dev/null +++ b/plugin/core/schemas/check-result.schema.json @@ -0,0 +1,63 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/check-result-v2.json", + "type": "object", + "additionalProperties": false, + "required": ["schema_version", "candidate_sha256", "target_oid", "id", "argv", "cwd", "required_to_pass", "status", "returncode", "error_code", "duration_ms", "stdout", "stderr"], + "properties": { + "schema_version": {"enum": [1, 2]}, + "candidate_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "target_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, + "id": {"type": "string", "minLength": 1}, + "argv": {"type": "array", "minItems": 1, "items": {"type": "string", "minLength": 1}}, + "cwd": {"type": "string"}, + "required_to_pass": {"type": "boolean"}, + "status": {"enum": ["passed", "failed", "timed_out", "launch_failed", "invalidated", "not_run"]}, + "output_paths": {"type": "array", "maxItems": 32, "uniqueItems": true, "items": {"type": "string", "minLength": 1}}, + "integrity": { + "type": "object", "additionalProperties": false, + "required": ["status", "reasons", "before_state_sha256", "after_state_sha256", "changes", "changes_truncated"], + "properties": { + "status": {"enum": ["verified", "violated", "not_run"]}, + "reasons": {"type": "array", "maxItems": 20, "items": {"type": "string", "minLength": 1, "maxLength": 200}}, + "before_state_sha256": {"type": ["string", "null"], "pattern": "^[0-9a-f]{64}$"}, + "after_state_sha256": {"type": ["string", "null"], "pattern": "^[0-9a-f]{64}$"}, + "changes_truncated": {"type": "boolean"}, + "changes": {"type": "array", "maxItems": 100, "items": { + "type": "object", "additionalProperties": false, + "required": ["workspace", "path", "before", "after"], + "properties": { + "workspace": {"enum": ["check", "review"]}, + "path": {"type": "string", "minLength": 1, "maxLength": 4096}, + "before": {"type": ["string", "null"], "pattern": "^([0-9a-f]{64}|undeclared)$"}, + "after": {"type": ["string", "null"], "pattern": "^([0-9a-f]{64}|undeclared)$"} + } + }} + } + }, + "returncode": {"type": ["integer", "null"]}, + "error_code": {"enum": [null, "TIMEOUT", "CLI_ERROR"]}, + "duration_ms": {"type": "integer", "minimum": 0}, + "stdout": {"$ref": "#/$defs/stream"}, + "stderr": {"$ref": "#/$defs/stream"} + }, + "allOf": [ + {"if": {"properties": {"schema_version": {"const": 2}}}, + "then": {"required": ["integrity", "output_paths"]}, + "else": {"not": {"anyOf": [{"required": ["integrity"]}, {"required": ["output_paths"]}]}, "properties": {"status": {"enum": ["passed", "failed", "timed_out", "launch_failed"]}}}} + ], + "$defs": { + "stream": { + "type": "object", + "additionalProperties": false, + "required": ["preview", "captured_bytes", "total_bytes", "truncated", "full_sha256"], + "properties": { + "preview": {"type": "string", "maxLength": 1024}, + "captured_bytes": {"type": "integer", "minimum": 0}, + "total_bytes": {"type": "integer", "minimum": 0}, + "truncated": {"type": "boolean"}, + "full_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"} + } + } + } +} diff --git a/plugin/core/schemas/council.schema.json b/plugin/core/schemas/council.schema.json new file mode 100644 index 0000000..e2717f6 --- /dev/null +++ b/plugin/core/schemas/council.schema.json @@ -0,0 +1,17 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/council-v1.json", + "type": "object", "additionalProperties": false, + "required": ["schema_version", "enabled", "automatic", "reason", "min_valid_proposals", "required_critics", "max_invocations", "seed", "evidence", "rubric"], + "properties": { + "schema_version": {"const": 1}, "enabled": {"const": true}, "automatic": {"const": false}, + "reason": {"type": "string", "minLength": 1, "maxLength": 16000}, + "min_valid_proposals": {"const": 2}, "required_critics": {"const": 1}, + "max_invocations": {"type": "integer", "minimum": 3, "maximum": 16}, + "seed": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "evidence": {"type": "array", "maxItems": 32, "items": {"type": "object", "additionalProperties": false, + "required": ["artifact_id", "sha256"], "properties": {"artifact_id": {"type": "string", "minLength": 1, "maxLength": 16000}, "sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}}}}, + "rubric": {"type": "array", "minItems": 1, "maxItems": 32, "items": {"type": "object", "additionalProperties": false, + "required": ["id", "description"], "properties": {"id": {"type": "string", "minLength": 1, "maxLength": 16000}, "description": {"type": "string", "minLength": 1, "maxLength": 16000}}}} + } +} diff --git a/plugin/core/schemas/execution-identity.schema.json b/plugin/core/schemas/execution-identity.schema.json new file mode 100644 index 0000000..80cab06 --- /dev/null +++ b/plugin/core/schemas/execution-identity.schema.json @@ -0,0 +1,13 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/execution-identity-v1.json", + "type": "object", "additionalProperties": false, + "required": ["harness", "harness_version", "model_provider", "model_family", "model", "effort", "tools", "permissions", "account_pool", "verification"], + "properties": { + "harness": {"type": "string", "minLength": 1}, "harness_version": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, + "model_provider": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, "model_family": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, "model": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, + "effort": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, "tools": {"type": "array", "items": {"type": "string", "minLength":1}, "uniqueItems": true}, + "permissions": {"enum": ["read_only", "workspace_write"]}, "account_pool": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, + "verification": {"enum": ["verified", "unverified", "unavailable", "unknown"]} + } +} diff --git a/plugin/core/schemas/launch-spec.schema.json b/plugin/core/schemas/launch-spec.schema.json new file mode 100644 index 0000000..481d795 --- /dev/null +++ b/plugin/core/schemas/launch-spec.schema.json @@ -0,0 +1,18 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/launch-spec-v1.json", + "type": "object", + "additionalProperties": false, + "required": ["schema_version", "adapter", "transport", "argv", "cwd", "stdin_path", "timeout_seconds", "requested", "environment"], + "properties": { + "schema_version": {"const": 1}, + "adapter": {"type": "string", "minLength": 1}, + "transport": {"enum": ["cli_exec", "native_protocol"]}, + "argv": {"type": "array", "minItems": 1, "items": {"type": "string", "minLength": 1}}, + "cwd": {"type": "string", "minLength": 1}, + "stdin_path": {"oneOf": [{"type": "string", "minLength": 1}, {"type": "null"}]}, + "timeout_seconds": {"type": "integer", "minimum": 1}, + "requested": {"$ref": "execution-identity.schema.json"}, + "environment": {"type": "object", "additionalProperties": false, "properties": {"DEVSQUAD_WORKER":{"type":"string"},"DEVSQUAD_RUN_ID":{"type":"string"},"DEVSQUAD_ATTEMPT_ID":{"type":"string"},"DEVSQUAD_DELEGATION_DEPTH":{"type":"string"}}} + } +} diff --git a/plugin/core/schemas/normalized-result.schema.json b/plugin/core/schemas/normalized-result.schema.json new file mode 100644 index 0000000..ae41124 --- /dev/null +++ b/plugin/core/schemas/normalized-result.schema.json @@ -0,0 +1,19 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/normalized-result-v1.json", + "type": "object", + "additionalProperties": false, + "required": ["schema_version", "execution_status", "error_code", "output", "artifact_status", "acceptance_status", "requested", "observed", "native_ids", "events"], + "properties": { + "schema_version": {"const": 1}, + "execution_status": {"enum": ["succeeded", "failed", "timed_out", "interrupted", "denied", "malformed"]}, + "error_code": {"type": ["string", "null"]}, + "output": {"type": ["string", "null"]}, + "artifact_status": {"enum": ["present", "missing", "not_required", "unknown"]}, + "acceptance_status": {"enum": ["pending", "accepted", "rejected", "not_evaluated"]}, + "requested": {"$ref": "execution-identity.schema.json"}, + "observed": {"oneOf": [{"$ref": "execution-identity.schema.json"}, {"type": "null"}]}, + "native_ids": {"type": "object", "propertyNames":{"minLength":1}, "additionalProperties": {"type": "string", "minLength":1}}, + "events": {"type": "array", "items": {"type": "object"}} + } +} diff --git a/plugin/core/schemas/policy.schema.json b/plugin/core/schemas/policy.schema.json new file mode 100644 index 0000000..fd91534 --- /dev/null +++ b/plugin/core/schemas/policy.schema.json @@ -0,0 +1,17 @@ +{ + "$schema":"https://json-schema.org/draft/2020-12/schema","$id":"https://devsquad.local/schemas/policy-v1.json", + "type":"object","additionalProperties":false, + "required":["schema_version","id","version","roles","task_classes","require_different_model_for_review","account_pools","experiment_budget"], + "properties":{ + "schema_version":{"const":1},"id":{"type":"string","minLength":1},"version":{"type":"integer","minimum":1}, + "roles":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher","proposer_a","proposer_b","critic"]},"additionalProperties":{"type":"array","minItems":1,"items":{"$ref":"#/$defs/candidate"}}}, + "task_classes":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"enum":["unvalidated","trial","proven","suspended"]}},"require_different_model_for_review":{"type":"boolean"},"prefer_different_harness_for_review":{"type":"boolean"}, + "account_pools":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"$ref":"#/$defs/account_pool"}},"experiment_budget":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"type":"integer","minimum":0}},"decision_helper":{"oneOf":[{"type":"object","additionalProperties":false,"required":["schema_version","mode"],"properties":{"schema_version":{"const":1},"mode":{"const":"off"}}},{"$ref":"#/$defs/decision_helper_enabled"}]} + }, + "$defs":{ + "candidate":{"type":"object","additionalProperties":false,"required":["kind","id"],"properties":{"kind":{"enum":["profile","alias"]},"id":{"type":"string","minLength":1}}}, + "account_pool":{"type":"object","additionalProperties":false,"required":["allowed_billing_modes","max_concurrency"],"properties":{"allowed_billing_modes":{"type":"array","minItems":1,"uniqueItems":true,"items":{"enum":["subscription","paid_api"]}},"max_concurrency":{"type":"integer","minimum":1},"unknown_capacity_policy":{"enum":["allow_bounded","block"]}}}, + "sha256":{"type":"string","pattern":"^[0-9a-f]{64}$"}, + "decision_helper_enabled":{"type":"object","additionalProperties":false,"required":["schema_version","mode","purpose","adapter","language","min_confidence","gate_evidence_sha256","budget"],"properties":{"schema_version":{"const":1},"mode":{"enum":["shadow","advisory"]},"purpose":{"type":"object","additionalProperties":false,"required":["id","version","question_sha256","rubric_sha256"],"properties":{"id":{"type":"string","minLength":1},"version":{"type":"integer","minimum":1},"question_sha256":{"$ref":"#/$defs/sha256"},"rubric_sha256":{"$ref":"#/$defs/sha256"}}},"adapter":{"type":"object","additionalProperties":false,"required":["id","model","runtime_revision","calibration_version"],"properties":{"id":{"type":"string","minLength":1},"model":{"type":"string","minLength":1},"runtime_revision":{"type":"string","minLength":1},"calibration_version":{"type":["string","null"],"minLength":1}}},"language":{"type":"string","minLength":1},"min_confidence":{"type":"number","minimum":0,"maximum":1},"gate_evidence_sha256":{"type":["string","null"],"pattern":"^[0-9a-f]{64}$"},"budget":{"type":"object","additionalProperties":false,"required":["max_calls","max_input_bytes","wall_seconds","max_cost_usd"],"properties":{"max_calls":{"const":1},"max_input_bytes":{"type":"integer","minimum":1},"wall_seconds":{"type":"integer","minimum":1},"max_cost_usd":{"type":"number","minimum":0}}}}} + } +} diff --git a/plugin/core/schemas/profile.schema.json b/plugin/core/schemas/profile.schema.json new file mode 100644 index 0000000..4f12512 --- /dev/null +++ b/plugin/core/schemas/profile.schema.json @@ -0,0 +1,20 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/profile-v1.json", + "type": "object", + "additionalProperties": false, + "required": ["id", "harness", "model_family", "model_id", "effort", "required_tools", "permission_policy", "account_pool_id", "billing_mode", "quality_status", "evidence_refs"], + "properties": { + "id": {"type": "string", "minLength": 1}, + "harness": {"type": "string", "minLength": 1}, + "model_family": {"type": "string", "minLength": 1}, + "model_id": {"type": "string", "minLength": 1}, + "effort": {"type": "object", "additionalProperties": false, "required": ["value", "transport"], "properties": {"value": {"type": ["string", "null"]}, "transport": {"enum": ["native", "model_variant", "provider_default"]}}}, + "required_tools": {"type": "array", "items": {"type": "string", "minLength":1}, "uniqueItems": true}, + "permission_policy": {"enum": ["read_only", "workspace_write"]}, + "account_pool_id": {"type": "string", "minLength": 1}, + "billing_mode": {"enum": ["subscription", "paid_api"]}, + "quality_status": {"enum": ["unvalidated", "trial", "proven", "suspended"]}, + "evidence_refs": {"type": "array", "items": {"type": "string", "minLength":1}, "uniqueItems": true} + } +} diff --git a/plugin/core/schemas/profiles.schema.json b/plugin/core/schemas/profiles.schema.json new file mode 100644 index 0000000..30ce652 --- /dev/null +++ b/plugin/core/schemas/profiles.schema.json @@ -0,0 +1,31 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/profiles-v1.json", + "type": "object", + "additionalProperties": false, + "required": ["schema_version", "profiles", "bindings"], + "properties": { + "schema_version": {"const": 1}, + "profiles": { + "type": "array", + "minItems": 1, + "items": {"$ref": "profile.schema.json"} + }, + "bindings": { + "type": "object", + "propertyNames": {"minLength": 1}, + "additionalProperties": {"$ref": "#/$defs/binding"} + } + }, + "$defs": { + "binding": { + "type": "object", + "additionalProperties": false, + "required": ["profile_id", "version"], + "properties": { + "profile_id": {"type": "string", "minLength": 1}, + "version": {"type": "integer", "minimum": 1} + } + } + } +} diff --git a/plugin/core/schemas/review-result.schema.json b/plugin/core/schemas/review-result.schema.json new file mode 100644 index 0000000..95e673d --- /dev/null +++ b/plugin/core/schemas/review-result.schema.json @@ -0,0 +1,38 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/review-result-v1.json", + "type": "object", + "additionalProperties": false, + "required": ["schema_version", "candidate_sha256", "base_oid", "target_oid", "review_mode", "verdict", "summary", "findings"], + "properties": { + "schema_version": {"const": 1}, + "candidate_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "base_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, + "target_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, + "review_mode": {"enum": ["standard", "adversarial"]}, + "verdict": {"enum": ["clean", "findings"]}, + "summary": {"type": "string", "minLength": 1, "maxLength": 20000}, + "findings": { + "type": "array", + "maxItems": 100, + "items": {"$ref": "#/$defs/finding"} + } + }, + "$defs": { + "finding": { + "type": "object", + "additionalProperties": false, + "required": ["id", "severity", "title", "description", "path", "start_line", "end_line", "evidence"], + "properties": { + "id": {"type": "string", "minLength": 1, "maxLength": 200}, + "severity": {"enum": ["critical", "high", "medium", "low"]}, + "title": {"type": "string", "minLength": 1, "maxLength": 500}, + "description": {"type": "string", "minLength": 1, "maxLength": 20000}, + "path": {"type": "string", "minLength": 1}, + "start_line": {"type": "integer", "minimum": 1}, + "end_line": {"type": "integer", "minimum": 1}, + "evidence": {"type": "string", "minLength": 1, "maxLength": 20000} + } + } + } +} diff --git a/plugin/core/schemas/task.schema.json b/plugin/core/schemas/task.schema.json new file mode 100644 index 0000000..ef50f63 --- /dev/null +++ b/plugin/core/schemas/task.schema.json @@ -0,0 +1,26 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", "$id": "https://devsquad.local/schemas/task-v1.json", + "type": "object", "additionalProperties": false, + "required": ["schema_version", "project", "workflow", "goal", "task_class", "acceptance", "checks", "scope", "lead", "routing", "budget", "origin"], + "properties": { + "schema_version": {"const": 1}, "workflow": {"enum": ["branch-review", "issue-delivery", "council-decision"]}, + "goal": {"type": "string", "minLength": 1}, "task_class": {"type": "string", "minLength": 1}, + "project": {"$ref": "#/$defs/project"}, "acceptance": {"type": "array", "minItems": 1, "maxItems": 100, "items": {"$ref": "#/$defs/acceptance"}}, + "checks": {"type": "array", "maxItems": 16, "items": {"$ref": "#/$defs/check"}}, "scope": {"$ref": "#/$defs/scope"}, + "lead": {"$ref": "#/$defs/lead"}, "routing": {"$ref": "#/$defs/routing"}, "budget": {"$ref": "#/$defs/budget"}, + "origin": {"$ref": "#/$defs/origin"}, "review": {"$ref": "#/$defs/review"}, "council": {"$ref": "council.schema.json"} + }, + "allOf": [{"if": {"properties": {"workflow": {"const": "council-decision"}}}, "then": {"required": ["council"], "properties": {"budget": {"properties": {"max_revisions": {"const": 0}}}, "scope": {"properties": {"write_paths": {"maxItems": 0}}}}, "not": {"required": ["review"]}}, "else": {"not": {"required": ["council"]}}}], + "$defs": { + "project": {"type":"object","additionalProperties":false,"required":["repo_path","base_ref","target_ref"],"properties":{"repo_path":{"type":"string","pattern":"^/"},"base_ref":{"type":"string","minLength":1},"target_ref":{"type":"string","minLength":1}}}, + "acceptance": {"type":"object","additionalProperties":false,"required":["id","description","evidence_kind"],"properties":{"id":{"type":"string","minLength":1},"description":{"type":"string","minLength":1},"evidence_kind":{"enum":["review","check","artifact","host"]}}}, + "check": {"type":"object","additionalProperties":false,"required":["id","argv","cwd","timeout_seconds","required_to_pass"],"properties":{"id":{"type":"string","minLength":1},"argv":{"type":"array","minItems":1,"items":{"type":"string","minLength":1}},"cwd":{"type":"string"},"timeout_seconds":{"type":"integer","minimum":1},"required_to_pass":{"type":"boolean"},"output_paths":{"type":"array","maxItems":32,"uniqueItems":true,"items":{"type":"string","minLength":1}}}}, + "scope": {"type":"object","additionalProperties":false,"required":["read_paths","write_paths"],"properties":{"read_paths":{"type":"array","maxItems":256,"uniqueItems":true,"items":{"type":"string","minLength":1}},"write_paths":{"type":"array","maxItems":256,"uniqueItems":true,"items":{"type":"string","minLength":1}}}}, + "lead": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["host","headless"]}}}, + "override": {"type":"object","additionalProperties":false,"required":["profile_id"],"properties":{"profile_id":{"type":"string","minLength":1},"fallback":{"enum":["none","policy"]}}}, + "routing": {"type":"object","additionalProperties":false,"properties":{"profiles_file":{"type":"string","minLength":1},"policy_file":{"type":"string","minLength":1},"profiles":{"type":"object"},"policy":{"type":"object"},"overrides":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher","proposer_a","proposer_b","critic"]},"additionalProperties":{"$ref":"#/$defs/override"}}},"oneOf":[{"required":["profiles_file","policy_file"],"not":{"anyOf":[{"required":["profiles"]},{"required":["policy"]}]}},{"required":["profiles","policy"],"not":{"anyOf":[{"required":["profiles_file"]},{"required":["policy_file"]}]}}]}, + "budget": {"type":"object","additionalProperties":false,"required":["wall_seconds","max_worker_invocations","max_revisions","max_fallbacks_per_step"],"properties":{"wall_seconds":{"type":"integer","minimum":1},"max_worker_invocations":{"type":"integer","minimum":1},"max_revisions":{"type":"integer","minimum":0},"max_fallbacks_per_step":{"type":"integer","minimum":0}}}, + "origin": {"type":"object","additionalProperties":false,"required":["surface"],"properties":{"surface":{"type":"string","minLength":1},"session_ref":{"type":"string","minLength":1}}}, + "review": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["standard","adversarial"]},"focus":{"type":"string","minLength":1}}} + } +} diff --git a/plugin/core/src/devsquad/__init__.py b/plugin/core/src/devsquad/__init__.py new file mode 100644 index 0000000..22a4c4a --- /dev/null +++ b/plugin/core/src/devsquad/__init__.py @@ -0,0 +1,3 @@ +"""DevSquad's surface-independent local core.""" + +__version__ = "0.1.0" diff --git a/plugin/core/src/devsquad/adapters.py b/plugin/core/src/devsquad/adapters.py new file mode 100644 index 0000000..87f759a --- /dev/null +++ b/plugin/core/src/devsquad/adapters.py @@ -0,0 +1,265 @@ +"""Manifest-driven M1 adapter preparation and output classification.""" + +from __future__ import annotations + +import json +import os +import re +import shutil +import subprocess +import shlex +import sys +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +from .contracts import ContractError, ExecutionIdentity, LaunchSpec, NormalizedResult, ProfileUnsupported, SCHEMA_VERSION +from .catalog import verified_efforts + +def _classification_policy() -> dict[str, str]: + source = Path(__file__).resolve().parents[2] / "adapters" / "classification-policy.conf" + if not source.exists(): + source = Path(sys.prefix) / "share" / "devsquad" / "adapters" / "classification-policy.conf" + values = {} + for line in source.read_text().splitlines(): + if line and not line.startswith("#"): + key, raw = line.split("=", 1); values[key] = shlex.split(raw)[0] + return values + + +_POLICY = _classification_policy() +ERROR_PATTERNS = (("AUTH_ERROR", re.compile(_POLICY["DEVSQUAD_AUTH_ERROR_PATTERN"], re.I)), ("RATE_LIMITED", re.compile(_POLICY["DEVSQUAD_RATE_LIMIT_PATTERN"], re.I))) +DENIED_PATTERN = re.compile(_POLICY["DEVSQUAD_DENIED_PATTERN"], re.I) + + +@dataclass(frozen=True) +class AdapterManifest: + name: str + transport: str + binary_candidates: tuple[str, ...] + model_provider: str | None + efforts_by_model: dict[str, tuple[str, ...]] + permission_profiles: dict[str, tuple[str, ...]] + permission_tools: dict[str, tuple[str, ...]] + output_format: str + verified_versions: tuple[str, ...] + + @classmethod + def load(cls, path: Path) -> "AdapterManifest": + raw = json.loads(path.read_text()) + required = {"schema_version", "name", "transport", "binary_candidates", "capabilities"} + missing = required - raw.keys() + if missing or raw["schema_version"] != 1: + raise ContractError(f"invalid adapter manifest: missing={sorted(missing)}") + capabilities = raw["capabilities"] + return cls( + name=raw["name"], transport=raw["transport"], + binary_candidates=tuple(raw["binary_candidates"]), + model_provider=raw.get("model_provider"), + efforts_by_model={k: tuple(v) for k, v in capabilities.get("efforts_by_model", {}).items()}, + permission_profiles={k: tuple(v) for k, v in raw.get("permission_profiles", {}).items()}, + permission_tools={k: tuple(v) for k, v in raw.get("permission_tools", {}).items()}, + output_format=raw.get("output_format", "text"), + verified_versions=tuple(raw.get("verified_harness_versions", [])), + ) + + def resolve_binary(self) -> str | None: + available = tuple( + dict.fromkeys( + path + for name in self.binary_candidates + if (path := shutil.which(name)) is not None + ) + ) + if self.verified_versions: + verified = next( + ( + path + for path in available + if harness_version(path) in self.verified_versions + ), + None, + ) + if verified is not None: + return verified + return available[0] if available else None + + def with_model_efforts(self, mapping: dict[str, tuple[str, ...]]) -> "AdapterManifest": + return AdapterManifest( + self.name, + self.transport, + self.binary_candidates, + self.model_provider, + mapping, + self.permission_profiles, + self.permission_tools, + self.output_format, + self.verified_versions, + ) + + +def _permission_args(manifest: AdapterManifest, permission: str) -> tuple[str, ...]: + try: + return manifest.permission_profiles[permission] + except KeyError as exc: + raise ContractError(f"unsupported permission profile: {permission}") from exc + + +def prepare_cli( + manifest: AdapterManifest, *, prompt: str, cwd: str, model: str | None, + effort: str | None, permission: str, timeout_seconds: int, + stdin_path: str | None = None, harness_version_value: str | None = None, +) -> LaunchSpec: + binary = manifest.resolve_binary() + if not binary: + raise ContractError(f"adapter unavailable: {manifest.name}") + if effort is not None: + supported = manifest.efforts_by_model.get(model or "") + if supported is None or effort not in supported: + raise ProfileUnsupported(f"unsupported or unverified effort {effort!r} for {manifest.name} model {model!r}") + verification = "unverified" + if harness_version_value is not None and manifest.verified_versions: + if harness_version_value not in manifest.verified_versions: + raise ProfileUnsupported( + f"unverified {manifest.name} harness version: {harness_version_value}" + ) + verification = "verified" + permission_args = _permission_args(manifest, permission) + args: list[str] + if manifest.name == "codex": + sandbox = "read-only" if permission == "read_only" else "workspace-write" + args = [binary, "exec", "--json", "--cd", cwd, "--sandbox", sandbox] + if model: + args += ["--model", model] + if effort: + args += ["-c", f'model_reasoning_effort="{effort}"'] + args += [prompt] + elif manifest.name == "antigravity": + args = [binary, "--print", prompt, "--output-format", "json", "--mode", "plan" if permission == "read_only" else "accept-edits"] + if model: + args += ["--model", model] + if effort: + args += ["--effort", effort] + elif manifest.name == "grok": + args = [binary, "--single", prompt, "--output-format", "json", "--cwd", cwd, "--permission-mode", "plan" if permission == "read_only" else "acceptEdits", "--no-subagents"] + if model: + args += ["--model", model] + if effort: + args += ["--reasoning-effort", effort] + elif manifest.name == "claude": + args = [ + binary, + "--print", + "--output-format", "json", + "--safe-mode", + "--disable-slash-commands", + "--no-session-persistence", + "--strict-mcp-config", + "--mcp-config", '{"mcpServers":{}}', + "--no-chrome", + ] + if model: + args += ["--model", model] + if effort: + args += ["--effort", effort] + args.extend(permission_args) + args += ["--", prompt] + permission_args = () + else: + raise ContractError(f"no argv builder for adapter: {manifest.name}") + args.extend(permission_args) + requested = ExecutionIdentity( + harness=manifest.name, harness_version=harness_version_value, + model_provider=manifest.model_provider, + model_family=None, model=model, effort=effort, permissions=permission, + tools=manifest.permission_tools.get(permission, ()), + verification=verification, + ) + return LaunchSpec(SCHEMA_VERSION, manifest.name, "cli_exec", tuple(args), str(Path(cwd).resolve()), stdin_path, timeout_seconds, requested, {"DEVSQUAD_WORKER": "1"}) + + +def prepare_native_codex(manifest: AdapterManifest, *, cwd: str, model: str, effort: str, permission: str, timeout_seconds: int, harness_version_value: str) -> LaunchSpec: + """Prepare the supervisor-owned app-server child without spawning it.""" + if manifest.name != "codex" or manifest.transport != "native_protocol": + raise ContractError("native Codex manifest required") + binary = manifest.resolve_binary() + if not binary: + raise ContractError("adapter unavailable: codex") + if harness_version_value not in manifest.verified_versions: + raise ProfileUnsupported(f"unverified Codex app-server version: {harness_version_value}") + supported = manifest.efforts_by_model.get(model) + if supported is None or effort not in supported: + raise ProfileUnsupported(f"unsupported or unverified effort {effort!r} for codex model {model!r}") + _permission_args(manifest, permission) + requested = ExecutionIdentity("codex", harness_version_value, "openai", None, model, effort, (), permission, None, "verified") + argv = (binary, "-c", f'model="{model}"', "-c", f'model_reasoning_effort="{effort}"', "app-server", "--listen", "stdio://") + return LaunchSpec(SCHEMA_VERSION, "codex", "native_protocol", argv, str(Path(cwd).resolve()), None, timeout_seconds, requested, {"DEVSQUAD_WORKER": "1"}) + + +def prepare_native_codex_from_catalog(manifest: AdapterManifest, snapshot: dict[str, Any], *, cwd: str, model: str, effort: str, permission: str, timeout_seconds: int, harness_version_value: str) -> LaunchSpec: + efforts = verified_efforts(snapshot, harness="codex", version=harness_version_value, model_id=model) + return prepare_native_codex(manifest.with_model_efforts({model: efforts}), cwd=cwd, model=model, effort=effort, permission=permission, timeout_seconds=timeout_seconds, harness_version_value=harness_version_value) + + +def _provider_records(adapter: str, stdout: str) -> tuple[list[dict[str, Any]], bool, bool, str | None]: + records: list[dict[str, Any]] = [] + deliverable = False + error_text = None + terminal = False + try: + document = json.loads(stdout) + source = document if isinstance(document, list) else [document] + except json.JSONDecodeError: + source = [json.loads(line) for line in stdout.splitlines() if line.strip()] + for item in source: + if not isinstance(item, dict): + raise json.JSONDecodeError("record is not an object", stdout, 0) + records.append(item) + if item.get("is_error") is True or item.get("error"): + error_text = str(item.get("error") or item.get("result") or item.get("message")) + kind = item.get("type") + if kind in {"result", "turn.completed"}: + terminal = True + if adapter == "codex" and kind == "item.completed": + native_item = item.get("item") or {} + if not isinstance(native_item, dict): + raise json.JSONDecodeError("item.completed item is not an object", stdout, 0) + if native_item.get("type") in {"agent_message", "agentMessage"}: + payload = native_item.get("text") or native_item.get("content") + deliverable = isinstance(payload, str) and bool(payload.strip()) + if kind in {"result", "assistant_message", "turn.completed"} and item.get("is_error") is not True: + payload = item.get("result") or item.get("message") or item.get("text") + if isinstance(payload, str) and payload.strip(): + deliverable = True + return records, deliverable, terminal, error_text + + +def classify_cli(spec: LaunchSpec, *, returncode: int, stdout: str, stderr: str, timed_out: bool = False) -> NormalizedResult: + code = next((code for code, pattern in ERROR_PATTERNS if pattern.search(stderr)), None) + status = "succeeded" + if code == "AUTH_ERROR" or code == "RATE_LIMITED": + status = "failed" + elif timed_out or returncode in (124, 137, 143): + status, code = "timed_out", "TIMEOUT" + elif returncode != 0: + status, code = "failed", "CLI_ERROR" + elif not stdout.strip(): + status, code = "malformed", "CLI_ERROR" + elif spec.adapter in {"codex", "antigravity", "grok", "claude"}: + try: + _, deliverable, terminal, provider_error = _provider_records(spec.adapter, stdout) + if provider_error: + code = next((candidate for candidate, pattern in ERROR_PATTERNS if pattern.search(provider_error)), "CLI_ERROR") + status = "denied" if DENIED_PATTERN.search(provider_error) else "failed" + elif not deliverable or not terminal: + status, code = "malformed", "CLI_ERROR" + except json.JSONDecodeError: + status, code = "malformed", "CLI_ERROR" + return NormalizedResult(SCHEMA_VERSION, status, code, stdout if stdout else None, "unknown", "not_evaluated", spec.requested, None) + + +def harness_version(binary: str) -> str | None: + try: + return subprocess.run([binary, "--version"], text=True, capture_output=True, timeout=3, check=False).stdout.strip() or None + except (OSError, subprocess.TimeoutExpired): + return None diff --git a/plugin/core/src/devsquad/attempt_runner.py b/plugin/core/src/devsquad/attempt_runner.py new file mode 100644 index 0000000..a4a3291 --- /dev/null +++ b/plugin/core/src/devsquad/attempt_runner.py @@ -0,0 +1,81 @@ +"""Frozen, gated worker owner that survives its detached coordinator.""" +import argparse, hashlib, json, os, signal, subprocess, threading, time +from pathlib import Path +from .store import Store +from .supervisor import process_start_identity, inspect_process, _live_group_exists + +def _atomic(path: Path, value): + temporary=path.with_name(path.name+f".tmp.{os.getpid()}") + with temporary.open("w") as stream: + stream.write(json.dumps(value,sort_keys=True)+"\n"); stream.flush(); os.fsync(stream.fileno()) + os.replace(temporary,path) + directory=os.open(path.parent,os.O_RDONLY) + try: os.fsync(directory) + finally: os.close(directory) + +def _drain(stream, capture: Path, limit: int, result: dict): + digest=hashlib.sha256(); total=0; kept=0 + with capture.open("wb") as output: + while True: + chunk=stream.read(65536) + if not chunk: break + total+=len(chunk); digest.update(chunk); remaining=limit-kept + if remaining>0: output.write(chunk[:remaining]); kept+=min(remaining,len(chunk)) + output.flush(); os.fsync(output.fileno()) + result.update(total_bytes=total,captured_bytes=kept,truncated=total>kept,full_sha256=digest.hexdigest(),captured_sha256=hashlib.sha256(capture.read_bytes()).hexdigest()) + +def main(argv=None): + p=argparse.ArgumentParser(); p.add_argument("--gate-fd",type=int,required=True); p.add_argument("--stdin-fd",type=int); p.add_argument("--database",type=Path,required=True); p.add_argument("--artifacts",type=Path,required=True); p.add_argument("--run-id",required=True); p.add_argument("--attempt-token",required=True); p.add_argument("--supervisor-token",type=int,required=True); p.add_argument("--stdout",type=Path,required=True); p.add_argument("--stderr",type=Path,required=True); p.add_argument("--exit-record",type=Path,required=True); p.add_argument("--child-record",type=Path,required=True); p.add_argument("--limit",type=int,required=True); p.add_argument("--timeout",type=float,required=True); p.add_argument("--grace",type=float,required=True); p.add_argument("command",nargs=argparse.REMAINDER) + a=p.parse_args(argv); command=a.command[1:] if a.command[:1]==["--"] else a.command + with os.fdopen(a.gate_fd,"rb",closefd=True) as gate: + if gate.read(1)!=b"1": + if a.stdin_fd is not None: os.close(a.stdin_fd) + return 125 + child_gate_read,child_gate_write=os.pipe() + gated=[os.sys.executable,"-P","-m","devsquad.worker_gate","--gate-fd",str(child_gate_read),"--",*command] + stdin_source=subprocess.DEVNULL if a.stdin_fd is None else a.stdin_fd + try: + child=subprocess.Popen(gated,stdin=stdin_source,stdout=subprocess.PIPE,stderr=subprocess.PIPE,start_new_session=True,pass_fds=(child_gate_read,)) + finally: + if a.stdin_fd is not None: os.close(a.stdin_fd) + os.close(child_gate_read) + started=process_start_identity(child.pid) + if started is None: + child.kill(); child.wait(); return 126 + _atomic(a.child_record,{"pid":child.pid,"pgid":child.pid,"process_start_id":started}) + os.write(child_gate_write,b"1"); os.close(child_gate_write) + out,err={},{}; threads=[threading.Thread(target=_drain,args=(child.stdout,a.stdout,a.limit,out)),threading.Thread(target=_drain,args=(child.stderr,a.stderr,a.limit,err))] + for thread in threads: thread.start() + store=Store(a.database,a.artifacts); cancelled=False; timed_out=False; deadline=time.monotonic()+a.timeout + try: + while child.poll() is None: + run=store.run(a.run_id) + if (run["state"]=="cancelling" or time.monotonic()>=deadline + or store.remaining_wall_seconds(a.run_id) == 0): + cancelled=run["state"]=="cancelling"; timed_out=not cancelled + if inspect_process(child.pid,child.pid,started)!="live": raise RuntimeError("child identity became unsafe") + try: os.killpg(child.pid,signal.SIGTERM) + except ProcessLookupError: pass + try: child.wait(timeout=a.grace) + except subprocess.TimeoutExpired: + try: os.killpg(child.pid,signal.SIGKILL) + except ProcessLookupError: pass + child.wait() + break + store.heartbeat_attempt(a.run_id,a.attempt_token,a.supervisor_token); time.sleep(.1) + returncode=child.wait() + finally: store.close() + if _live_group_exists(child.pid): + try: os.killpg(child.pid,signal.SIGKILL) + except ProcessLookupError: pass + cleanup_deadline=time.monotonic()+a.grace + while _live_group_exists(child.pid) and time.monotonic() datetime: + current = datetime.now(timezone.utc) if value is None else value + if not isinstance(current, datetime) or current.tzinfo is None or current.utcoffset() is None: + raise ContractError("capacity evaluation time must include a timezone") + return current.astimezone(timezone.utc) + + +def _timestamp(value: Any, field: str, *, nullable: bool = False) -> datetime | None: + if value is None and nullable: + return None + if not isinstance(value, str) or not value: + suffix = " or null" if nullable else "" + raise ContractError(f"capacity {field} must be an ISO timestamp{suffix}") + try: + parsed = datetime.fromisoformat(value) + except ValueError as exc: + raise ContractError(f"capacity {field} must be an ISO timestamp") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ContractError(f"capacity {field} must include a timezone") + return parsed.astimezone(timezone.utc) + + +def _identifier(value: Any, field: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ContractError(f"capacity {field} must be a non-empty string") + return value + + +def _measurement(value: Any, field: str) -> int | float | None: + if value is None: + return None + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ContractError(f"capacity {field} must be a finite non-negative number or null") + if not math.isfinite(value) or value < 0: + raise ContractError(f"capacity {field} must be a finite non-negative number or null") + return value + + +def _selector(value: Any) -> dict[str, list[str]]: + if not isinstance(value, dict) or set(value) != APPLIES_TO_FIELDS: + raise ContractError("capacity applies_to fields are invalid") + normalized: dict[str, list[str]] = {} + for field in sorted(APPLIES_TO_FIELDS): + entries = value[field] + if not isinstance(entries, list): + raise ContractError(f"capacity applies_to.{field} must be an array") + if any(not isinstance(item, str) or not item for item in entries): + raise ContractError( + f"capacity applies_to.{field} must contain non-empty strings", + ) + if len(set(entries)) != len(entries): + raise ContractError(f"capacity applies_to.{field} must be unique") + normalized[field] = sorted(entries) + return normalized + + +def validate_observation( + value: dict[str, Any], *, now: datetime | None = None, +) -> dict[str, Any]: + """Validate and normalize one schema-v1 capacity observation.""" + if not isinstance(value, dict) or set(value) != OBSERVATION_FIELDS: + unknown = sorted(set(value) - OBSERVATION_FIELDS) if isinstance(value, dict) else [] + missing = sorted(OBSERVATION_FIELDS - set(value)) if isinstance(value, dict) else [] + raise ContractError( + f"capacity observation fields invalid: unknown={unknown} missing={missing}", + ) + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("capacity observation schema_version is invalid") + normalized = { + **value, + "observation_id": _identifier(value["observation_id"], "observation_id"), + "pool_id": _identifier(value["pool_id"], "pool_id"), + "window_id": _identifier(value["window_id"], "window_id"), + "applies_to": _selector(value["applies_to"]), + } + if not isinstance(value["source"], str) or value["source"] not in SOURCES: + raise ContractError("capacity source is invalid") + if not isinstance(value["unit"], str) or value["unit"] not in UNITS: + raise ContractError("capacity unit is invalid") + if not isinstance(value["confidence"], str) or value["confidence"] not in CONFIDENCE: + raise ContractError("capacity confidence is invalid") + normalized["used"] = _measurement(value["used"], "used") + normalized["limit"] = _measurement(value["limit"], "limit") + if value["unit"] == "percent" and any( + item is not None and item > 100 + for item in (normalized["used"], normalized["limit"]) + ): + raise ContractError("capacity percent measurements must be at most 100") + observed = _timestamp(value["observed_at"], "observed_at") + expires = _timestamp(value["expires_at"], "expires_at") + _timestamp(value["resets_at"], "resets_at", nullable=True) + if observed > expires: + raise ContractError("capacity observed_at must not be after expires_at") + current = _authoritative_now(now) + if observed > current + MAX_CLOCK_SKEW: + raise ContractError("capacity observed_at exceeds allowed clock skew") + # A canonical round trip proves every retained value is finite JSON and + # prevents callers from mutating the source object after validation. + return json.loads(canonical_json(normalized)) + + +def _matches(selector: dict[str, list[str]], target: dict[str, Any] | None) -> bool: + mapping = { + "harnesses": "harness", + "model_families": "model_family", + "model_ids": "model_id", + } + for selector_field, target_field in mapping.items(): + allowed = selector[selector_field] + if not allowed: + continue + if target is None or target.get(target_field) not in allowed: + return False + return True + + +def derive_pool_capacity( + pool_id: str, + observations: list[dict[str, Any]], + *, + target: dict[str, Any] | None = None, + in_flight: int = 0, + now: datetime | None = None, +) -> dict[str, Any]: + """Derive hard availability from the latest applicable observation per window.""" + _identifier(pool_id, "pool_id") + if not isinstance(observations, list): + raise ContractError("capacity observations must be an array") + if type(in_flight) is not int or in_flight < 0: + raise ContractError("capacity in_flight must be a non-negative integer") + if target is not None: + if not isinstance(target, dict) or any( + not isinstance(target.get(field), str) or not target[field] + for field in ("harness", "model_family", "model_id") + ): + raise ContractError("capacity target identity is invalid") + current = _authoritative_now(now) + latest: dict[tuple[str, str], tuple[datetime, str, dict[str, Any]]] = {} + seen_ids: set[str] = set() + for raw in observations: + observation = validate_observation(raw, now=current) + observation_id = observation["observation_id"] + if observation_id in seen_ids: + raise ContractError("capacity observation ids must be unique") + seen_ids.add(observation_id) + if observation["pool_id"] != pool_id or not _matches( + observation["applies_to"], target, + ): + continue + key = (observation["window_id"], canonical_json(observation["applies_to"])) + ordering = ( + _timestamp(observation["observed_at"], "observed_at"), + observation_id, + ) + prior = latest.get(key) + if prior is None or ordering[:2] > prior[:2]: + latest[key] = (ordering[0], ordering[1], observation) + + windows = [] + for key in sorted(latest): + observation = latest[key][2] + fresh = current <= _timestamp(observation["expires_at"], "expires_at") + authoritative = ( + observation["source"] != "estimated" + and observation["confidence"] != "estimated" + ) + if not fresh: + status, reason = "unknown", "stale_observation" + elif not authoritative: + status, reason = "unknown", "estimated_observation" + elif observation["used"] is None or observation["limit"] is None: + status, reason = "unknown", "unknown_measurement" + elif observation["used"] >= observation["limit"]: + status, reason = "exhausted", "window_exhausted" + else: + status, reason = "available", "window_available" + windows.append({ + "observation_id": observation["observation_id"], + "window_id": observation["window_id"], + "applies_to": observation["applies_to"], + "observed_at": observation["observed_at"], + "expires_at": observation["expires_at"], + "resets_at": observation["resets_at"], + "source": observation["source"], + "confidence": observation["confidence"], + "used": observation["used"], + "limit": observation["limit"], + "unit": observation["unit"], + "fresh": fresh, + "authoritative": authoritative, + "status": status, + "reason": reason, + }) + + if any(window["status"] == "exhausted" for window in windows): + status = "exhausted" + elif not windows or any(window["status"] == "unknown" for window in windows): + status = "unknown" + else: + status = "available" + observed_at = None + if windows: + observed_at = max( + windows, + key=lambda window: _timestamp(window["observed_at"], "observed_at"), + )["observed_at"] + reasons = [ + f"{window['reason']}:{window['window_id']}" + for window in windows + if window["status"] != "available" + ] + if not windows: + reasons = ["no_applicable_observations"] + return { + "schema_version": 1, + "pool_id": pool_id, + "status": status, + "in_flight": in_flight, + "observed_at": observed_at, + "evaluated_at": current.isoformat(), + "target": None if target is None else { + field: target[field] for field in ("harness", "model_family", "model_id") + }, + "windows": windows, + "reasons": reasons, + } diff --git a/plugin/core/src/devsquad/catalog.py b/plugin/core/src/devsquad/catalog.py new file mode 100644 index 0000000..bb5091d --- /dev/null +++ b/plugin/core/src/devsquad/catalog.py @@ -0,0 +1,220 @@ +"""Structured, last-good model catalog handling for M1.""" + +from __future__ import annotations + +import hashlib +import json +import os +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Iterable + +from .contracts import ContractError +from .store import canonical_json +from .validation import validate_profile + + +CATALOG_CHANGE_FIELDS = { + "schema_version", "harness", "previous_sha256", "current_sha256", + "added_model_ids", "removed_model_ids", "changed_model_ids", + "same_id_revision_unknown", "affected_profile_ids", + "unavailable_profile_ids", "profile_scope", "unqualified_candidate_ids", + "binding_changes_applied", +} + + +def validate_catalog_change(value: dict[str, Any]) -> dict[str, Any]: + """Validate complete-catalog drift evidence before lifecycle mutation.""" + if not isinstance(value, dict) or set(value) != CATALOG_CHANGE_FIELDS: + raise ContractError("catalog change fields are invalid") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("catalog change schema_version is invalid") + if not isinstance(value["harness"], str) or not value["harness"]: + raise ContractError("catalog change harness is invalid") + for field in ("previous_sha256", "current_sha256"): + digest = value[field] + if (digest is None and field == "previous_sha256"): + continue + if (not isinstance(digest, str) or len(digest) != 64 + or any(character not in "0123456789abcdef" for character in digest)): + raise ContractError(f"catalog change {field} is invalid") + list_fields = CATALOG_CHANGE_FIELDS - { + "schema_version", "harness", "previous_sha256", "current_sha256", + "profile_scope", "binding_changes_applied", + } + normalized = dict(value) + for field in list_fields: + items = value[field] + if (not isinstance(items, list) + or any(not isinstance(item, str) or not item for item in items) + or len(items) != len(set(items))): + raise ContractError(f"catalog change {field} is invalid") + normalized[field] = sorted(items) + if value["profile_scope"] not in {"provided", "unavailable"}: + raise ContractError("catalog change profile_scope is invalid") + if value["binding_changes_applied"] is not False: + raise ContractError("catalog discovery cannot apply binding changes") + if set(normalized["same_id_revision_unknown"]) - set(normalized["changed_model_ids"]): + raise ContractError("catalog revision uncertainty is not changed-model scoped") + if normalized["unqualified_candidate_ids"] != normalized["added_model_ids"]: + raise ContractError("catalog additions must remain unqualified candidates") + if set(normalized["unavailable_profile_ids"]) - set(normalized["affected_profile_ids"]): + raise ContractError("catalog unavailable profiles must be affected") + if (value["profile_scope"] == "unavailable" + and (normalized["affected_profile_ids"] + or normalized["unavailable_profile_ids"])): + raise ContractError("catalog change cannot infer profiles without scope") + return json.loads(canonical_json(normalized)) + + +def model_fingerprint(harness: str, version: str | None, model: dict[str, Any]) -> str: + # Provider ordering/default hints are not capability or serving revisions. + # Their movement must not invalidate an explicitly approved alias. + capabilities = {key: value for key, value in model.items() if key not in {"isDefault", "is_default"}} + stable = {"harness": harness, "version": version, "model": capabilities} + return hashlib.sha256(json.dumps(stable, sort_keys=True, separators=(",", ":")).encode()).hexdigest() + + +def normalize_models(harness: str, version: str | None, models: Iterable[dict[str, Any]]) -> list[dict[str, Any]]: + normalized = [] + identifiers = set() + for raw in models: + if not isinstance(raw, dict): + raise ContractError("catalog model must be an object") + model_id = raw.get("id") or raw.get("model") + if not isinstance(model_id, str) or not model_id: + raise ContractError("catalog model missing id") + if model_id in identifiers: + raise ContractError("catalog model ids must be unique") + identifiers.add(model_id) + effort_values = raw.get("supportedReasoningEfforts") or raw.get("supported_reasoning_efforts") or [] + efforts = [item.get("reasoningEffort") if isinstance(item, dict) else item for item in effort_values] + if not all(isinstance(effort, str) and effort for effort in efforts): + raise ContractError("catalog model effort metadata is invalid") + normalized.append({ + "id": model_id, + "display_name": raw.get("displayName") or raw.get("display_name") or model_id, + "family": raw.get("family"), + "default_effort": raw.get("defaultReasoningEffort") or raw.get("default_reasoning_effort"), + "supported_efforts": efforts, + "modalities": raw.get("inputModalities") or raw.get("input_modalities") or [], + "revision": raw.get("modelRevision") or raw.get("revision"), + "is_default": bool(raw.get("isDefault") or raw.get("is_default")), + "qualification": "unqualified", + "fingerprint": model_fingerprint(harness, version, raw), + }) + return normalized + + +def analyze_catalog_drift( + previous: dict[str, Any] | None, + current: dict[str, Any], + *, + profiles: Iterable[dict[str, Any]] | None = None, +) -> dict[str, Any]: + """Identify exact model/profile drift without changing any binding.""" + if (not isinstance(current, dict) or current.get("complete") is not True + or not isinstance(current.get("harness"), str) + or not isinstance(current.get("models"), list)): + raise ContractError("current catalog snapshot is invalid") + if previous is not None and ( + not isinstance(previous, dict) or previous.get("complete") is not True + or previous.get("harness") != current["harness"] + or not isinstance(previous.get("models"), list)): + raise ContractError("previous catalog snapshot is invalid") + + def models_by_id(snapshot): + result = {} + for model in snapshot.get("models", []): + if (not isinstance(model, dict) + or not isinstance(model.get("id"), str) + or model["id"] in result): + raise ContractError("catalog snapshot model identity is invalid") + result[model["id"]] = model + return result + + old_models = {} if previous is None else models_by_id(previous) + new_models = models_by_id(current) + added = sorted(set(new_models) - set(old_models)) + removed = sorted(set(old_models) - set(new_models)) + changed = sorted( + model_id for model_id in set(old_models) & set(new_models) + if old_models[model_id].get("fingerprint") + != new_models[model_id].get("fingerprint") + ) + unknown_revision = sorted( + model_id for model_id in changed + if (old_models[model_id].get("revision") is None + or new_models[model_id].get("revision") is None) + ) + profile_rows = [] if profiles is None else list(profiles) + for profile in profile_rows: + validate_profile(profile) + affected_ids = set(changed) | set(removed) + affected_profiles = sorted( + profile["id"] for profile in profile_rows + if (profile["harness"] == current["harness"] + and profile["model_id"] in affected_ids) + ) + unavailable_profiles = sorted( + profile["id"] for profile in profile_rows + if (profile["harness"] == current["harness"] + and profile["model_id"] in removed) + ) + previous_sha256 = ( + hashlib.sha256(canonical_json(previous).encode()).hexdigest() + if previous is not None else None + ) + return validate_catalog_change({ + "schema_version": 1, + "harness": current["harness"], + "previous_sha256": previous_sha256, + "current_sha256": hashlib.sha256( + canonical_json(current).encode(), + ).hexdigest(), + "added_model_ids": added, + "removed_model_ids": removed, + "changed_model_ids": changed, + "same_id_revision_unknown": unknown_revision, + "affected_profile_ids": affected_profiles, + "unavailable_profile_ids": unavailable_profiles, + "profile_scope": "provided" if profiles is not None else "unavailable", + "unqualified_candidate_ids": added, + "binding_changes_applied": False, + }) + + +def update_last_good(path: Path, *, harness: str, version: str | None, models: Iterable[dict[str, Any]] | None, complete: bool, error: str | None = None, profiles: Iterable[dict[str, Any]] | None = None) -> dict[str, Any]: + old = json.loads(path.read_text()) if path.exists() else None + now = datetime.now(timezone.utc).isoformat() + if error or not complete or models is None: + if old: + old["last_refresh"] = {"at": now, "status": "error" if error else "incomplete", "error": error} + _atomic_json(path, old) + return old + raise ContractError(error or "catalog response incomplete and no last-good snapshot exists") + value = {"schema_version": 1, "harness": harness, "harness_version": version, "fetched_at": now, "complete": True, "models": normalize_models(harness, version, models), "last_refresh": {"at": now, "status": "ok", "error": None}} + value["catalog_change"] = analyze_catalog_drift( + old, value, profiles=profiles, + ) + _atomic_json(path, value) + return value + + +def _atomic_json(path: Path, value: dict[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + tmp = path.with_name(f"{path.name}.tmp.{os.getpid()}") + tmp.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") + os.replace(tmp, path) + + +def verified_efforts(snapshot: dict[str, Any], *, harness: str, version: str, model_id: str) -> tuple[str, ...]: + if snapshot.get("complete") is not True or snapshot.get("harness") != harness or snapshot.get("harness_version") != version: + raise ContractError("catalog snapshot does not verify this harness version") + matches = [m for m in snapshot.get("models", []) if m.get("id") == model_id] + if len(matches) != 1: + raise ContractError(f"model is not uniquely present in verified catalog: {model_id}") + efforts = matches[0].get("supported_efforts") + if not isinstance(efforts, list) or not all(isinstance(v, str) for v in efforts): + raise ContractError("model effort metadata is unknown") + return tuple(efforts) diff --git a/plugin/core/src/devsquad/claude_delivery_worker.py b/plugin/core/src/devsquad/claude_delivery_worker.py new file mode 100644 index 0000000..28d0e80 --- /dev/null +++ b/plugin/core/src/devsquad/claude_delivery_worker.py @@ -0,0 +1,269 @@ +"""Run one frozen Claude CLI implementation inside the durable writer fence.""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +from typing import Any + +from .contracts import CapabilityUnavailable, ContractError, ProfileUnsupported +from .claude_identity import ( + ClaudeResultError, failure_diagnostics, failure_envelope, native_result, + observed_identity, strict_json, +) +from .store import canonical_json +from .workflows import build_implementation_prompt, make_implementation_evidence + + +MAX_SNAPSHOT_BYTES = 2 * 1024 * 1024 +MAX_CLAUDE_OUTPUT_BYTES = 2 * 1024 * 1024 +ADAPTER_FIELDS = { + "schema_version", "harness", "transport", "binary", "binary_sha256", + "harness_version", "model_provider", "permission_args", "error_patterns", + "denied_pattern", +} + + +def _manifest_path() -> Path: + source = Path(__file__).resolve().parents[2] / "adapters" / "claude" / "adapter.json" + if source.is_file(): + return source + installed = ( + Path(sys.prefix) / "share" / "devsquad" / "adapters" / "claude" + / "adapter.json" + ) + if installed.is_file(): + return installed + raise CapabilityUnavailable("Claude adapter manifest is unavailable") + + +def freeze_claude_implementer(selected: dict[str, Any]) -> dict[str, Any]: + """Resolve one exact subscription Claude writer during run preflight.""" + from .adapters import ( + AdapterManifest, + DENIED_PATTERN, + ERROR_PATTERNS, + harness_version, + ) + if not isinstance(selected, dict) or not isinstance(selected.get("profile"), dict): + raise ContractError("frozen implementer selection is invalid") + profile = selected["profile"] + if profile.get("harness") != "claude": + raise CapabilityUnavailable( + f"selected implementer harness is not Claude: {profile.get('harness')}" + ) + if profile.get("permission_policy") != "workspace_write": + raise ProfileUnsupported("Claude implementer requires workspace_write") + if set(profile.get("required_tools", [])) - {"read", "write"}: + raise ProfileUnsupported("Claude implementer requests unsupported tools") + effort = profile.get("effort") + if (not isinstance(effort, dict) or effort.get("transport") != "native" + or not isinstance(effort.get("value"), str) or not effort["value"]): + raise ProfileUnsupported("Claude implementer requires an explicit effort") + if not isinstance(profile.get("model_id"), str) or not profile["model_id"]: + raise ProfileUnsupported("Claude implementer requires an exact model id") + + manifest = AdapterManifest.load(_manifest_path()) + binary_name = manifest.resolve_binary() + if not binary_name: + raise CapabilityUnavailable("Claude executable is unavailable") + binary = Path(binary_name).resolve(strict=True) + version = harness_version(str(binary)) + if not version: + raise CapabilityUnavailable("Claude version could not be observed") + if version not in manifest.verified_versions: + raise ProfileUnsupported(f"unverified Claude CLI version: {version}") + return { + "schema_version": 1, + "harness": "claude", + "transport": "cli_exec", + "binary": str(binary), + "binary_sha256": hashlib.sha256(binary.read_bytes()).hexdigest(), + "harness_version": version, + "model_provider": manifest.model_provider or "anthropic", + "permission_args": list(manifest.permission_profiles["workspace_write"]), + "error_patterns": { + code: pattern.pattern for code, pattern in ERROR_PATTERNS + }, + "denied_pattern": DENIED_PATTERN.pattern, + } + + +def _validated(snapshot: dict[str, Any]) -> tuple[dict[str, Any], dict[str, Any]]: + adapter = snapshot.get("implementation_adapter") + try: + profile = snapshot["routing"]["roles"]["implementer"]["selected"]["profile"] + except (KeyError, TypeError) as exc: + raise ContractError("frozen Claude implementer selection is missing") from exc + if not isinstance(adapter, dict) or set(adapter) != ADAPTER_FIELDS: + raise ContractError("frozen Claude implementer adapter fields are invalid") + if (adapter["schema_version"] != 1 + or adapter["harness"] != "claude" + or adapter["transport"] != "cli_exec" + or adapter["model_provider"] != "anthropic"): + raise ContractError("frozen Claude implementer adapter identity is invalid") + if (not isinstance(adapter["permission_args"], list) + or adapter["permission_args"] != [ + "--permission-mode", "acceptEdits", "--tools", + "Read,Glob,Grep,Edit,Write", + ] + or not isinstance(adapter["error_patterns"], dict) + or set(adapter["error_patterns"]) != {"AUTH_ERROR", "RATE_LIMITED"} + or not all( + isinstance(value, str) and value + for value in adapter["error_patterns"].values() + ) + or not isinstance(adapter["denied_pattern"], str) + or not adapter["denied_pattern"]): + raise ContractError("frozen Claude implementer policy is invalid") + if (not isinstance(profile, dict) or profile.get("harness") != "claude" + or profile.get("permission_policy") != "workspace_write"): + raise ContractError("frozen profile is not a Claude implementation writer") + return adapter, profile + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("delivery snapshot must be an object") + adapter, profile = _validated(snapshot) + binary = Path(adapter["binary"]) + try: + resolved = binary.resolve(strict=True) + except OSError as exc: + raise CapabilityUnavailable("frozen Claude executable is missing") from exc + if (resolved != binary + or hashlib.sha256(binary.read_bytes()).hexdigest() + != adapter["binary_sha256"]): + raise CapabilityUnavailable("frozen Claude executable changed after preflight") + try: + version = subprocess.run( + [str(binary), "--version"], text=True, capture_output=True, + timeout=3, check=False, + ) + except (OSError, subprocess.TimeoutExpired, UnicodeError) as exc: + raise CapabilityUnavailable("Claude version could not be re-observed") from exc + if version.returncode != 0 or version.stdout.strip() != adapter["harness_version"]: + raise CapabilityUnavailable("Claude version changed after preflight") + + workspace = Path(snapshot["delivery_workspace"]["path"]).resolve(strict=True) + effort = profile["effort"]["value"] + model = profile["model_id"] + prompt = build_implementation_prompt( + snapshot["task"], snapshot["delivery_workspace"], + snapshot.get("revision_request"), + ) + argv = [ + str(binary), "--print", "--output-format", "stream-json", "--verbose", "--safe-mode", + "--disable-slash-commands", "--no-session-persistence", + "--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}', + "--no-chrome", "--model", model, "--effort", effort, + *adapter["permission_args"], "--", prompt, + ] + timeout_seconds = snapshot["task"]["budget"]["wall_seconds"] + try: + completed = subprocess.run( + argv, + cwd=workspace, + env={**os.environ, "DEVSQUAD_WORKER": "1"}, + text=False, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + timeout=timeout_seconds, + check=False, + ) + timed_out = False + except subprocess.TimeoutExpired as exc: + completed = subprocess.CompletedProcess( + argv, 124, exc.stdout or "", exc.stderr or "", + ) + timed_out = True + # Preserve exact native bytes, including invalid UTF-8 and CRLF. Only + # stderr classification uses replacement decoding; stdout is parsed strictly. + stdout = completed.stdout + stderr = completed.stderr.decode("utf-8", "replace") if isinstance( + completed.stderr, bytes + ) else completed.stderr + if (len(stdout.encode() if isinstance(stdout, str) else stdout) > MAX_CLAUDE_OUTPUT_BYTES + or len(stderr.encode()) > MAX_CLAUDE_OUTPUT_BYTES): + raise ClaudeResultError("CLI_ERROR", failure_diagnostics( + stdout, None, "output_limit", + )) + import re + error_code = next(( + code for code, pattern in adapter["error_patterns"].items() + if re.search(pattern, stderr, re.IGNORECASE) + ), None) + if timed_out: + error_code = "TIMEOUT" + provider_document = None + try: + from .claude_identity import decode_native_result + provider_document, writer_messages = decode_native_result(stdout) + summary, native = native_result(provider_document, writer_messages) + observed = observed_identity(native, adapter, profile) + except ContractError as exc: + safe_document = provider_document if isinstance(provider_document, dict) else {} + provider_text = str( + safe_document.get("result") or safe_document.get("error") or "" + ) + native_code = next(( + code for code, pattern in adapter["error_patterns"].items() + if re.search(pattern, provider_text, re.IGNORECASE) + ), None) + if (safe_document.get("is_error") is True + and re.search(r"\bnot logged in\b", provider_text, re.IGNORECASE)): + native_code = "AUTH_ERROR" + if error_code != "TIMEOUT": + reported_codes = {error_code, native_code} + error_code = next((code for code in ("AUTH_ERROR", "RATE_LIMITED") + if code in reported_codes), error_code or native_code) + if error_code is None and re.search( + adapter["denied_pattern"], provider_text, re.IGNORECASE, + ): + error_code = "CLI_ERROR" + raise ClaudeResultError(error_code or "CLI_ERROR", failure_diagnostics( + stdout, provider_document, "native_result_invalid", + )) from exc + if completed.returncode != 0 and error_code is None: + error_code = "CLI_ERROR" + if error_code is not None: + raise ClaudeResultError(error_code, failure_diagnostics( + stdout, provider_document, "execution_failed", + )) + return make_implementation_evidence( + snapshot, + summary, + observed_identity=observed, + native_ids={"session_id": native["session_id"]}, + native_model_requests=None, + usage=native["usage"], + ) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_SNAPSHOT_BYTES + 1) + if len(payload) > MAX_SNAPSHOT_BYTES: + raise ContractError("delivery snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("delivery snapshot is not valid UTF-8 JSON") from exc + try: + result = run(snapshot) + except ClaudeResultError as exc: + selected = snapshot["routing"]["roles"]["implementer"]["selected"] + sys.stdout.write(canonical_json(failure_envelope( + exc, selected["profile_sha256"], + )) + "\n") + sys.stderr.write(str(exc) + "\n") + return 1 + sys.stdout.write(canonical_json(result) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/claude_identity.py b/plugin/core/src/devsquad/claude_identity.py new file mode 100644 index 0000000..e0c60c0 --- /dev/null +++ b/plugin/core/src/devsquad/claude_identity.py @@ -0,0 +1,343 @@ +"""Bounded Claude result evidence; requested settings are never observations. + +Legacy JSON results need a single concrete usage model. Native streams can +identify a unique writer through correlated assistant-message model reports, +while retaining separately reported auxiliary usage without guessing its role. +Pricing aliases are metadata. Effective effort and serving revision are unknown. +""" + +from __future__ import annotations + +import hashlib +import json +import math +import re +from typing import Any + +from .contracts import ContractError + + +MAX_OUTPUT_BYTES = 2 * 1024 * 1024 +MAX_COUNTER = 2 ** 63 - 1 +_MODEL = re.compile(r"[A-Za-z0-9][A-Za-z0-9._:/-]{0,255}\Z") +_SESSION = re.compile(r"[A-Za-z0-9][A-Za-z0-9._:-]{0,499}\Z") +_COUNTERS = { + "inputTokens", "outputTokens", "cacheReadInputTokens", + "cacheCreationInputTokens", "webSearchRequests", "contextWindow", + "maxOutputTokens", +} +_ALIASES = {"sonnet", "opus", "haiku"} +_NATIVE_FIELDS = { + "schema_version", "type", "subtype", "is_error", "session_id", + "model_usage", "top_level_model", "usage", +} +ERROR_CODES = {"AUTH_ERROR", "RATE_LIMITED", "TIMEOUT", "CLI_ERROR"} + + +def _exact(value: Any, fields: set[str]) -> dict[str, Any]: + if not isinstance(value, dict) or set(value) != fields: + raise ContractError("Claude evidence fields are invalid") + return value + + +def _model(value: Any) -> str: + if not isinstance(value, str) or not _MODEL.fullmatch(value): + raise ContractError("Claude reported model is invalid") + return value + + +def _session(value: Any) -> str: + if not isinstance(value, str) or not _SESSION.fullmatch(value): + raise ContractError("Claude result session is invalid") + return value + + +def _counter(value: Any) -> int: + if type(value) is not int or not 0 <= value <= MAX_COUNTER: + raise ContractError("Claude reported counter is invalid") + return value + + +def strict_json(payload: bytes | str) -> Any: + def pairs(items): + result = {} + for key, value in items: + if key in result: + raise ValueError("duplicate key") + result[key] = value + return result + + def nonfinite(value): + raise ValueError("non-finite number") + + def finite_float(value): + result = float(value) + if not math.isfinite(result): + raise ValueError("non-finite number") + return result + + try: + encoded = payload.encode("utf-8") if isinstance(payload, str) else payload + if len(encoded) > MAX_OUTPUT_BYTES: + raise ValueError("output limit") + return json.loads(encoded.decode("utf-8"), object_pairs_hook=pairs, + parse_constant=nonfinite, parse_float=finite_float) + except (UnicodeError, ValueError, TypeError, RecursionError) as exc: + raise ContractError("Claude result is not bounded strict UTF-8 JSON") from exc + + +def unknown_usage() -> dict[str, Any]: + return {"input_tokens": None, "output_tokens": None, + "total_tokens": None, "source": "unavailable"} + + +def reported_usage(raw: Any) -> dict[str, Any]: + if not isinstance(raw, dict): + return unknown_usage() + try: + incoming = _counter(raw.get("input_tokens")) + outgoing = _counter(raw.get("output_tokens")) + total = _counter(incoming + outgoing) + except ContractError: + return unknown_usage() + return {"input_tokens": incoming, "output_tokens": outgoing, + "total_tokens": total, "source": "native_reported"} + + +def _usage_evidence(value: Any) -> dict[str, Any]: + _exact(value, set(unknown_usage())) + expected = reported_usage(value) + if expected != value: + raise ContractError("Claude usage evidence is inconsistent") + return expected + + +def model_usage(raw: Any) -> dict[str, Any]: + if not isinstance(raw, dict) or not 1 <= len(raw) <= 32: + raise ContractError("Claude result has no bounded model usage") + result = {} + for model, entry in raw.items(): + _model(model) + if not isinstance(entry, dict): + raise ContractError("Claude model usage is invalid") + normalized = {key: _counter(entry[key]) for key in + ("inputTokens", "outputTokens") if key in entry} + if len(normalized) != 2: + raise ContractError("Claude model usage counters are missing") + for key in _COUNTERS & entry.keys(): + normalized[key] = _counter(entry[key]) + if "costUSD" in entry: + cost = entry["costUSD"] + if (type(cost) not in (int, float) or not 0 <= cost <= MAX_COUNTER + or not math.isfinite(cost)): + raise ContractError("Claude model cost is invalid") + normalized["costUSD"] = cost + for key in ("canonicalModel", "provider"): + if key in entry: + normalized[key] = _model(entry[key]) + result[model] = normalized + return result + + +def decode_native_result(payload: bytes | str) -> tuple[dict[str, Any], dict[str, Any] | None]: + """Decode strict legacy JSON or a bounded, session-correlated native stream.""" + try: + document = strict_json(payload) + except ContractError: + encoded = payload.encode("utf-8") if isinstance(payload, str) else payload + if not isinstance(encoded, bytes) or len(encoded) > MAX_OUTPUT_BYTES: + raise ContractError("Claude stream exceeds its bound") + records = [strict_json(line) for line in encoded.splitlines() if line.strip()] + if not 1 <= len(records) <= 8192 or any(not isinstance(item, dict) for item in records): + raise ContractError("Claude stream records are invalid") + terminals = [item for item in records if item.get("type") == "result"] + if len(terminals) != 1 or records[-1] is not terminals[0]: + raise ContractError("Claude stream requires one final result") + document = terminals[0] + session = _session(document.get("session_id")) + # Native authentication failures emit a synthetic assistant model. + # Retain the error terminal for classification, never writer identity. + if document.get("is_error") is True: + return document, None + models = [] + for item in records: + if item.get("type") != "assistant": + continue + message = item.get("message") + if (item.get("session_id") != session or item.get("parent_tool_use_id") is not None + or not isinstance(message, dict) or message.get("role") != "assistant"): + raise ContractError("Claude writer message is not session-correlated") + models.append(_model(message.get("model"))) + if not models: + raise ContractError("Claude stream has no reported writer messages") + return document, {"session_id": session, "models": sorted(set(models)), "message_count": len(models)} + if not isinstance(document, dict): + raise ContractError("Claude result must be an object") + return document, None + + +def native_result(document: Any, writer_messages: dict[str, Any] | None = None) -> tuple[str, dict[str, Any]]: + if (not isinstance(document, dict) or document.get("type") != "result" + or document.get("subtype") != "success" + or document.get("is_error") is not False + or not isinstance(document.get("result"), str) + or not document["result"].strip()): + raise ContractError("Claude implementation has no successful result") + summary = document["result"].strip() + if len(summary) > 20_000: + raise ContractError("Claude summary exceeds its limit") + return summary, { + "schema_version": 1 if writer_messages is None else 2, "type": "result", "subtype": "success", + "is_error": False, "session_id": _session(document.get("session_id")), + "model_usage": model_usage(document.get("modelUsage")), + "top_level_model": _model(document["model"]) if "model" in document else None, + "usage": reported_usage(document.get("usage")), + **({"writer_messages": writer_messages} if writer_messages is not None else {}), + } + + +def observed_identity(native: Any, adapter: dict[str, Any], + profile: dict[str, Any]) -> dict[str, Any]: + if not isinstance(native, dict): + raise ContractError("Claude native identity envelope is invalid") + version = native.get("schema_version") + _exact(native, _NATIVE_FIELDS | ({"writer_messages"} if version == 2 else set())) + if (type(version) is not int or version not in {1, 2} + or native["type"] != "result" or native["subtype"] != "success" + or native["is_error"] is not False): + raise ContractError("Claude native identity envelope is invalid") + _session(native["session_id"]) + _usage_evidence(native["usage"]) + models = model_usage(native["model_usage"]) + if models != native["model_usage"]: + raise ContractError("Claude native model usage is inconsistent") + if version == 1: + if len(models) != 1: + raise ContractError("Claude result cannot identify a unique writer model") + model = next(iter(models)) + model_source = "claude.result.modelUsage" + else: + messages = _exact(native["writer_messages"], {"session_id", "models", "message_count"}) + if (messages["session_id"] != native["session_id"] + or type(messages["message_count"]) is not int or not 1 <= messages["message_count"] <= 8192 + or not isinstance(messages["models"], list) or len(messages["models"]) != 1): + raise ContractError("Claude stream cannot identify a unique correlated writer model") + model = _model(messages["models"][0]) + if model not in models: + raise ContractError("Claude stream writer is missing from terminal model usage") + model_source = "claude.stream.assistant.message.model" + if model in _ALIASES: + raise ContractError("Claude reported identity is an unresolved alias") + requested = _model(profile.get("model_id")) + alias = requested in _ALIASES + if (alias and not model.startswith(f"claude-{requested}-")) or ( + not alias and requested != model): + raise ContractError("Claude reported model does not match the requested profile") + top = native["top_level_model"] + if top is not None and _model(top) != model: + raise ContractError("Claude result contains contradictory model identity") + if (adapter.get("harness") != "claude" + or adapter.get("model_provider") != "anthropic" + or not isinstance(adapter.get("harness_version"), str) + or not adapter["harness_version"] + or profile.get("harness") != "claude" + or profile.get("permission_policy") != "workspace_write"): + raise ContractError("Claude identity does not match the frozen adapter") + # Native 2.1.220 uses a transport label, not a model-provider ID. Retain + # the raw field and map only the label observed under this verified CLI. + allowed_providers = {adapter["model_provider"]} + if adapter["harness_version"] == "2.1.220 (Claude Code)": + allowed_providers.add("firstParty") + if any(entry.get("provider", adapter["model_provider"]) not in allowed_providers for entry in models.values()): + raise ContractError("Claude reported provider contradicts the frozen adapter") + return { + "harness": "claude", "harness_version": adapter["harness_version"], + "model_provider": adapter["model_provider"], "model_id": model, + "effort": None, "backing_revision": None, + "permission_policy": "workspace_write", "verification": "verified", + "verification_scope": "reported_model", + "model_source": model_source, + "alias_resolution": {"requested": requested, "reported": model} if alias else None, + "native_evidence": native, + } + + +def validate_observation(value: Any, adapter: dict[str, Any], + profile: dict[str, Any], ids: Any, usage: Any) -> None: + if not isinstance(value, dict): + raise ContractError("Claude observed identity is missing") + native = value.get("native_evidence") + expected = observed_identity(native, adapter, profile) + # Canonical JSON comparison distinguishes booleans from integers. + if json.dumps(value, sort_keys=True) != json.dumps(expected, sort_keys=True): + raise ContractError("Claude observed identity differs from native evidence") + if ids != {"session_id": native["session_id"]} or usage != native["usage"]: + raise ContractError("Claude session or usage differs from native evidence") + + +def failure_diagnostics(payload: bytes | str, document: Any, reason: str) -> dict[str, Any]: + """Keep only bounded typed native fields, never provider prose or stderr.""" + document = document if isinstance(document, dict) else {} + session = document.get("session_id") + try: + _session(session) + except ContractError: + session = None + models = {} + raw = document.get("modelUsage") + if isinstance(raw, dict) and len(raw) <= 32: + for key, value in raw.items(): + try: + models.update(model_usage({key: value})) + except (ContractError, OverflowError): + continue + encoded = payload.encode("utf-8") if isinstance(payload, str) else payload + return { + "schema_version": 1, "identity_status": "unverified", "reason": reason, + "output_sha256": hashlib.sha256(encoded).hexdigest(), + "output_bytes": len(encoded), "session_id": session, + "model_usage": models, "usage": reported_usage(document.get("usage")), + } + + +class ClaudeResultError(ContractError): + def __init__(self, code: str, diagnostics: dict[str, Any]): + super().__init__(f"{code}: Claude implementation failed") + self.code = code + self.diagnostics = diagnostics + + +def failure_envelope(error: ClaudeResultError, profile_sha256: str) -> dict[str, Any]: + return {"schema_version": 1, "type": "claude_implementation_failure", + "profile_sha256": profile_sha256, "error": error.code, + "native_diagnostics": error.diagnostics} + + +def validate_failure(value: Any, profile_sha256: str) -> dict[str, Any]: + _exact(value, {"schema_version", "type", "profile_sha256", "error", + "native_diagnostics"}) + if (type(value["schema_version"]) is not int or value["schema_version"] != 1 + or value["type"] != "claude_implementation_failure" + or value["profile_sha256"] != profile_sha256 + or not isinstance(value["error"], str) + or value["error"] not in ERROR_CODES): + raise ContractError("Claude failed attempt identity is invalid") + data = _exact(value["native_diagnostics"], { + "schema_version", "identity_status", "reason", "output_sha256", + "output_bytes", "session_id", "model_usage", "usage", + }) + if (type(data["schema_version"]) is not int or data["schema_version"] != 1 + or data["identity_status"] != "unverified" + or not isinstance(data["reason"], str) + or data["reason"] not in {"native_result_invalid", "execution_failed", "output_limit"} + or not isinstance(data["output_sha256"], str) + or not re.fullmatch(r"[0-9a-f]{64}", data["output_sha256"])): + raise ContractError("Claude failure diagnostics are invalid") + _counter(data["output_bytes"]) + if data["session_id"] is not None: + _session(data["session_id"]) + if data["model_usage"] != {}: + if model_usage(data["model_usage"]) != data["model_usage"]: + raise ContractError("Claude failed model usage is invalid") + _usage_evidence(data["usage"]) + return value diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py new file mode 100644 index 0000000..ea9b622 --- /dev/null +++ b/plugin/core/src/devsquad/cli.py @@ -0,0 +1,930 @@ +"""Versioned JSON command surface for local DevSquad operations.""" + +from __future__ import annotations + +import argparse +import json +import os +import shlex +import sys +import time +from pathlib import Path +from typing import Any + +from . import __version__ +from .adapters import AdapterManifest, classify_cli, harness_version, prepare_cli, prepare_native_codex_from_catalog +from .contracts import ContractError, envelope, error_payload +from .diagnostics import build_doctor_report +from .integrations import LocalIntegrationManager, load_integrations +from .service import Service +from .store import ConflictError, SchemaVersionError +from .task_entry import ( + build_managed_task, + discover_codex_identity, + parse_checks, + resolve_repository, +) + +SOURCE_ROOT = Path(__file__).resolve().parents[2] +CORE_ROOT = SOURCE_ROOT if (SOURCE_ROOT / "adapters").is_dir() else Path(sys.prefix) / "share" / "devsquad" +WAIT_POLL_SECONDS = 0.25 +WAIT_EXIT_CODES = { + "succeeded": 0, + "blocked": 2, + "awaiting_host": 2, + "failed": 3, + "cancelled": 4, +} +WAIT_ACTIVE_STATES = {"queued", "running", "cancelling"} +HUMAN_COMMANDS = {"review", "fix", "council", "council-finish", "status", "result", "doctor", "setup", "finish", "resume", "cancel"} + + +def manifests() -> list[tuple[Path, AdapterManifest]]: + return [(path, AdapterManifest.load(path)) for path in sorted((CORE_ROOT / "adapters").glob("*/adapter.json"))] + + +def command_doctor(args: argparse.Namespace) -> tuple[dict, int]: + report = build_doctor_report( + project=Path(args.project_dir), + squad_executable=( + Path(args.squad_executable) if args.squad_executable else None + ), + ) + return envelope(data=report), 0 if report["ready"] else 1 + + +def command_setup(args: argparse.Namespace) -> tuple[dict, int]: + templates = load_integrations() + selected = set(args.host or (template.id for template in templates)) + manager = LocalIntegrationManager( + project=Path(args.project_dir), + squad_executable=( + Path(args.squad_executable) if args.squad_executable else None + ), + ) + rows = [ + manager.setup(template, dry_run=args.dry_run) + for template in templates + if template.id in selected + ] + successful_actions = { + "added", "updated", "unchanged", "would_add", "would_update", + } + completed = all(row["action"] in successful_actions for row in rows) + ready = all(row["ready"] for row in rows) + return envelope(data={ + "completed": completed, + "ready": ready, + "dry_run": args.dry_run, + "hosts": rows, + }), 0 if completed else 1 + + +def _read_json(path: str, label: str) -> Any: + def object_pairs(pairs): + value = {} + for key, item in pairs: + if key in value: + raise ValueError(f"duplicate key: {key}") + value[key] = item + return value + + def reject_constant(value): + raise ValueError(f"non-finite number: {value}") + + try: + return json.loads( + Path(path).read_text(), + object_pairs_hook=object_pairs, + parse_constant=reject_constant, + ) + except (OSError, UnicodeError, json.JSONDecodeError, ValueError) as exc: + raise ContractError(f"cannot read {label}: {exc}") from exc + + +def command_prepare(args: argparse.Namespace) -> tuple[dict, int]: + manifest = AdapterManifest.load(CORE_ROOT / "adapters" / args.adapter / "adapter.json") + transport = args.transport or manifest.transport + if transport == "native_protocol": + if manifest.name != "codex" or not args.catalog_file or not args.model or not args.effort: + raise ContractError("native preparation requires Codex, --catalog-file, --model and --effort") + binary = manifest.resolve_binary() + version = harness_version(binary) if binary else None + snapshot = _read_json(args.catalog_file, "catalog file") + spec = prepare_native_codex_from_catalog(manifest, snapshot, cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout, harness_version_value=version or "unknown") + elif transport == "cli_exec": + spec = prepare_cli(manifest, prompt=args.prompt, cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) + else: + raise ContractError(f"unsupported transport: {transport}") + return envelope(data=spec.to_dict()), 0 + + +def command_classify(args: argparse.Namespace) -> tuple[dict, int]: + manifest = AdapterManifest.load(CORE_ROOT / "adapters" / args.adapter / "adapter.json") + spec = prepare_cli(manifest, prompt="classification", cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) + try: + stdout = Path(args.stdout_file).read_text() + stderr = Path(args.stderr_file).read_text() + except (OSError, UnicodeError) as exc: + raise ContractError(f"cannot read classification output: {exc}") from exc + result = classify_cli(spec, returncode=args.returncode, stdout=stdout, stderr=stderr) + return envelope(data=result.to_dict()), 0 + + +def _service(args: argparse.Namespace) -> Service: + return Service(Path(args.runtime_dir)) + + +def _wait_for_run( + service: Service, + started: dict[str, Any], + *, + resume_candidate_review: bool = False, +) -> tuple[dict[str, Any], int]: + run_id = started["run_id"] + resumed_versions: set[int] = set() + try: + while True: + status = service.status(run_id) + state = status.get("state") + version = status.get("version") + if (state == "awaiting_host" + and status.get("next_action") == "continue_headless_lead"): + if type(version) is int and version not in resumed_versions: + resumed_versions.add(version) + try: + service.resume(run_id) + except ConflictError: + # The detached owner may have continued the same + # headless handoff between observation and resume. + if service.status(run_id).get("version") == version: + raise + time.sleep(WAIT_POLL_SECONDS) + continue + if state in WAIT_EXIT_CODES: + return envelope(data=status), WAIT_EXIT_CODES[state] + if state not in WAIT_ACTIVE_STATES: + raise RuntimeError(f"service returned unsupported run state: {state!r}") + if (resume_candidate_review + and state == "queued" + and status.get("next_action") == "resume_candidate_review" + and type(version) is int + and version not in resumed_versions): + resumed_versions.add(version) + service.resume(run_id) + time.sleep(WAIT_POLL_SECONDS) + except KeyboardInterrupt: + cancel_command = f"squad cancel {run_id}" + print( + f"Stopped observing run {run_id}; the run was not cancelled and remains saved. " + f"To cancel it explicitly, run: {cancel_command}", + file=sys.stderr, + ) + return envelope(data={ + "run_id": run_id, + "state": started.get("state"), + "observation_stopped": True, + "cancelled": False, + "next_action": cancel_command, + }), 130 + + +def command_start(args: argparse.Namespace) -> tuple[dict, int]: + task = _read_json(args.task_file, "task file") + service = _service(args) + started = service.start(task, args.idempotency_key, args.supersedes_run) + if not args.wait: + return envelope(data=started), 0 + return _wait_for_run(service, started) + + +def command_trial(args: argparse.Namespace) -> tuple[dict, int]: + service = _service(args) + started = service.trial_start( + _read_json(args.experiment, "experiment file"), args.case, args.arm, + _read_json(args.task_file, "trial task file"), args.idempotency_key, + ) + if args.wait: + return _wait_for_run(service, started, resume_candidate_review=True) + return envelope(data=started), 0 + + +def _normal_entry_result( + summary: dict[str, Any], + idempotency_key: str, + run: dict[str, Any] | None, +) -> dict[str, Any]: + run_id = run.get("run_id") if run is not None else None + state = run.get("state") if run is not None else "not_started" + if run_id is None: + next_action = "rerun this command without --dry-run" + elif state == "succeeded": + next_action = f"squad result {run_id} --json" + else: + next_action = f"squad status {run_id} --json" + return { + "dry_run": run is None, + "run_id": run_id, + "state": state, + "service": run, + "idempotency_key": idempotency_key, + **summary, + "next_action": next_action, + } + + +def _command_normal_entry( + args: argparse.Namespace, + *, + workflow: str, +) -> tuple[dict, int]: + if args.dry_run and args.wait: + raise ContractError("--wait cannot be combined with --dry-run") + repo = resolve_repository(args.project_dir) + reviewer_model = args.model if workflow == "branch-review" else args.review_model + reviewer_effort = args.effort if workflow == "branch-review" else args.review_effort + pins = tuple(role for role, explicit in ( + ("reviewer", reviewer_model is not None or reviewer_effort is not None), + ("implementer", workflow == "issue-delivery" and (args.implementer_model is not None or args.implementer_effort is not None)), + ) if explicit) + bindings = ( + _service(args).normal_entry_bindings(workflow, pinned_roles=pins) + if (Path(args.runtime_dir) / "state.sqlite3").is_file() else {} + ) + incumbent = bindings.get("reviewer", {}).get("profile", {}) + codex_identity = discover_codex_identity( + repo, + requested_model=reviewer_model or incumbent.get("model_id"), + requested_effort=reviewer_effort or incumbent.get("effort", {}).get("value"), + runtime=Path(args.runtime_dir), + ) + if workflow == "branch-review": + mode = args.mode + focus = args.focus + goal = ( + f"Review exact target {args.target} against base {args.base}" + + (f" with adversarial focus on {focus.strip()}" if focus else "") + + "." + ) + write_paths: tuple[str, ...] = () + claude_model = "sonnet" + claude_effort = "high" + else: + mode = args.review_mode + focus = args.review_focus + goal = args.issue + write_paths = tuple(args.write_path or ()) + incumbent = bindings.get("implementer", {}).get("profile", {}) + claude_model = args.implementer_model or incumbent.get("model_id", "sonnet") + claude_effort = args.implementer_effort or incumbent.get("effort", {}).get("value", "high") + task, summary = build_managed_task( + workflow=workflow, + project_dir=repo, + base_ref=args.base, + target_ref=args.target, + goal=goal, + codex_identity=codex_identity, + write_paths=write_paths, + checks=parse_checks(args.check), + check_timeout=args.check_timeout, + review_mode=mode, + review_focus=focus, + claude_model=claude_model, + claude_effort=claude_effort, + role_bindings=bindings, + pinned_roles=pins, + ) + idempotency_key = args.idempotency_key or ( + f"normal-{workflow}-{summary['task_sha256']}" + ) + if args.dry_run: + return envelope(data=_normal_entry_result( + summary, idempotency_key, None, + )), 0 + service = _service(args) + started = service.start(task, idempotency_key, None) + run, code = ( + _wait_for_run( + service, + started, + resume_candidate_review=(workflow == "issue-delivery"), + ) if args.wait + else (envelope(data=started), 0) + ) + service_data = run["data"] + if not args.json: + service_data = _display_status(service, service_data) + return envelope(data=_normal_entry_result( + summary, idempotency_key, service_data, + )), code + + +def command_review(args: argparse.Namespace) -> tuple[dict, int]: + return _command_normal_entry(args, workflow="branch-review") + + +def command_fix(args: argparse.Namespace) -> tuple[dict, int]: + return _command_normal_entry(args, workflow="issue-delivery") + + +def command_council(args: argparse.Namespace) -> tuple[dict, int]: + from .council_task_entry import build_council_task + if args.dry_run and args.wait: + raise ContractError("--wait cannot be combined with --dry-run") + repo = resolve_repository(args.project_dir) + identities = discover_codex_identity(repo, requested_effort=args.effort, runtime=Path(args.runtime_dir), all_models=True) + if args.model: + if len(args.model) != 3 or len(set(args.model)) != 3: + raise ContractError("Council --model must name exactly three distinct model IDs in proposer_a/proposer_b/critic order") + catalog = {value["model_id"]: value for value in identities} + if any(model not in catalog for model in args.model): + raise ContractError("Council pinned model is not available with verified effort metadata") + identities = [catalog[model] for model in args.model] + else: + # Prefer available catalog cross-family IDs; catalog entitlement is not + # tested quality qualification. The same Codex subscription harness + # with three entitled distinct IDs remains a valid bounded baseline. + chosen = [] + remaining = list(identities) + while remaining and len(chosen) < 3: + families = {identity["model_family"] for identity in chosen} + remaining.sort(key=lambda identity: identity["model_family"] in families) + chosen.append(remaining.pop(0)) + identities = chosen + evidence = [] + for value in args.evidence or []: + artifact_id, separator, sha256 = value.rpartition(":") + if not separator: + raise ContractError("Council --evidence must be ARTIFACT_ID:SHA256") + evidence.append({"artifact_id": artifact_id, "sha256": sha256}) + rubric = None + if args.criterion: + rubric = [] + for value in args.criterion: + criterion_id, separator, description = value.partition("=") + if not separator: + raise ContractError("Council --criterion must be ID=DESCRIPTION") + rubric.append({"id": criterion_id, "description": description}) + task, summary = build_council_task(project_dir=repo, goal=args.question, identities=identities, + base_ref=args.base, target_ref=args.target, read_paths=args.read_path or (), checks=parse_checks(args.check), + lead_mode=args.lead, max_invocations=args.max_invocations, evidence=evidence, rubric=rubric) + key = args.idempotency_key or f"normal-council-{summary['task_sha256']}" + if args.dry_run: + return envelope(data=_normal_entry_result(summary, key, None)), 0 + service = _service(args) + started = service.start(task, key) + run, code = _wait_for_run(service, started) if args.wait else (envelope(data=started), 0) + service_data = run["data"] if args.json else _display_status(service, run["data"]) + return envelope(data=_normal_entry_result(summary, key, service_data)), code + + +def command_council_finish(args: argparse.Namespace) -> tuple[dict, int]: + return command_finish(args) + + +def command_capacity_observe(args: argparse.Namespace) -> tuple[dict, int]: + observation = _read_json(args.file, "capacity observation file") + return envelope(data=_service(args).capacity_observe(observation)), 0 + + +def command_outcome_add(args: argparse.Namespace) -> tuple[dict, int]: + outcome = _read_json(args.file, "outcome file") + return envelope(data=_service(args).outcome_add(args.run, outcome)), 0 + + +def command_report(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).learning_report(args.project)), 0 + + +def command_policy_evaluate(args: argparse.Namespace) -> tuple[dict, int]: + experiment = _read_json(args.experiment, "experiment file") + if args.revision_id is not None or args.previous_evaluation_sha256 is not None: + return envelope(data=_service(args).policy_evaluate( + experiment, revision_id=args.revision_id, + previous_evaluation_sha256=args.previous_evaluation_sha256, + )), 0 + return envelope(data=_service(args).policy_evaluate(experiment)), 0 + + +def command_learn_propose(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).learning_propose(args.project)), 0 + + +def command_profile_template_add(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_template_add( + _read_json(args.file, "profile template file"), + )), 0 + + +def command_profile_binding_bootstrap(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_binding_bootstrap( + _read_json(args.file, "profile binding bootstrap file"), + )), 0 + + +def command_profile_qualification_add(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_qualification_add( + _read_json(args.file, "profile qualification file"), + )), 0 + + +def command_profile_binding_change(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_binding_change( + _read_json(args.file, "profile binding change file"), + )), 0 + + +def command_profile_binding_fallback(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_binding_fallback( + _read_json(args.file, "profile binding fallback file"), + )), 0 + + +def command_profile_binding_show(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_binding_status(args.alias)), 0 + + +def _selected_run(args: argparse.Namespace, service: Service) -> str: + return args.run if args.run is not None else service.resolve_run_id(None, Path(args.project_dir)) + + +def _display_status(service: Service, data: dict[str, Any]) -> dict[str, Any]: + if data.get("state") == "awaiting_host" and (data.get("handoff") or {}).get("status") == "open": + try: + view = service.handoff_view(data["run_id"]) + except ConflictError as exc: + # A live headless lead can advance while its status is rendered. + return {**data, "handoff_view_error": str(exc)} + if view["version"] == data.get("version"): + return {**data, "handoff_view": view} + return data + + +def command_status(args: argparse.Namespace) -> tuple[dict, int]: + service = _service(args) + data = service.status(_selected_run(args, service)) + return envelope(data=data if args.json else _display_status(service, data)), 0 +def command_events(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).events(args.run, args.after, args.limit)), 0 +def command_result(args: argparse.Namespace) -> tuple[dict, int]: + service = _service(args) + return envelope(data=service.result(_selected_run(args, service))), 0 +def command_cancel(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).cancel(args.run)), 0 +def command_resume(args: argparse.Namespace) -> tuple[dict, int]: + recovery = _read_json(args.recovery_file, "recovery file") if args.recovery_file else None + return envelope(data=_service(args).resume(args.run, recovery)), 0 + + +def command_finish(args: argparse.Namespace) -> tuple[dict, int]: + service = _service(args) + run_id = _selected_run(args, service) + choice = { + "chosen": args.choose, "supported_claims": args.supported_claim, + "discarded_alternatives": args.discarded_alternative, "validation": args.validation, + } + # Ordinary R6 completion keeps its call contract; Council fields are opt-in. + if all(value is None for value in choice.values()): + return envelope(data=service.finish(run_id, args.disposition, args.reason)), 0 + return envelope(data=service.finish(run_id, args.disposition, args.reason, **choice)), 0 + + +def command_handoff_claim(args: argparse.Namespace) -> tuple[dict, int]: + prior_claim = _read_json(args.claim_file, "claim file") if args.claim_file else None + return envelope(data=_service(args).handoff_claim( + args.run, args.expected_version, args.owner, prior_claim, + )), 0 + + +def command_handoff_complete(args: argparse.Namespace) -> tuple[dict, int]: + claim = _read_json(args.claim_file, "claim file") + decision = _read_json(args.decision_file, "decision file") + return envelope(data=_service(args).handoff_complete(args.run, claim, decision)), 0 + + +def command_mcp_serve(args: argparse.Namespace) -> int: + # Keep this import inside the explicitly requested command. Importing the + # ordinary CLI must remain valid when the optional SDK is absent. + from .mcp_server import MCPDependencyUnavailable, serve_stdio + + try: + serve_stdio( + Path(args.runtime_dir), + caller_surface=args.surface, + caller_session_ref=args.session_ref, + ) + except MCPDependencyUnavailable as exc: + # stdout is the MCP protocol channel, including during startup. + print(str(exc), file=sys.stderr) + return 69 + return 0 + + +class ContractParser(argparse.ArgumentParser): + def error(self, message: str) -> None: + raise ContractError(message) + + +def parser() -> argparse.ArgumentParser: + p = ContractParser(prog="squad") + p.add_argument("--version", action="version", version=f"squad {__version__}") + sub = p.add_subparsers(dest="command", required=True) + doctor = sub.add_parser("doctor") + doctor.add_argument("--json", action="store_true") + doctor.add_argument("--project-dir", default=str(Path.cwd())) + doctor.add_argument("--squad-executable") + doctor.set_defaults(func=command_doctor) + setup = sub.add_parser("setup") + setup.add_argument( + "--host", + action="append", + choices=("codex", "claude-code", "antigravity", "grok"), + ) + setup.add_argument("--dry-run", action="store_true") + setup.add_argument("--json", action="store_true") + setup.add_argument("--project-dir", default=str(Path.cwd())) + setup.add_argument("--squad-executable") + setup.set_defaults(func=command_setup) + for name, fn in (("prepare", command_prepare), ("classify", command_classify)): + cmd = sub.add_parser(name) + cmd.add_argument("adapter", choices=("codex", "antigravity", "grok")) + cmd.add_argument("--cwd", default=str(Path.cwd())) + cmd.add_argument("--model") + cmd.add_argument("--effort") + cmd.add_argument("--permission", choices=("read_only", "workspace_write"), default="read_only") + cmd.add_argument("--timeout", type=int, default=90) + cmd.add_argument("--transport", choices=("cli_exec", "native_protocol")) + cmd.add_argument("--catalog-file") + if name == "prepare": + cmd.add_argument("--prompt", required=True) + else: + cmd.add_argument("--returncode", type=int, required=True) + cmd.add_argument("--stdout-file", required=True) + cmd.add_argument("--stderr-file", required=True) + cmd.set_defaults(func=fn) + runtime_default = os.environ.get("DEVSQUAD_RUNTIME_DIR", str(Path.home() / ".devsquad" / "runtime")) + council = sub.add_parser("council", help="manual read-only Council decision, automatic triggering off") + council.add_argument("question") + council.add_argument("--project-dir", default=str(Path.cwd())) + council.add_argument("--base", default="HEAD") + council.add_argument("--target", default="HEAD") + council.add_argument("--read-path", action="append") + council.add_argument("--evidence", action="append", help="saved ARTIFACT_ID:SHA256") + council.add_argument("--criterion", action="append", help="ID=DESCRIPTION; freezes the complete rubric") + council.add_argument("--check", action="append") + council.add_argument("--model", action="append", help="exact model IDs, repeated three times") + council.add_argument("--effort") + council.add_argument("--lead", choices=("headless", "host"), default="headless") + council.add_argument("--max-invocations", type=int, default=4) + council.add_argument("--idempotency-key") + council.add_argument("--dry-run", action="store_true") + council.add_argument("--wait", action="store_true") + council.add_argument("--json", action="store_true") + council.add_argument("--runtime-dir", default=runtime_default) + council.set_defaults(func=command_council) + council_finish = sub.add_parser("council-finish", help="explicit host Council choice without decision JSON") + council_finish.add_argument("run", nargs="?") + council_disposition = council_finish.add_mutually_exclusive_group(required=True) + council_disposition.add_argument("--accept", dest="disposition", action="store_const", const="accept") + council_disposition.add_argument("--reject", dest="disposition", action="store_const", const="reject") + council_finish.add_argument("--reason", required=True) + council_finish.add_argument("--choose", choices=("A", "B", "synthesis"), required=True) + council_finish.add_argument("--supported-claim", action="append") + council_finish.add_argument("--discarded-alternative", action="append") + council_finish.add_argument("--validation", required=True) + council_finish.add_argument("--json", action="store_true") + council_finish.add_argument("--project-dir", default=str(Path.cwd())) + council_finish.add_argument("--runtime-dir", default=runtime_default) + council_finish.set_defaults(func=command_council_finish) + review = sub.add_parser( + "review", + help="start an exact-commit Codex branch review without task JSON", + ) + review.add_argument("--base", default="main") + review.add_argument("--target", default="HEAD") + review.add_argument("--project-dir", default=str(Path.cwd())) + review.add_argument("--model") + review.add_argument("--effort") + review.add_argument("--mode", choices=("standard", "adversarial"), default="standard") + review.add_argument("--focus") + review.add_argument("--check", action="append") + review.add_argument("--check-timeout", type=int, default=600) + review.add_argument("--idempotency-key") + review.add_argument("--dry-run", action="store_true") + review.add_argument("--wait", action="store_true") + review.add_argument("--json", action="store_true") + review.add_argument("--runtime-dir", default=runtime_default) + review.set_defaults(func=command_review) + fix = sub.add_parser( + "fix", + help="start bounded Claude implementation and independent Codex review", + ) + fix.add_argument("issue") + fix.add_argument("--base", default="HEAD") + fix.add_argument("--target", default="HEAD") + fix.add_argument("--project-dir", default=str(Path.cwd())) + fix.add_argument("--write-path", action="append") + fix.add_argument("--check", action="append") + fix.add_argument("--check-timeout", type=int, default=600) + fix.add_argument("--review-model") + fix.add_argument("--review-effort") + fix.add_argument( + "--review-mode", choices=("standard", "adversarial"), + default="standard", + ) + fix.add_argument("--review-focus") + fix.add_argument("--implementer-model") + fix.add_argument("--implementer-effort") + fix.add_argument("--idempotency-key") + fix.add_argument("--dry-run", action="store_true") + fix.add_argument("--wait", action="store_true") + fix.add_argument("--json", action="store_true") + fix.add_argument("--runtime-dir", default=runtime_default) + fix.set_defaults(func=command_fix) + start = sub.add_parser("start"); start.add_argument("--task-file", required=True); start.add_argument("--idempotency-key", required=True); start.add_argument("--supersedes-run"); start.add_argument("--wait", action="store_true"); start.add_argument("--json", action="store_true"); start.add_argument("--runtime-dir", default=runtime_default); start.set_defaults(func=command_start) + trial = sub.add_parser("trial", help="explicitly run one predeclared bounded experiment arm") + trial.add_argument("--experiment", required=True) + trial.add_argument("--case", required=True) + trial.add_argument("--arm", choices=("control", "candidate"), required=True) + trial.add_argument("--task-file", required=True) + trial.add_argument("--idempotency-key", required=True) + trial.add_argument("--wait", action="store_true") + trial.add_argument("--json", action="store_true") + trial.add_argument("--runtime-dir", default=runtime_default) + trial.set_defaults(func=command_trial) + for name, fn in (("status",command_status),("result",command_result),("cancel",command_cancel),("resume",command_resume)): + cmd=sub.add_parser(name); cmd.add_argument("run", nargs="?" if name in {"status", "result"} else None); cmd.add_argument("--json",action="store_true"); cmd.add_argument("--runtime-dir",default=runtime_default) + if name in {"status", "result"}: cmd.add_argument("--project-dir", default=str(Path.cwd())) + if name == "resume": cmd.add_argument("--recovery-file") + cmd.set_defaults(func=fn) + finish = sub.add_parser("finish", help="decide the current host handoff without decision JSON") + finish.add_argument("run", nargs="?") + disposition = finish.add_mutually_exclusive_group(required=True) + for value in ("accept", "reject", "revise"): + disposition.add_argument(f"--{value}", dest="disposition", action="store_const", const=value) + finish.add_argument("--reason", required=True) + finish.add_argument("--choose", choices=("A", "B", "synthesis"), help="required for Council, never inferred") + finish.add_argument("--supported-claim", action="append", help="Council supported claim, repeat as needed") + finish.add_argument("--discarded-alternative", action="append", help="Council discarded alternative, repeat as needed") + finish.add_argument("--validation", help="required Council objective validation and remaining uncertainty") + finish.add_argument("--project-dir", default=str(Path.cwd())) + finish.add_argument("--json", action="store_true") + finish.add_argument("--runtime-dir", default=runtime_default) + finish.set_defaults(func=command_finish) + events=sub.add_parser("events"); events.add_argument("run"); events.add_argument("--after",type=int,default=0); events.add_argument("--limit",type=int,default=100); events.add_argument("--json",action="store_true"); events.add_argument("--runtime-dir",default=runtime_default); events.set_defaults(func=command_events) + capacity = sub.add_parser("capacity") + capacity_sub = capacity.add_subparsers(dest="capacity_command", required=True) + observe = capacity_sub.add_parser("observe") + observe.add_argument("--file", required=True) + observe.add_argument("--json", action="store_true") + observe.add_argument("--runtime-dir", default=runtime_default) + observe.set_defaults(func=command_capacity_observe) + outcome = sub.add_parser("outcome") + outcome_sub = outcome.add_subparsers(dest="outcome_command", required=True) + outcome_add = outcome_sub.add_parser("add") + outcome_add.add_argument("run") + outcome_add.add_argument("--file", required=True) + outcome_add.add_argument("--json", action="store_true") + outcome_add.add_argument("--runtime-dir", default=runtime_default) + outcome_add.set_defaults(func=command_outcome_add) + report = sub.add_parser("report") + report.add_argument("--project", required=True) + report.add_argument("--json", action="store_true") + report.add_argument("--runtime-dir", default=runtime_default) + report.set_defaults(func=command_report) + policy = sub.add_parser("policy") + policy_sub = policy.add_subparsers(dest="policy_command", required=True) + evaluate = policy_sub.add_parser("evaluate") + evaluate.add_argument("--experiment", required=True) + evaluate.add_argument("--revision-id") + evaluate.add_argument("--previous-evaluation-sha256") + evaluate.add_argument("--json", action="store_true") + evaluate.add_argument("--runtime-dir", default=runtime_default) + evaluate.set_defaults(func=command_policy_evaluate) + learn = sub.add_parser("learn") + learn_sub = learn.add_subparsers(dest="learn_command", required=True) + propose = learn_sub.add_parser("propose") + propose.add_argument("--project", required=True) + propose.add_argument("--json", action="store_true") + propose.add_argument("--runtime-dir", default=runtime_default) + propose.set_defaults(func=command_learn_propose) + profile = sub.add_parser("profile") + profile_sub = profile.add_subparsers(dest="profile_command", required=True) + for name, fn in ( + ("template-add", command_profile_template_add), + ("binding-bootstrap", command_profile_binding_bootstrap), + ("qualification-add", command_profile_qualification_add), + ("binding-change", command_profile_binding_change), + ("binding-fallback", command_profile_binding_fallback), + ): + operation = profile_sub.add_parser(name) + operation.add_argument("--file", required=True) + operation.add_argument("--json", action="store_true") + operation.add_argument("--runtime-dir", default=runtime_default) + operation.set_defaults(func=fn) + binding_show = profile_sub.add_parser("binding-show") + binding_show.add_argument("alias") + binding_show.add_argument("--json", action="store_true") + binding_show.add_argument("--runtime-dir", default=runtime_default) + binding_show.set_defaults(func=command_profile_binding_show) + handoff = sub.add_parser("handoff") + handoff_sub = handoff.add_subparsers(dest="handoff_command", required=True) + claim = handoff_sub.add_parser("claim") + claim.add_argument("run") + claim.add_argument("--expected-version", type=int, required=True) + claim.add_argument("--owner", required=True) + claim.add_argument("--claim-file") + claim.add_argument("--json", action="store_true") + claim.add_argument("--runtime-dir", default=runtime_default) + claim.set_defaults(func=command_handoff_claim) + complete = handoff_sub.add_parser("complete") + complete.add_argument("run") + complete.add_argument("--claim-file", required=True) + complete.add_argument("--decision-file", required=True) + complete.add_argument("--json", action="store_true") + complete.add_argument("--runtime-dir", default=runtime_default) + complete.set_defaults(func=command_handoff_complete) + mcp = sub.add_parser("mcp") + mcp_sub = mcp.add_subparsers(dest="mcp_command", required=True) + serve = mcp_sub.add_parser("serve") + serve.add_argument("--runtime-dir", default=runtime_default) + serve.add_argument("--surface") + serve.add_argument("--session-ref") + serve.set_defaults(stream_func=command_mcp_serve) + return p + + +def _next_command(data: dict[str, Any]) -> str: + run_id = data.get("run_id", "RUN") + action = data.get("next_action") + if action == "claim_handoff": + pending = (data.get("handoff_view") or {}).get("pending_finish") + if pending: + command = f"squad finish {run_id} --{pending['disposition']} --reason={shlex.quote(pending['reason'])}" + choice = pending.get("council_choice") + if choice is not None: + command += f" --choose={shlex.quote(choice['chosen'])} --validation={shlex.quote(choice['validation'])}" + for value in choice["supported_claims"]: + command += f" --supported-claim={shlex.quote(value)}" + for value in choice["discarded_alternatives"]: + command += f" --discarded-alternative={shlex.quote(value)}" + return command + owner = (data.get("handoff") or {}).get("claimed_by") + if owner: + return f"complete or renew the saved claim in {owner}; this handoff already has an owner" + if (data.get("handoff_view") or {}).get("workflow") == "council-decision": + return f"inspect the Council evidence, then squad finish {run_id} with explicit disposition, --choose, --reason and --validation" + return f'squad finish {run_id} --accept --reason="your assessment of the saved evidence"' + if action in {"continue_headless_lead", "resume_candidate_review", "resume_council_stage", "handoff_submission_saved"}: + return f"squad resume {run_id}" + if action == "recovery_file_required": + return f"squad status {run_id} --json; inspect the recovery evidence before squad resume {run_id} --recovery-file FILE" + if data.get("state") in {"succeeded", "failed", "cancelled"}: + return f"squad result {run_id}" + if isinstance(action, str) and action: + return action.removesuffix(" --json") + return f"squad status {run_id}" + + +def _next_lines(data: dict[str, Any]) -> list[str]: + lines = [f"Next: {_next_command(data)}"] + if data.get("next_action") == "claim_handoff": + if (data.get("handoff_view") or {}).get("pending_finish"): + lines.append("Guidance: retry the exact saved intent; all disposition, reason and choice inputs must match.") + elif (data.get("handoff_view") or {}).get("workflow") == "council-decision": + lines.append("Guidance: choose A, B or synthesis yourself; --supported-claim and --discarded-alternative repeat; dissent is retained. Extra rounds require a new capped Council run.") + elif not (data.get("handoff") or {}).get("claimed_by"): + lines.append("Guidance: use --reject or --revise instead of --accept if the evidence requires it.") + return lines + + +def _handoff_lines(data: dict[str, Any]) -> list[str]: + view = data.get("handoff_view") + if not view: + return [f"Evidence unavailable: {data['handoff_view_error']}"] if data.get("handoff_view_error") else [] + if view.get("workflow") == "council-decision": + lines = ["Council: independent proposals committed before the distinct critic; automatic use off."] + for label, proposal in sorted(view["proposals"].items()): + lines.append(f"Proposal {label}: {proposal['summary']}") + critique = view["critique"] + lines.append(f"Critic: {critique['summary']}") + for assessment in critique["assessments"]: + lines.append(f"Criterion {assessment['criterion_id']} ({assessment['label']}): {assessment['status']} — {assessment['reason']}") + for objection in critique["objections"]: + lines.append(f"Dissent {objection['id']} ({objection['label']}): {objection['reason']}") + for check in view["checks"]: + lines.append(f"Check {check['id']}: {check['status']}") + artifact = view.get("report_artifact") + lines.append(f"Evidence {artifact['name']}: {artifact['path']}" if artifact else "Evidence: verified saved Council handoff report") + return lines + review = view["review"] + lines = [f"Review: {review['verdict']} — {review['summary']}"] + for finding in review.get("findings", []): + lines.append(f" {finding['severity']}: {finding['title']} ({finding['path']}:{finding['start_line']})") + for check in view["checks"]: + lines.append(f"Check {check['id']}: {check['status']}") + lines.append(f"Evidence {view['report']['name']}: {view['report']['path']}") + return lines + + +def _human_response(command: str, response: dict[str, Any]) -> str: + if not response["ok"]: + error = response["error"] + return f"{error['code']}: {error['message']}\nThe command did not complete. Existing runs remain saved; inspect squad status RUN." + data = response["data"] + lines = [] + if command == "doctor": + lines.append(f"DevSquad {data.get('core_version', __version__)}: {'ready' if data.get('ready') else 'needs attention'}") + for row in data.get("adapters", []): + auth = row.get("authentication", {}) + verified = row.get("operation_verified") + lines.append( + f"{row['adapter']}: {row.get('version') or 'not installed'}; " + f"adapter {row.get('status', 'unknown')}; " + f"authentication {auth.get('status', 'unknown')}; " + f"operation {'verified' if verified is True else 'unverified' if verified is False else 'unknown'}" + ) + if auth.get("next_action"): + lines.append(f" Next: {auth['next_action']}") + for workflow, row in data.get("supported_workflows", {}).items(): + lines.append(f"{workflow}: {'ready' if row.get('ready') else 'unavailable' if not row.get('supported') else 'needs attention'}") + if row.get("implemented_partial"): + lines.append(f" Implemented partial; native ready: false; automatic off. {row['reason']}") + for row in data.get("local_apps", []): + lines.append(f"{row.get('id', 'app')} registration: {row.get('status', 'unknown')}") + lines.append("Next: squad setup --dry-run" if not data.get("ready") else "Next: squad review --base main --dry-run") + elif command == "setup": + lines.append(f"Setup {'preview' if data.get('dry_run') else 'completed' if data.get('completed') else 'needs attention'}") + for row in data.get("hosts", []): + lines.append(f"{row.get('id', 'app')}: {row.get('action', 'unknown')}; registration {'ready' if row.get('ready') else 'needs attention'}") + lines.append("Next: squad setup" if data.get("dry_run") and data.get("completed") else "Next: squad doctor") + elif command in {"review", "fix", "council"}: + lines.append(f"{data.get('workflow', command)}: {data.get('state', 'unknown')}") + if data.get("run_id"): + lines.append(f"Run: {data['run_id']}") + lines.append(f"Project: {data.get('project', 'unknown')}") + lines.append(f"Commits: {data.get('base_oid', '')} → {data.get('target_oid', '')}") + for role, row in data.get("planned_roles", {}).items(): + lines.append(f"{role}: {row.get('harness', 'unknown')} {row.get('model_id', '')} / {row.get('effort', 'unknown')} ({row.get('selection_mode', 'unknown')})") + if data.get("selection_reason"): + lines.append(f"Selection: {data['selection_reason']}") + scope = data.get("scope", {}) + lines.append(f"Read scope: {', '.join(scope.get('read_paths', []))}; write scope: {', '.join(scope.get('write_paths', [])) or 'none'}") + for check in data.get("check_plan", []): + lines.append(f"Check {check['id']}: {shlex.join(check['argv'])} ({'required' if check['required_to_pass'] else 'report only'})") + if not data.get("check_plan"): + lines.append(f"Checks: {', '.join(data.get('checks', []))}") + if command == "council": + lines.append(f"Worker cap: {data['budget']['max_worker_invocations']}; rounds: 1; lead: {data['lead']['mode']}; automatic off") + lines.append("Native readiness: unavailable until exact-boundary backend attestation; dry-run is preparation only.") + next_data = data.get("service") or data + lines.extend(_handoff_lines(next_data)) + lines.extend(_next_lines(next_data)) + else: + run_id = data.get("run_id", "unknown") + lines.append(f"Run {run_id}: {data.get('state', 'unknown')}") + if data.get("phase"): + lines.append(f"Progress: {data['phase']}") + if command == "result": + if not data.get("ready"): + lines.append("Result is not ready; the run is saved.") + for artifact in data.get("artifacts", []): + lines.append(f"{artifact['name']}: {artifact['path']}") + if data.get("active_attempt"): + lines.append(f"Worker: {data['active_attempt'].get('status', 'unknown')}") + lines.extend(_handoff_lines(data)) + if data.get("disposition"): + lines.append(f"Disposition: {data['disposition']}") + if command == "cancel" and data.get("state") == "cancelling": + lines.append("Cancellation is saved; worker cleanup is still running.") + if command != "result" or not data.get("ready"): + lines.extend(_next_lines(data)) + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + arguments = list(sys.argv[1:] if argv is None else argv) + command = arguments[0] if arguments else "" + def emit(response): + if command in HUMAN_COMMANDS and "--json" not in arguments: + print(_human_response(command, response)) + else: + print(json.dumps(response, sort_keys=True)) + try: + args = parser().parse_args(arguments) + if hasattr(args, "stream_func"): + return args.stream_func(args) + response, code = args.func(args) + emit(response) + return code + except (ConflictError, SchemaVersionError) as exc: + emit(envelope(error=error_payload(exc.code, str(exc)))) + return 75 + except ContractError as exc: + emit(envelope(error=error_payload(exc.code, str(exc)))) + return 64 + except Exception as exc: + emit(envelope(error=error_payload("INTERNAL_ERROR", str(exc)))) + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/codex_lead_worker.py b/plugin/core/src/devsquad/codex_lead_worker.py new file mode 100644 index 0000000..5477b40 --- /dev/null +++ b/plugin/core/src/devsquad/codex_lead_worker.py @@ -0,0 +1,284 @@ +"""Drive one frozen read-only Codex headless-lead turn inside the M2 worker.""" + +from __future__ import annotations + +import hashlib +import io +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import time +from typing import Any + +from .catalog import normalize_models, verified_efforts +from .codex_protocol import ( + JsonLinePeer, + NativeTurnState, + discover_models, + initialize_request, + initialized_notification, + receive_response, + thread_start_request, + turn_start_request, +) +from .codex_review_worker import ( + MAX_PROTOCOL_EVENT_BYTES, + MAX_PROTOCOL_EVENTS, + MAX_SERVER_STDERR_BYTES, + _isolated_codex_environment, + _remaining, + _response_result, + _stop_server, + _usage, + _validated_adapter, + freeze_codex_role, +) +from .contracts import CapabilityUnavailable, ContractError, ProfileUnsupported +from .store import canonical_json +from .supervisor import BoundedDrain +from .workflows import ( + MAX_EVIDENCE_BYTES, + MAX_LEAD_BYTES, + build_lead_prompt, + decode_headless_lead_choice, + lead_output_schema, + make_headless_lead_evidence, +) + + +def freeze_codex_lead(selected: dict[str, Any]) -> dict[str, Any]: + return freeze_codex_role( + selected, role="lead", output_schema=lead_output_schema(), + ) + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("workflow snapshot must be an object") + adapter, profile = _validated_adapter( + snapshot, + role="lead", + adapter_key="lead_adapter", + output_schema=lead_output_schema(), + ) + handoff = snapshot.get("headless_handoff") + if not isinstance(handoff, dict) or not isinstance(handoff.get("packet"), dict): + raise ContractError("headless lead handoff is missing") + binary = Path(adapter["binary"]) + try: + resolved = binary.resolve(strict=True) + except OSError as exc: + raise CapabilityUnavailable("frozen Codex executable is missing") from exc + if (resolved != binary + or hashlib.sha256(binary.read_bytes()).hexdigest() != adapter["binary_sha256"]): + raise CapabilityUnavailable("frozen Codex executable changed after preflight") + try: + version = subprocess.run( + [str(binary), "--version"], text=True, capture_output=True, + timeout=3, check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise CapabilityUnavailable( + "frozen Codex version could not be re-observed" + ) from exc + if version.returncode != 0 or version.stdout.strip() != adapter["harness_version"]: + raise CapabilityUnavailable("Codex version changed after preflight") + + effort = profile["effort"]["value"] + model = profile["model_id"] + review_root = Path(snapshot["workspace"]["path"]).resolve(strict=True) + argv = [ + str(binary), + "-c", f'model="{model}"', + "-c", f'model_reasoning_effort="{effort}"', + "--disable", "apps", + "--disable", "plugins", + "--disable", "browser_use", + "--disable", "computer_use", + "--disable", "multi_agent", + "--enable", "skip_host_skill_discovery", + "app-server", "--listen", "stdio://", + ] + process: subprocess.Popen[bytes] | None = None + writer: io.TextIOWrapper | None = None + stderr: BoundedDrain | None = None + codex_home: tempfile.TemporaryDirectory | None = None + protocol_events: list[dict[str, Any]] = [] + protocol_bytes = 0 + deadline = time.monotonic() + snapshot["task"]["budget"]["wall_seconds"] + + def record(message: dict[str, Any]) -> None: + nonlocal protocol_bytes + encoded = canonical_json(message).encode() + protocol_bytes += len(encoded) + if (len(protocol_events) >= MAX_PROTOCOL_EVENTS + or protocol_bytes > MAX_PROTOCOL_EVENT_BYTES): + raise ContractError("native Codex protocol evidence exceeds its bound") + protocol_events.append(message) + + try: + codex_home, environment = _isolated_codex_environment(adapter["auth_file"]) + process = subprocess.Popen( + argv, + cwd=review_root, + env=environment, + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + start_new_session=False, + close_fds=True, + ) + assert process.stdin is not None and process.stdout is not None and process.stderr is not None + writer = io.TextIOWrapper(process.stdin, encoding="utf-8", write_through=True) + peer = JsonLinePeer(process.stdout, writer) + stderr = BoundedDrain(process.stderr, MAX_SERVER_STDERR_BYTES) + stderr.start() + + peer.send(initialize_request(1)) + _response_result(receive_response( + peer, 1, timeout_seconds=_remaining(deadline), on_notification=record, + ), "Codex initialize") + peer.send(initialized_notification()) + models = discover_models( + peer, first_request_id=10, timeout_seconds=_remaining(deadline), + ) + catalog = { + "complete": True, + "harness": "codex", + "harness_version": adapter["harness_version"], + "models": normalize_models("codex", adapter["harness_version"], models), + } + if effort not in verified_efforts( + catalog, + harness="codex", + version=adapter["harness_version"], + model_id=model, + ): + raise ProfileUnsupported( + f"unsupported or unverified effort {effort!r} for codex model {model!r}" + ) + + peer.send(thread_start_request( + 100, cwd=str(review_root), model=model, permission="read_only", + ephemeral=True, + )) + thread_result = _response_result(receive_response( + peer, 100, timeout_seconds=_remaining(deadline), on_notification=record, + ), "Codex thread/start") + thread = thread_result.get("thread") + thread_id = thread.get("id") if isinstance(thread, dict) else None + reported_version = thread.get("cliVersion") if isinstance(thread, dict) else None + if not isinstance(thread_id, str) or not thread_id: + raise ContractError("Codex thread/start returned no thread id") + if reported_version and f"codex-cli {reported_version}" != adapter["harness_version"]: + raise ContractError("Codex thread reported a different harness version") + reported_cwd = thread_result.get("cwd") + expected_policy = {"type": "readOnly", "networkAccess": False} + if (not isinstance(reported_cwd, str) or not reported_cwd + or thread_result.get("model") != model + or thread_result.get("reasoningEffort") != effort + or thread_result.get("modelProvider") != adapter["model_provider"] + or thread_result.get("approvalPolicy") != "never" + or thread_result.get("sandbox") != expected_policy + or Path(reported_cwd).resolve() != review_root): + raise ContractError("Codex thread did not preserve the frozen execution identity") + + prompt = build_lead_prompt(snapshot["task"], handoff["packet"]) + peer.send(turn_start_request( + 101, + thread_id=thread_id, + prompt=prompt, + model=model, + effort=effort, + cwd=str(review_root), + permission="read_only", + output_schema=lead_output_schema(), + )) + turn_result = _response_result(receive_response( + peer, 101, timeout_seconds=_remaining(deadline), on_notification=record, + ), "Codex turn/start") + turn = turn_result.get("turn") + turn_id = turn.get("id") if isinstance(turn, dict) else None + if not isinstance(turn_id, str) or not turn_id: + raise ContractError("Codex turn/start returned no turn id") + state = NativeTurnState(thread_id=thread_id, turn_id=turn_id) + for event in protocol_events: + state.consume(event) + output_bytes = sum(len(part.encode()) for part in state.output) + while not state.terminal: + message = peer.receive(_remaining(deadline)) + if "id" in message and "method" in message: + raise ContractError("native Codex requested an unsupported host action") + record(message) + prior = len(state.output) + state.consume(message) + output_bytes += sum(len(part.encode()) for part in state.output[prior:]) + if output_bytes > MAX_LEAD_BYTES: + raise ContractError("native Codex lead output exceeds its byte limit") + if state.terminal_status != "completed": + detail = ( + canonical_json(state.error)[:2000] + if state.error is not None else "no provider error was reported" + ) + raise ContractError( + "native Codex lead did not complete successfully: " + f"{state.terminal_status}; {detail}" + ) + lead_payload = state.final_output().strip() + if len(lead_payload.encode("utf-8")) > MAX_LEAD_BYTES: + raise ContractError("native Codex lead output exceeds its byte limit") + if not lead_payload: + raise ContractError( + "native Codex lead completed without output; protocol_summary=" + + canonical_json(state.output_diagnostics()) + ) + try: + choice = decode_headless_lead_choice(lead_payload, handoff["packet"]) + except ContractError as exc: + raise ContractError( + f"{exc}; protocol_summary=" + + canonical_json(state.output_diagnostics()) + ) from exc + usage = _usage(protocol_events, thread_id, turn_id) + finally: + if process is not None: + _stop_server(process, writer, stderr) + if codex_home is not None: + codex_home.cleanup() + + observed_identity = { + "harness": "codex", + "harness_version": adapter["harness_version"], + "model_provider": adapter["model_provider"], + "model_id": model, + "effort": effort, + "permission_policy": "read_only", + "verification": "verified", + } + return make_headless_lead_evidence( + snapshot, + handoff, + choice, + observed_identity=observed_identity, + native_ids={"thread_id": thread_id, "turn_id": turn_id}, + native_model_requests=None, + usage=usage, + ) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_EVIDENCE_BYTES + 1) + if len(payload) > MAX_EVIDENCE_BYTES: + raise ContractError("workflow snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("workflow snapshot is not valid UTF-8 JSON") from exc + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py new file mode 100644 index 0000000..854b6d2 --- /dev/null +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -0,0 +1,387 @@ +"""Typed JSON-RPC preparation/state normalization for Codex app-server. + +This module does not spawn or supervise the server. M2 gives it a connected +stdio stream owned by the run supervisor. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +import json +import os +import selectors +import time +from typing import Any + +from .contracts import ContractError + + +class JsonLinePeer: + """Protocol codec over supervisor-owned streams; does not own the process.""" + def __init__(self, reader: Any, writer: Any, *, max_frame_bytes: int = 4 * 1024 * 1024): + self.reader, self.writer = reader, writer + self._fd = reader.fileno() + self._buffer = bytearray() + self._max_frame_bytes = max_frame_bytes + + def send(self, message: dict[str, Any]) -> None: + self.writer.write(json.dumps(message, separators=(",", ":")) + "\n") + self.writer.flush() + + def receive(self, timeout_seconds: float) -> dict[str, Any]: + deadline = time.monotonic() + timeout_seconds + while b"\n" not in self._buffer: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("native protocol response timed out") + selector = selectors.DefaultSelector() + try: + selector.register(self._fd, selectors.EVENT_READ) + if not selector.select(remaining): + raise TimeoutError("native protocol response timed out") + finally: + selector.close() + chunk = os.read(self._fd, 65536) + if not chunk: + raise EOFError("native protocol disconnected") + self._buffer.extend(chunk) + if len(self._buffer) > self._max_frame_bytes: + raise ContractError("native protocol frame exceeds byte bound") + raw, _, remainder = self._buffer.partition(b"\n") + self._buffer = bytearray(remainder) + try: + value = json.loads(raw.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("native protocol returned malformed JSON") from exc + if not isinstance(value, dict): + raise ContractError("native protocol message must be an object") + return value + + +def request(request_id: int, method: str, params: dict[str, Any] | None = None) -> dict[str, Any]: + return {"id": request_id, "method": method, "params": params or {}} + + +def initialize_request(request_id: int = 1) -> dict[str, Any]: + return request(request_id, "initialize", {"clientInfo": {"name": "devsquad", "version": "0.1.0"}, "capabilities": {"experimentalApi": True}}) + + +def initialized_notification() -> dict[str, Any]: + return {"method": "initialized", "params": {}} + + +def thread_start_request( + request_id: int, + *, + cwd: str, + model: str, + permission: str, + ephemeral: bool = False, +) -> dict[str, Any]: + sandbox = {"read_only": "read-only", "workspace_write": "workspace-write"}.get(permission) + if sandbox is None: + raise ContractError(f"unsupported native permission: {permission}") + if type(ephemeral) is not bool: + raise ContractError("native thread ephemeral flag must be boolean") + return request(request_id, "thread/start", {"cwd": cwd, "model": model, "sandbox": sandbox, "approvalPolicy": "never", "ephemeral": ephemeral}) + + +def model_list_request(request_id: int, cursor: str | None = None, limit: int = 100) -> dict[str, Any]: + params: dict[str, Any] = {"limit": limit} + if cursor: + params["cursor"] = cursor + return request(request_id, "model/list", params) + + +def turn_start_request(request_id: int, *, thread_id: str, prompt: str, model: str | None, effort: str | None, cwd: str, permission: str, output_schema: dict[str, Any] | None = None) -> dict[str, Any]: + policy = {"read_only": {"type": "readOnly", "networkAccess": False}, "workspace_write": {"type": "workspaceWrite", "writableRoots": [cwd], "networkAccess": False}}.get(permission) + if policy is None: + raise ContractError(f"unsupported native permission: {permission}") + params: dict[str, Any] = {"threadId": thread_id, "input": [{"type": "text", "text": prompt}], "cwd": cwd, "sandboxPolicy": policy, "approvalPolicy": "never"} + if model: + params["model"] = model + if effort: + params["effort"] = effort + if output_schema is not None: + params["outputSchema"] = output_schema + return request(request_id, "turn/start", params) + + +def review_start_request(request_id: int, *, thread_id: str, target: dict[str, Any]) -> dict[str, Any]: + allowed = {"uncommittedChanges", "baseBranch", "commit", "custom"} + if target.get("type") not in allowed: + raise ContractError("unsupported review target") + return request(request_id, "review/start", {"threadId": thread_id, "target": target, "delivery": "inline"}) + + +def turn_interrupt_request(request_id: int, *, thread_id: str, turn_id: str) -> dict[str, Any]: + return request(request_id, "turn/interrupt", {"threadId": thread_id, "turnId": turn_id}) + + +@dataclass +class NativeTurnState: + thread_id: str | None = None + turn_id: str | None = None + terminal: bool = False + interrupted_acknowledged: bool = False + output: list[str] = field(default_factory=list) + completed_output: str | None = None + events: list[dict[str, Any]] = field(default_factory=list) + terminal_status: str | None = None + error: dict[str, Any] | None = None + + def final_output(self) -> str: + """Prefer the last authoritative completed agent message over deltas.""" + if self.completed_output is not None: + return self.completed_output + return "".join(self.output) + + def output_diagnostics(self) -> dict[str, Any]: + """Return bounded structural evidence without retaining model text.""" + method_counts: dict[str, int] = {} + matching_deltas = 0 + matching_delta_bytes = 0 + matching_completed_messages = 0 + matching_completed_message_bytes = 0 + matching_terminal_turns = 0 + matching_terminal_agent_messages = 0 + matching_terminal_agent_message_bytes = 0 + terminal_item_types: set[str] = set() + uncorrelated_output_events = 0 + for event in self.events: + method = event.get("method") + if not isinstance(method, str): + method = "" + method_counts[method] = method_counts.get(method, 0) + 1 + params = event.get("params") + if not isinstance(params, dict): + continue + event_turn = params.get("turn") + turn = event_turn if isinstance(event_turn, dict) else {} + message_thread = params.get("threadId") + message_turn = turn.get("id") or params.get("turnId") + correlated = ( + message_thread == self.thread_id and message_turn == self.turn_id + ) + if method in {"item/agentMessage/delta", "turn/output/delta"}: + if not correlated: + uncorrelated_output_events += 1 + continue + delta = params.get("delta") + if isinstance(delta, str): + matching_deltas += 1 + matching_delta_bytes += len(delta.encode("utf-8")) + elif method == "item/completed": + item = params.get("item") + if not isinstance(item, dict) or item.get("type") not in { + "agentMessage", "agent_message", + }: + continue + if not correlated: + uncorrelated_output_events += 1 + continue + content = item.get("text", item.get("content")) + if isinstance(content, str): + matching_completed_messages += 1 + matching_completed_message_bytes += len(content.encode("utf-8")) + elif method == "turn/completed" and correlated: + matching_terminal_turns += 1 + items = turn.get("items") + if not isinstance(items, list): + continue + for item in items: + if not isinstance(item, dict): + terminal_item_types.add("") + continue + item_type = item.get("type") + terminal_item_types.add( + item_type if isinstance(item_type, str) else "" + ) + if item_type not in {"agentMessage", "agent_message"}: + continue + content = item.get("text", item.get("content")) + if isinstance(content, str): + matching_terminal_agent_messages += 1 + matching_terminal_agent_message_bytes += len( + content.encode("utf-8") + ) + return { + "collected_output_bytes": sum( + len(part.encode("utf-8")) for part in self.output + ), + "collected_output_parts": len(self.output), + "completed_output_bytes": ( + len(self.completed_output.encode("utf-8")) + if self.completed_output is not None else None + ), + "event_count": len(self.events), + "matching_completed_message_bytes": matching_completed_message_bytes, + "matching_completed_messages": matching_completed_messages, + "matching_delta_bytes": matching_delta_bytes, + "matching_deltas": matching_deltas, + "matching_terminal_agent_message_bytes": ( + matching_terminal_agent_message_bytes + ), + "matching_terminal_agent_messages": matching_terminal_agent_messages, + "matching_terminal_turns": matching_terminal_turns, + "method_counts": dict(sorted(method_counts.items())), + "terminal_item_types": sorted(terminal_item_types), + "uncorrelated_output_events": uncorrelated_output_events, + } + + def consume(self, message: dict[str, Any]) -> None: + if not isinstance(message, dict): + raise ContractError("native message must be an object") + self.events.append(message) + method = message.get("method", "") + params = message.get("params", message.get("result", {})) + if method in {"thread/started", "thread/start/completed", "turn/started", "item/agentMessage/delta", "turn/output/delta", "item/completed", "turn/completed", "error"} and not isinstance(params, dict): + raise ContractError("native event params must be an object") + if not isinstance(params, dict): + return + if method in {"thread/started", "thread/start/completed"}: + thread = params.get("thread", {}) + if not isinstance(thread, dict): + raise ContractError("native thread must be an object") + candidate = thread.get("id") or params.get("threadId") + if candidate is not None and not isinstance(candidate, str): + raise ContractError("native thread id must be a string") + if self.thread_id and candidate and candidate != self.thread_id: + return + self.thread_id = candidate or self.thread_id + message_thread = params.get("threadId") + turn = params.get("turn") or {} + if not isinstance(turn, dict): + raise ContractError("native turn must be an object") + message_turn = turn.get("id") or params.get("turnId") + if message_thread is not None and not isinstance(message_thread, str): + raise ContractError("native thread id must be a string") + if message_turn is not None and not isinstance(message_turn, str): + raise ContractError("native turn id must be a string") + if self.thread_id and message_thread and message_thread != self.thread_id: + return + if self.turn_id and message_turn and message_turn != self.turn_id: + return + if method == "turn/started": + self.thread_id = message_thread or self.thread_id + self.turn_id = message_turn or self.turn_id + if method in {"item/agentMessage/delta", "turn/output/delta"}: + delta = params.get("delta", "") + if not isinstance(delta, str): + raise ContractError("native output delta must be a string") + if self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id: + self.output.append(delta) + if (method == "item/completed" and self.thread_id and self.turn_id + and message_thread == self.thread_id and message_turn == self.turn_id): + item = params.get("item") + if not isinstance(item, dict): + raise ContractError("native completed item must be an object") + if item.get("type") in {"agentMessage", "agent_message"}: + content = item.get("text", item.get("content")) + if not isinstance(content, str): + raise ContractError("native completed agent message must contain text") + if content.strip(): + self.completed_output = content + if method == "turn/completed" and self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id: + items = turn.get("items") + if items is not None: + if not isinstance(items, list): + raise ContractError("native terminal turn items must be an array") + completed_output = None + for item in items: + if not isinstance(item, dict): + raise ContractError("native terminal turn item must be an object") + if item.get("type") in {"agentMessage", "agent_message"}: + content = item.get("text", item.get("content")) + if not isinstance(content, str): + raise ContractError( + "native terminal agent message must contain text" + ) + if content.strip(): + completed_output = content + if completed_output is not None: + self.completed_output = completed_output + self.terminal = True + self.terminal_status = turn.get("status") + turn_error = turn.get("error") + if turn_error is not None: + if not isinstance(turn_error, dict): + raise ContractError("native terminal turn error must be an object") + self.error = turn_error + if method == "error" and self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id and not params.get("willRetry", False): + self.terminal = True + self.terminal_status = "failed" + self.error = params.get("error") + + def acknowledge_interrupt(self, response: dict[str, Any]) -> None: + if "error" in response: + raise ContractError(f"turn/interrupt failed: {response['error']}") + self.interrupted_acknowledged = True + + def disconnected(self) -> None: + if not self.terminal: + self.terminal_status = "transport_disconnected" + + +def parse_model_page(response: dict[str, Any]) -> tuple[list[dict[str, Any]], str | None]: + if "error" in response: + raise ContractError(f"model/list failed: {response['error']}") + result = response.get("result") + if not isinstance(result, dict): + raise ContractError("model/list missing result") + models = result["data"] if "data" in result else result.get("models") + if not isinstance(models, list): + raise ContractError("model/list is incomplete") + cursor = result.get("nextCursor") or result.get("next_cursor") + return models, cursor + + +def collect_model_pages(fetch_page: Any, *, max_pages: int = 100) -> list[dict[str, Any]]: + cursor = None + seen: set[str] = set() + all_models: list[dict[str, Any]] = [] + for _ in range(max_pages): + models, next_cursor = parse_model_page(fetch_page(cursor)) + all_models.extend(models) + if next_cursor is None: + return all_models + if next_cursor in seen: + raise ContractError("model/list repeated pagination cursor") + seen.add(next_cursor) + cursor = next_cursor + raise ContractError("model/list exceeded page bound") + + +def receive_response(peer: JsonLinePeer, request_id: int, *, timeout_seconds: float, on_notification: Any | None = None) -> dict[str, Any]: + deadline = time.monotonic() + timeout_seconds + while True: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError(f"native request {request_id} timed out") + message = peer.receive(remaining) + if message.get("id") == request_id: + return message + if "method" in message and "id" not in message: + if on_notification: + on_notification(message) + continue + raise ContractError(f"unexpected native response while awaiting request {request_id}") + + +def discover_models(peer: JsonLinePeer, *, first_request_id: int = 10, timeout_seconds: float = 5, max_pages: int = 100) -> list[dict[str, Any]]: + """Collect a complete native snapshot from an already initialized peer.""" + request_id, cursor = first_request_id, None + deadline = time.monotonic() + timeout_seconds + responses: list[dict[str, Any]] = [] + for _ in range(max_pages): + peer.send(model_list_request(request_id, cursor)) + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("model/list discovery deadline expired") + response = receive_response(peer, request_id, timeout_seconds=remaining) + responses.append(response) + _, cursor = parse_model_page(response) + if cursor is None: + return collect_model_pages(lambda ignored: responses.pop(0), max_pages=len(responses)) + request_id += 1 + raise ContractError("model/list exceeded page bound") diff --git a/plugin/core/src/devsquad/codex_review_worker.py b/plugin/core/src/devsquad/codex_review_worker.py new file mode 100644 index 0000000..b94c4f7 --- /dev/null +++ b/plugin/core/src/devsquad/codex_review_worker.py @@ -0,0 +1,537 @@ +"""Drive one frozen read-only Codex app-server review inside the M2 worker.""" + +from __future__ import annotations + +import hashlib +import io +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import time +from typing import Any + +from .catalog import normalize_models, verified_efforts +from .codex_protocol import ( + JsonLinePeer, + NativeTurnState, + discover_models, + initialize_request, + initialized_notification, + receive_response, + thread_start_request, + turn_start_request, +) +from .contracts import CapabilityUnavailable, ContractError, ProfileUnsupported +from .review_worker import run_review_and_checks +from .store import canonical_json +from .supervisor import BoundedDrain +from .workflows import ( + MAX_REVIEW_BYTES, + build_review_prompt, + decode_review_document, + review_output_schema, +) + + +MAX_SNAPSHOT_BYTES = 2 * 1024 * 1024 +MAX_PROTOCOL_EVENT_BYTES = 4 * 1024 * 1024 +MAX_PROTOCOL_EVENTS = 10_000 +MAX_SERVER_STDERR_BYTES = 256 * 1024 +ADAPTER_FIELDS = { + "schema_version", "harness", "transport", "binary", "binary_sha256", + "harness_version", "model_provider", "output_schema_sha256", "auth_file", +} + + +def _adapter_manifest_path() -> Path: + source = Path(__file__).resolve().parents[2] / "adapters" / "codex" / "adapter.json" + if source.is_file(): + return source + installed = Path(sys.prefix) / "share" / "devsquad" / "adapters" / "codex" / "adapter.json" + if installed.is_file(): + return installed + raise CapabilityUnavailable("Codex adapter manifest is unavailable") + + +def _subscription_auth_file(value: str | None = None) -> Path: + try: + if value is None: + configured_home = os.environ.get("CODEX_HOME") + root = Path(configured_home).expanduser() if configured_home else Path.home() / ".codex" + auth_file = (root / "auth.json").resolve(strict=True) + else: + auth_file = Path(value).resolve(strict=True) + metadata = auth_file.stat() + except OSError as exc: + raise CapabilityUnavailable("Codex subscription auth file is unavailable") from exc + if not auth_file.is_file(): + raise CapabilityUnavailable("Codex subscription auth path is not a file") + if hasattr(os, "getuid") and metadata.st_uid != os.getuid(): + raise CapabilityUnavailable("Codex subscription auth file is not user-owned") + if os.name != "nt" and metadata.st_mode & 0o077: + raise CapabilityUnavailable("Codex subscription auth file permissions are too broad") + return auth_file + + +def freeze_codex_role( + selected: dict[str, Any], + *, + role: str, + output_schema: dict[str, Any], +) -> dict[str, Any]: + """Resolve and verify one read-only Codex role during preflight.""" + from .adapters import AdapterManifest, harness_version + + if not isinstance(selected, dict) or not isinstance(selected.get("profile"), dict): + raise ContractError(f"frozen {role} selection is invalid") + profile = selected["profile"] + if profile.get("harness") != "codex": + raise CapabilityUnavailable( + f"selected {role} harness is not implemented for M3: {profile.get('harness')}" + ) + if profile.get("permission_policy") != "read_only": + raise ProfileUnsupported(f"Codex {role} requires read_only permission") + if set(profile.get("required_tools", [])) - {"read"}: + raise ProfileUnsupported(f"Codex {role} profile requests unsupported tools") + effort = profile.get("effort") + if (not isinstance(effort, dict) or effort.get("transport") != "native" + or not isinstance(effort.get("value"), str) or not effort["value"]): + raise ProfileUnsupported(f"Codex {role} requires an explicit native effort") + if not isinstance(profile.get("model_id"), str) or not profile["model_id"]: + raise ProfileUnsupported(f"Codex {role} requires an exact model id") + + manifest = AdapterManifest.load(_adapter_manifest_path()) + if manifest.name != "codex" or manifest.transport != "native_protocol": + raise CapabilityUnavailable("installed Codex adapter is not native_protocol") + binary_name = manifest.resolve_binary() + if not binary_name: + raise CapabilityUnavailable("Codex executable is unavailable") + binary = Path(binary_name).resolve(strict=True) + version = harness_version(str(binary)) + if not version: + raise CapabilityUnavailable("Codex version could not be observed") + if version not in manifest.verified_versions: + raise ProfileUnsupported(f"unverified Codex app-server version: {version}") + content = binary.read_bytes() + schema_hash = hashlib.sha256(canonical_json(output_schema).encode()).hexdigest() + return { + "schema_version": 1, + "harness": "codex", + "transport": "native_protocol", + "binary": str(binary), + "binary_sha256": hashlib.sha256(content).hexdigest(), + "harness_version": version, + "model_provider": manifest.model_provider or "openai", + "output_schema_sha256": schema_hash, + "auth_file": str(_subscription_auth_file()), + } + + +def freeze_codex_reviewer(selected: dict[str, Any]) -> dict[str, Any]: + """Resolve and verify the non-model Codex reviewer identity.""" + return freeze_codex_role( + selected, role="reviewer", output_schema=review_output_schema(), + ) + + +def _validated_adapter( + snapshot: dict[str, Any], + *, + role: str = "reviewer", + adapter_key: str = "review_adapter", + output_schema: dict[str, Any] | None = None, +) -> tuple[dict[str, Any], dict[str, Any]]: + adapter = snapshot.get(adapter_key) + try: + selected = snapshot["routing"]["roles"][role]["selected"] + profile = selected["profile"] + except (KeyError, TypeError) as exc: + raise ContractError(f"frozen Codex {role} selection is missing") from exc + if not isinstance(adapter, dict) or set(adapter) != ADAPTER_FIELDS: + raise ContractError(f"frozen Codex {role} adapter fields are invalid") + if (adapter["schema_version"] != 1 or type(adapter["schema_version"]) is not int + or adapter["harness"] != "codex" + or adapter["transport"] != "native_protocol" + or adapter["model_provider"] != "openai"): + raise ContractError(f"frozen Codex {role} adapter identity is invalid") + for field in ( + "binary", "binary_sha256", "harness_version", "output_schema_sha256", + "auth_file", + ): + if not isinstance(adapter[field], str) or not adapter[field]: + raise ContractError(f"frozen Codex {role} adapter value is invalid") + if (len(adapter["binary_sha256"]) != 64 + or len(adapter["output_schema_sha256"]) != 64): + raise ContractError(f"frozen Codex {role} adapter hash is invalid") + if not Path(adapter["auth_file"]).is_absolute(): + raise ContractError("frozen Codex auth path is not absolute") + if (not isinstance(profile, dict) or profile.get("harness") != "codex" + or profile.get("permission_policy") != "read_only"): + raise ContractError(f"frozen profile is not a read-only Codex {role}") + schema = review_output_schema() if output_schema is None else output_schema + schema_hash = hashlib.sha256(canonical_json(schema).encode()).hexdigest() + if schema_hash != adapter["output_schema_sha256"]: + raise ContractError(f"frozen {role} output schema changed") + return adapter, profile + + +def _remaining(deadline: float) -> float: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("native Codex review exceeded its deadline") + return remaining + + +def _response_result(response: dict[str, Any], label: str) -> dict[str, Any]: + if "error" in response: + raise ContractError(f"{label} failed: {response['error']}") + result = response.get("result") + if not isinstance(result, dict): + raise ContractError(f"{label} returned no result") + return result + + +def _usage(events: list[dict[str, Any]], thread_id: str, turn_id: str) -> dict[str, Any]: + for event in reversed(events): + if event.get("method") != "thread/tokenUsage/updated": + continue + params = event.get("params") + if (not isinstance(params, dict) or params.get("threadId") != thread_id + or params.get("turnId") not in {None, turn_id}): + continue + token_usage = params.get("tokenUsage") + total = token_usage.get("total") if isinstance(token_usage, dict) else None + if not isinstance(total, dict): + continue + values = [total.get(key) for key in ("inputTokens", "outputTokens", "totalTokens")] + if all(type(value) is int and value >= 0 for value in values): + return { + "input_tokens": values[0], + "output_tokens": values[1], + "total_tokens": values[2], + "source": "native_reported", + } + return { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + } + + +def _stop_server( + process: subprocess.Popen[bytes], + writer: io.TextIOWrapper | None, + stderr: BoundedDrain | None, +) -> None: + if writer is not None: + try: + writer.close() + except OSError: + pass + if process.poll() is None: + process.terminate() + try: + process.wait(timeout=2) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=2) + if stderr is not None: + metadata = stderr.finish() + if stderr.content: + sys.stderr.buffer.write(bytes(stderr.content)) + if metadata["truncated"]: + sys.stderr.write( + f"\n[Codex stderr truncated; total_bytes={metadata['total_bytes']} " + f"sha256={metadata['full_sha256']}]\n" + ) + + +def _isolated_codex_environment( + auth_file_value: str, +) -> tuple[tempfile.TemporaryDirectory, dict[str, str]]: + """Expose subscription auth without loading user sessions, config or plugins.""" + auth_file = _subscription_auth_file(auth_file_value) + home = tempfile.TemporaryDirectory(prefix="devsquad-codex-home-") + root = Path(home.name) + try: + root.chmod(0o700) + (root / "auth.json").symlink_to(auth_file) + except Exception: + home.cleanup() + raise + environment = os.environ.copy() + environment["CODEX_HOME"] = str(root) + return home, environment + + +def run(snapshot: dict[str, Any], *, council_role: str | None = None, + council_prompt: str | None = None) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("workflow snapshot must be an object") + schema = review_output_schema() + if council_role is not None: + from .council import output_schema + schema = output_schema(council_role) + adapter, profile = _validated_adapter(snapshot, role=council_role or "reviewer", + adapter_key="council_adapter" if council_role else "review_adapter", + output_schema=schema) + binary = Path(adapter["binary"]) + try: + resolved = binary.resolve(strict=True) + except OSError as exc: + raise CapabilityUnavailable("frozen Codex executable is missing") from exc + if (resolved != binary + or hashlib.sha256(binary.read_bytes()).hexdigest() != adapter["binary_sha256"]): + raise CapabilityUnavailable("frozen Codex executable changed after preflight") + try: + version = subprocess.run( + [str(binary), "--version"], + text=True, + capture_output=True, + timeout=3, + check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise CapabilityUnavailable( + "frozen Codex version could not be re-observed" + ) from exc + if version.returncode != 0 or version.stdout.strip() != adapter["harness_version"]: + raise CapabilityUnavailable("Codex version changed after preflight") + + effort = profile["effort"]["value"] + model = profile["model_id"] + workspace = snapshot["workspace"] + review_root = Path(workspace["path"]).resolve(strict=True) + argv = [ + str(binary), + "-c", f'model="{model}"', + "-c", f'model_reasoning_effort="{effort}"', + "--disable", "apps", + "--disable", "plugins", + "--disable", "browser_use", + "--disable", "computer_use", + "--disable", "multi_agent", + "--enable", "skip_host_skill_discovery", + "app-server", "--listen", "stdio://", + ] + process: subprocess.Popen[bytes] | None = None + writer = None + stderr = None + codex_home: tempfile.TemporaryDirectory | None = None + protocol_events: list[dict[str, Any]] = [] + protocol_bytes = 0 + deadline = time.monotonic() + snapshot["task"]["budget"]["wall_seconds"] + + def record(message: dict[str, Any]) -> None: + nonlocal protocol_bytes + encoded = canonical_json(message).encode() + protocol_bytes += len(encoded) + if (len(protocol_events) >= MAX_PROTOCOL_EVENTS + or protocol_bytes > MAX_PROTOCOL_EVENT_BYTES): + raise ContractError("native Codex protocol evidence exceeds its bound") + protocol_events.append(message) + + try: + codex_home, environment = _isolated_codex_environment(adapter["auth_file"]) + if council_role is not None: + import shutil + from .council_isolation import command + auth = Path(codex_home.name) / "auth.json" + auth.unlink() + shutil.copyfile(adapter["auth_file"], auth) + auth.chmod(0o600) + scratch = Path(codex_home.name).resolve(strict=True) + environment["CODEX_HOME"] = str(scratch) + environment["HOME"] = str(scratch) + environment["TMPDIR"] = str(scratch) + argv = command(snapshot["council_boundary"], argv, scratch=scratch) + process = subprocess.Popen( + argv, + cwd=review_root, + env=environment, + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + start_new_session=False, + close_fds=True, + ) + assert process.stdin is not None and process.stdout is not None and process.stderr is not None + writer = io.TextIOWrapper(process.stdin, encoding="utf-8", write_through=True) + peer = JsonLinePeer(process.stdout, writer) + stderr = BoundedDrain(process.stderr, MAX_SERVER_STDERR_BYTES) + stderr.start() + + peer.send(initialize_request(1)) + _response_result( + receive_response( + peer, 1, timeout_seconds=_remaining(deadline), + on_notification=record, + ), + "Codex initialize", + ) + peer.send(initialized_notification()) + models = discover_models( + peer, + first_request_id=10, + timeout_seconds=_remaining(deadline), + ) + catalog = { + "complete": True, + "harness": "codex", + "harness_version": adapter["harness_version"], + "models": normalize_models("codex", adapter["harness_version"], models), + } + if effort not in verified_efforts( + catalog, + harness="codex", + version=adapter["harness_version"], + model_id=model, + ): + raise ProfileUnsupported( + f"unsupported or unverified effort {effort!r} for codex model {model!r}" + ) + + peer.send(thread_start_request( + 100, + cwd=str(review_root), + model=model, + permission="read_only", + ephemeral=True, + )) + thread_result = _response_result( + receive_response( + peer, 100, timeout_seconds=_remaining(deadline), + on_notification=record, + ), + "Codex thread/start", + ) + thread = thread_result.get("thread") + thread_id = thread.get("id") if isinstance(thread, dict) else None + reported_version = thread.get("cliVersion") if isinstance(thread, dict) else None + if not isinstance(thread_id, str) or not thread_id: + raise ContractError("Codex thread/start returned no thread id") + if reported_version and f"codex-cli {reported_version}" != adapter["harness_version"]: + raise ContractError("Codex thread reported a different harness version") + expected_policy = {"type": "readOnly", "networkAccess": False} + reported_cwd = thread_result.get("cwd") + if not isinstance(reported_cwd, str) or not reported_cwd: + raise ContractError("Codex thread returned an invalid working directory") + if (thread_result.get("model") != model + or thread_result.get("reasoningEffort") != effort + or thread_result.get("modelProvider") != adapter["model_provider"] + or thread_result.get("approvalPolicy") != "never" + or thread_result.get("sandbox") != expected_policy + or Path(reported_cwd).resolve() != review_root): + raise ContractError("Codex thread did not preserve the frozen execution identity") + + prompt = council_prompt if council_role is not None else build_review_prompt(snapshot["task"], workspace) + peer.send(turn_start_request( + 101, + thread_id=thread_id, + prompt=prompt, + model=model, + effort=effort, + cwd=str(review_root), + permission="read_only", + output_schema=schema, + )) + turn_result = _response_result( + receive_response( + peer, 101, timeout_seconds=_remaining(deadline), + on_notification=record, + ), + "Codex turn/start", + ) + turn = turn_result.get("turn") + turn_id = turn.get("id") if isinstance(turn, dict) else None + if not isinstance(turn_id, str) or not turn_id: + raise ContractError("Codex turn/start returned no turn id") + state = NativeTurnState(thread_id=thread_id, turn_id=turn_id) + for event in protocol_events: + state.consume(event) + output_bytes = sum(len(part.encode()) for part in state.output) + while not state.terminal: + message = peer.receive(_remaining(deadline)) + if "id" in message and "method" in message: + raise ContractError("native Codex requested an unsupported host action") + record(message) + prior = len(state.output) + state.consume(message) + output_bytes += sum(len(part.encode()) for part in state.output[prior:]) + if output_bytes > MAX_REVIEW_BYTES: + raise ContractError("native Codex review output exceeds its byte limit") + if state.terminal_status != "completed": + detail = ( + canonical_json(state.error)[:2000] + if state.error is not None else "no provider error was reported" + ) + raise ContractError( + "native Codex review did not complete successfully: " + f"{state.terminal_status}; {detail}" + ) + review_payload = state.final_output().strip() + if len(review_payload.encode("utf-8")) > MAX_REVIEW_BYTES: + raise ContractError("native Codex review output exceeds its byte limit") + if not review_payload: + raise ContractError( + "native Codex review completed without output; protocol_summary=" + + canonical_json(state.output_diagnostics()) + ) + try: + if council_role is not None: + from .council import decode + review = decode(review_payload) + else: + review = decode_review_document(review_payload, snapshot["task"], workspace) + except ContractError as exc: + raise ContractError( + f"{exc}; protocol_summary=" + + canonical_json(state.output_diagnostics()) + ) from exc + usage = _usage(protocol_events, thread_id, turn_id) + finally: + if process is not None: + _stop_server(process, writer, stderr) + if codex_home is not None: + codex_home.cleanup() + + observed_identity = { + "harness": "codex", + "harness_version": adapter["harness_version"], + "model_provider": adapter["model_provider"], + "model_id": model, + "effort": effort, + "permission_policy": "read_only", + "verification": "verified", + } + if council_role is not None: + return {"document": review, "observed_identity": observed_identity, + "native_ids": {"thread_id": thread_id, "turn_id": turn_id}, "usage": usage} + return run_review_and_checks( + snapshot, + review, + observed_identity=observed_identity, + native_ids={"thread_id": thread_id, "turn_id": turn_id}, + native_model_requests=None, + usage=usage, + ) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_SNAPSHOT_BYTES + 1) + if len(payload) > MAX_SNAPSHOT_BYTES: + raise ContractError("workflow snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("workflow snapshot is not valid UTF-8 JSON") from exc + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/contracts.py b/plugin/core/src/devsquad/contracts.py new file mode 100644 index 0000000..9c0d1b2 --- /dev/null +++ b/plugin/core/src/devsquad/contracts.py @@ -0,0 +1,168 @@ +"""Strict M1 contracts for prepared invocations and normalized results.""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import Any, Literal + +SCHEMA_VERSION = 1 +Transport = Literal["cli_exec", "native_protocol"] +Verification = Literal["verified", "unverified", "unavailable", "unknown"] + + +class ContractError(ValueError): + """Raised before launch when an invocation contract is unsupported.""" + + code = "INPUT_INVALID" + + +class ProfileUnsupported(ContractError): + code = "PROFILE_UNSUPPORTED" + + +class CapabilityUnavailable(ContractError): + code = "CAPABILITY_UNAVAILABLE" + + +class BudgetExhausted(ContractError): + code = "BUDGET_EXHAUSTED" + + +class PolicyDenied(ContractError): + code = "POLICY_DENIED" + + +@dataclass(frozen=True) +class ExecutionIdentity: + harness: str + harness_version: str | None + model_provider: str | None + model_family: str | None + model: str | None + effort: str | None + tools: tuple[str, ...] = () + permissions: str = "read_only" + account_pool: str | None = None + verification: Verification = "unknown" + + def __post_init__(self) -> None: + if not isinstance(self.harness, str) or not self.harness: + raise ContractError("identity harness must be non-empty") + if not isinstance(self.permissions, str) or self.permissions not in {"read_only", "workspace_write"}: + raise ContractError("identity permission is invalid") + if not isinstance(self.verification, str) or self.verification not in {"verified", "unverified", "unavailable", "unknown"}: + raise ContractError("identity verification is invalid") + if not isinstance(self.tools, tuple) or not all(isinstance(v, str) for v in self.tools): + raise ContractError("identity tools must be a string tuple") + if len(set(self.tools)) != len(self.tools) or any(not v for v in self.tools): + raise ContractError("identity tools must contain unique non-empty strings") + for name in ("harness_version", "model_provider", "model_family", "model", "effort", "account_pool"): + value = getattr(self, name) + if value is not None and (not isinstance(value, str) or not value): + raise ContractError(f"identity {name} must be a non-empty string or null") + + +@dataclass(frozen=True) +class LaunchSpec: + """A launch description. M2 owns spawning, timeout, cancellation and reaping.""" + + schema_version: int + adapter: str + transport: Transport + argv: tuple[str, ...] + cwd: str + stdin_path: str | None + timeout_seconds: int + requested: ExecutionIdentity + environment: dict[str, str] = field(default_factory=dict) + + def __post_init__(self) -> None: + if type(self.schema_version) is not int or self.schema_version != SCHEMA_VERSION: + raise ContractError("unsupported schema_version") + if not isinstance(self.transport, str) or self.transport not in ("cli_exec", "native_protocol"): + raise ContractError("unsupported transport") + if not isinstance(self.adapter, str) or not self.adapter: + raise ContractError("adapter must be non-empty") + if not isinstance(self.argv, tuple) or not self.argv or not all(isinstance(v, str) and v for v in self.argv): + raise ContractError("argv must be a non-empty string array") + if type(self.timeout_seconds) is not int or self.timeout_seconds <= 0: + raise ContractError("timeout_seconds must be positive") + if not isinstance(self.cwd, str) or not self.cwd or not Path(self.cwd).is_absolute(): + raise ContractError("cwd must be absolute") + if self.stdin_path is not None and (not isinstance(self.stdin_path, str) or not self.stdin_path): + raise ContractError("stdin_path must be a non-empty string or null") + if not isinstance(self.requested, ExecutionIdentity): + raise ContractError("requested must be an execution identity") + allowed_env = {"DEVSQUAD_WORKER", "DEVSQUAD_RUN_ID", "DEVSQUAD_ATTEMPT_ID", "DEVSQUAD_DELEGATION_DEPTH", "DEVSQUAD_COUNCIL_ROLE"} + if not isinstance(self.environment, dict) or set(self.environment) - allowed_env or not all(isinstance(k, str) and isinstance(v, str) for k, v in self.environment.items()): + raise ContractError("environment contains non-allowlisted or non-string values") + if self.environment.get("DEVSQUAD_COUNCIL_ROLE", "") not in {"", "proposer_a", "proposer_b", "critic", "lead"}: + raise ContractError("Council worker role marker is invalid") + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + +@dataclass(frozen=True) +class NormalizedResult: + """Execution evidence without conflating artifacts or acceptance.""" + + schema_version: int + execution_status: Literal[ + "succeeded", "failed", "timed_out", "interrupted", "denied", "malformed" + ] + error_code: str | None + output: str | None + artifact_status: Literal["present", "missing", "not_required", "unknown"] + acceptance_status: Literal["pending", "accepted", "rejected", "not_evaluated"] + requested: ExecutionIdentity + observed: ExecutionIdentity | None + native_ids: dict[str, str] = field(default_factory=dict) + events: tuple[dict[str, Any], ...] = () + + def __post_init__(self) -> None: + if type(self.schema_version) is not int or self.schema_version != SCHEMA_VERSION: + raise ContractError("unsupported schema_version") + if not isinstance(self.execution_status, str) or self.execution_status not in {"succeeded", "failed", "timed_out", "interrupted", "denied", "malformed"}: + raise ContractError("invalid execution_status") + if not isinstance(self.artifact_status, str) or self.artifact_status not in {"present", "missing", "not_required", "unknown"}: + raise ContractError("invalid artifact_status") + if not isinstance(self.acceptance_status, str) or self.acceptance_status not in {"pending", "accepted", "rejected", "not_evaluated"}: + raise ContractError("invalid acceptance_status") + for name in ("error_code", "output"): + value = getattr(self, name) + if value is not None and not isinstance(value, str): + raise ContractError(f"{name} must be a string or null") + if not isinstance(self.requested, ExecutionIdentity) or (self.observed is not None and not isinstance(self.observed, ExecutionIdentity)): + raise ContractError("requested/observed identity is invalid") + if not isinstance(self.native_ids, dict) or not all(isinstance(k, str) and k and isinstance(v, str) and v for k, v in self.native_ids.items()): + raise ContractError("native_ids must contain non-empty string pairs") + if not isinstance(self.events, tuple) or not all(isinstance(v, dict) for v in self.events): + raise ContractError("events must be an object tuple") + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + +def envelope(*, data: Any = None, error: dict[str, Any] | None = None) -> dict[str, Any]: + if (data is None) == (error is None): + raise ContractError("exactly one of data and error is required") + return {"schema_version": SCHEMA_VERSION, "ok": error is None, "data": data, "error": error} + + +def error_payload(code: str, message: str, *, retryable: bool = False, details: dict[str, Any] | None = None) -> dict[str, Any]: + return {"code": code, "message": message, "retryable": retryable, "details": details or {}} + + +def validate_launch_payload(value: dict[str, Any]) -> None: + if not isinstance(value, dict): + raise ContractError("LaunchSpec must be an object") + expected = {"schema_version", "adapter", "transport", "argv", "cwd", "stdin_path", "timeout_seconds", "requested", "environment"} + if set(value) != expected: + raise ContractError(f"LaunchSpec fields differ: {sorted(set(value) ^ expected)}") + if not isinstance(value["argv"], (list, tuple)) or isinstance(value["argv"], (str, bytes)): + raise ContractError("argv must be an array") + if not isinstance(value["requested"], dict) or not isinstance(value["requested"].get("tools"), (list, tuple)): + raise ContractError("requested identity is invalid") + LaunchSpec(**{**value, "argv": tuple(value["argv"]), "requested": ExecutionIdentity(**{**value["requested"], "tools": tuple(value["requested"]["tools"])})}) diff --git a/plugin/core/src/devsquad/council.py b/plugin/core/src/devsquad/council.py new file mode 100644 index 0000000..6055604 --- /dev/null +++ b/plugin/core/src/devsquad/council.py @@ -0,0 +1,242 @@ +"""Strict contracts for the explicitly invoked, read-only Council workflow.""" +from __future__ import annotations + +import hashlib +import json +from typing import Any + +from .contracts import ContractError +from .store import canonical_json + +ROLES = ("proposer_a", "proposer_b", "critic") +MAX_PACKET_BYTES = 512 * 1024 +PROMPT_VERSION = "council-role-packet-v1" + + +def role_prompt(role: str, packet: dict) -> str: + return (f"You are the single read-only Council {role}. Read evidence.json in your directory. " + "Use the frozen rubric and cite only supplied evidence IDs. Produce the requested structured document. " + "No peer research or implementation writes. The lead retains every objection ID and required validation. " + "Preference cannot override a failed mandatory check.\n" + canonical_json(packet)) + + +def prompt_digest(role: str, packet: dict) -> str: + return hashlib.sha256(role_prompt(role, packet).encode()).hexdigest() + + +def digest(value: Any) -> str: + return hashlib.sha256(canonical_json(value).encode()).hexdigest() + + +def exact(value: Any, fields: set[str], label: str) -> dict: + if not isinstance(value, dict) or set(value) != fields: + raise ContractError(f"{label} fields are invalid") + return value + + +def text(value: Any, label: str) -> str: + if not isinstance(value, str) or not value.strip() or len(value) > 16000: + raise ContractError(f"{label} must be bounded non-empty text") + return value + + +def sha(value: Any) -> str: + if not isinstance(value, str) or len(value) != 64 or any(c not in "0123456789abcdef" for c in value): + raise ContractError("Council hash is invalid") + return value + + +def validate_spec(spec: Any) -> dict: + exact(spec, {"schema_version", "enabled", "automatic", "reason", "min_valid_proposals", + "required_critics", "max_invocations", "seed", "evidence", "rubric"}, "CouncilSpec") + if type(spec["schema_version"]) is not int or spec["schema_version"] != 1: + raise ContractError("CouncilSpec version is invalid") + if spec["enabled"] is not True or spec["automatic"] is not False: + raise ContractError("Council requires explicit invocation; automatic Council is disabled") + if type(spec["min_valid_proposals"]) is not int or spec["min_valid_proposals"] != 2: + raise ContractError("Council requires two valid proposals") + if type(spec["required_critics"]) is not int or spec["required_critics"] != 1: + raise ContractError("Council requires one distinct critic") + if type(spec["max_invocations"]) is not int or not 3 <= spec["max_invocations"] <= 16: + raise ContractError("Council invocation cap must be between 3 and 16") + text(spec["reason"], "Council reason") + sha(spec["seed"]) + if not isinstance(spec["evidence"], list) or len(spec["evidence"]) > 32: + raise ContractError("Council evidence must be a bounded array") + identifiers = set() + for item in spec["evidence"]: + exact(item, {"artifact_id", "sha256"}, "Council evidence reference") + text(item["artifact_id"], "artifact_id") + sha(item["sha256"]) + if item["artifact_id"] in identifiers: + raise ContractError("Council evidence IDs must be unique") + identifiers.add(item["artifact_id"]) + if not isinstance(spec["rubric"], list) or not 1 <= len(spec["rubric"]) <= 32: + raise ContractError("Council rubric must be a non-empty bounded array") + identifiers = set() + for item in spec["rubric"]: + exact(item, {"id", "description"}, "Council rubric item") + text(item["id"], "rubric id") + text(item["description"], "rubric description") + if item["id"] in identifiers: + raise ContractError("Council rubric IDs must be unique") + identifiers.add(item["id"]) + return spec + + +def label_mapping(seed: str, authors: list[str]) -> dict[str, str]: + sha(seed) + if len(authors) != 2 or len(set(authors)) != 2: + raise ContractError("Council label mapping requires two distinct authors") + ordered = sorted(authors, key=lambda author: hashlib.sha256((seed + ":" + author).encode()).hexdigest()) + return dict(zip(("A", "B"), ordered)) + + +def evidence_ids(value: Any, spec: dict) -> list[str]: + allowed = {ref["artifact_id"] for ref in spec["evidence"]} | set(spec.get("source_ids", [])) + if (not isinstance(value, list) or not all(isinstance(v, str) and v in allowed for v in value) + or len(value) != len(set(value))): + raise ContractError("Council evidence IDs are unknown or duplicated") + return value + + +def sanitized_proposals(snapshot: dict) -> dict: + """Strip supplied identity tokens from text; raw originals remain sealed.""" + import re + mapping = label_mapping(snapshot["task"]["council"]["seed"], ["proposer_a", "proposer_b"]) + documents = snapshot["council_state"]["documents"] + sensitive = set() + for role in ("proposer_a", "proposer_b"): + evidence = documents[role] + profile = evidence["profile"] + sensitive.update([role, profile["id"], profile["model_id"], profile["account_pool_id"]]) + sensitive.update(str(value) for value in (evidence.get("observed_identity") or {}).values() + if isinstance(value, str) and len(value) > 3) + def sanitize(value, key=None): + if key == "evidence_ids": + return value # Frozen provenance IDs must not be rewritten. + if isinstance(value, str): + for token in sorted(sensitive, key=len, reverse=True): + value = re.sub(re.escape(token), "[identity omitted]", value, flags=re.IGNORECASE) + return value + if isinstance(value, list): + return [sanitize(item) for item in value] + if isinstance(value, dict): + return {field: sanitize(item, field) for field, item in value.items()} + return value + return {label: sanitize(documents[author]["document"]) for label, author in mapping.items()} + + +def validate_proposal(value: Any, spec: dict) -> dict: + exact(value, {"summary", "approach", "claims", "validation"}, "Council proposal") + for field in ("summary", "approach", "validation"): + text(value[field], field) + if not isinstance(value["claims"], list) or not 1 <= len(value["claims"]) <= 64: + raise ContractError("Council proposal claims must be non-empty and bounded") + for claim in value["claims"]: + exact(claim, {"text", "evidence_ids"}, "Council claim") + text(claim["text"], "claim text") + evidence_ids(claim["evidence_ids"], spec) + return value + + +def validate_critique(value: Any, spec: dict) -> dict: + exact(value, {"summary", "assessments", "objections"}, "Council critique") + text(value["summary"], "critique summary") + required = {(label, criterion["id"]) for label in ("A", "B") for criterion in spec["rubric"]} + seen = set() + if not isinstance(value["assessments"], list) or len(value["assessments"]) != len(required): + raise ContractError("Council critique must assess every label and rubric criterion") + for item in value["assessments"]: + exact(item, {"label", "criterion_id", "status", "reason", "evidence_ids"}, "Council assessment") + pair = (item["label"], item["criterion_id"]) + if pair not in required or pair in seen or item["status"] not in {"supported", "unsupported", "uncertain"}: + raise ContractError("Council assessment label/criterion/status is invalid or duplicated") + seen.add(pair) + text(item["reason"], "assessment reason") + evidence_ids(item["evidence_ids"], spec) + if not isinstance(value["objections"], list) or len(value["objections"]) > 64: + raise ContractError("Council objections must be bounded") + seen = set() + for item in value["objections"]: + exact(item, {"id", "label", "reason", "evidence_ids"}, "Council objection") + if item["label"] not in {"A", "B"} or item["id"] in seen: + raise ContractError("Council objection label/ID is invalid") + seen.add(text(item["id"], "objection id")) + text(item["reason"], "objection reason") + evidence_ids(item["evidence_ids"], spec) + return value + + +def validate_choice(value: Any, spec: dict, critique: dict) -> dict: + exact(value, {"disposition", "reason", "chosen", "supported_claims", "discarded_alternatives", + "unresolved_objections", "validation"}, "Council decision") + if value["disposition"] not in {"accept", "reject", "revise"} or value["chosen"] not in {"A", "B", "synthesis"}: + raise ContractError("Council disposition/chosen label is invalid") + for field in ("reason", "validation"): + text(value[field], field) + for field in ("supported_claims", "discarded_alternatives", "unresolved_objections"): + if not isinstance(value[field], list) or len(value[field]) > 64 or not all(isinstance(v, str) and v for v in value[field]): + raise ContractError(f"Council decision {field} must be a bounded text array") + for entry in value[field]: + text(entry, f"Council decision {field} entry") + objections = {item["id"] for item in critique["objections"]} + # Retain every objection; disposition does not silently erase a dissenting source. + if set(value["unresolved_objections"]) != objections or len(value["unresolved_objections"]) != len(objections): + raise ContractError("Council decision must retain every recorded objection ID") + if value["disposition"] == "accept" and not value["supported_claims"]: + raise ContractError("Council acceptance needs supported claims") + return value + + +def output_schema(role: str) -> dict: + string = {"type": "string", "minLength": 1} + array = {"type": "array", "items": string} + def obj(properties): + return {"type": "object", "additionalProperties": False, "required": list(properties), "properties": properties} + if role in {"proposer_a", "proposer_b"}: + return obj({"summary": string, "approach": string, "claims": {"type": "array", "minItems": 1, + "items": obj({"text": string, "evidence_ids": array})}, "validation": string}) + if role == "critic": + return obj({"summary": string, "assessments": {"type": "array", "items": obj({ + "label": {"enum": ["A", "B"]}, "criterion_id": string, + "status": {"enum": ["supported", "unsupported", "uncertain"]}, "reason": string, + "evidence_ids": array})}, "objections": {"type": "array", "items": obj({ + "id": string, "label": {"enum": ["A", "B"]}, "reason": string, "evidence_ids": array})}}) + if role == "lead": + return obj({"disposition": {"enum": ["accept", "reject", "revise"]}, "reason": string, + "chosen": {"enum": ["A", "B", "synthesis"]}, "supported_claims": array, + "discarded_alternatives": array, "unresolved_objections": array, "validation": string}) + raise ContractError("Council role is invalid") + + +def verify_identities(documents: dict[str, dict], *, fixture: bool) -> None: + identities = [] + native_ids = set() + for role in ROLES: + value = documents.get(role) + if not isinstance(value, dict): + raise ContractError(f"Council is missing {role}; no valid quorum") + identity = value.get("observed_identity") + if fixture: + if identity is not None or value.get("identity_scope") != "all_fixture": + raise ContractError("Council mixed fixture/native identities cannot establish quorum") + identities.append(value["profile"]["model_id"].casefold()) + else: + if (not isinstance(identity, dict) or identity.get("verification") != "verified" + or not isinstance(identity.get("model_id"), str) or not identity["model_id"]): + raise ContractError("Council requires verified observed model identities") + identities.append(identity["model_id"].casefold()) + ids = value.get("native_ids", {}) + pair = (ids.get("thread_id"), ids.get("turn_id")) + if not all(isinstance(item, str) and item for item in pair) or pair in native_ids: + raise ContractError("Council native role turns must be distinct correlated thread/turn identities") + native_ids.add(pair) + if len(set(identities)) != 3: + raise ContractError("Council critic/proposers must have three distinct model identities") + + +def decode(value: bytes | str) -> dict: + from .workflows import _strict_json_object + result = _strict_json_object(value, "Council evidence", maximum=MAX_PACKET_BYTES) + return result diff --git a/plugin/core/src/devsquad/council_comparison.py b/plugin/core/src/devsquad/council_comparison.py new file mode 100644 index 0000000..b418118 --- /dev/null +++ b/plugin/core/src/devsquad/council_comparison.py @@ -0,0 +1,166 @@ +"""Predeclared workflow mechanics comparison, separate from R3 eligibility. + +Public fixture runs support observable overhead/failure mechanics, not native +quality. No fabricated score, profile promotion or automatic Council authority +can be produced by this bounded comparison. +""" +from __future__ import annotations + +from datetime import datetime, timezone +import hashlib +import json +import os +from pathlib import Path + +from .contracts import ContractError +from .council import digest, exact, sha, text +from .store import canonical_json, ConflictError + +ARMS = ("control", "council") +CONTRACT_FIELDS = ("project", "goal", "acceptance", "checks", "scope") + + +def _file_sha(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _task_version(task: dict) -> dict: + if set(task["routing"]) != {"profiles", "policy"}: + raise ContractError("Workflow comparison requires embedded frozen routing documents") + module = "council.py" if task["workflow"] == "council-decision" else "workflows.py" + return {"task_sha256": digest(task), "profiles_sha256": digest(task["routing"]["profiles"]), + "policy_sha256": digest(task["routing"]["policy"]), "prompt_module": module, + "prompt_module_sha256": _file_sha(Path(__file__).with_name(module))} + + +def predeclare(path: Path, cases: list[dict]) -> dict: + """Create once before runs. Each case fixes both task versions and split.""" + if not isinstance(cases, list) or not 2 <= len(cases) <= 16: + raise ContractError("Comparison requires bounded matched and held-out cases") + declared, identifiers, splits, contracts = [], set(), set(), set() + for case in cases: + exact(case, {"id", "split", "control", "council"}, "workflow comparison case") + text(case["id"], "comparison case ID") + if case["id"] in identifiers or case["split"] not in {"matched", "heldout"}: + raise ContractError("Comparison case IDs/splits are invalid") + identifiers.add(case["id"]) + splits.add(case["split"]) + control, council = case["control"], case["council"] + if control["workflow"] != "branch-review" or council["workflow"] != "council-decision": + raise ContractError("Comparison arms must be branch-review and Council") + contract = {field: control[field] for field in CONTRACT_FIELDS} + if contract != {field: council[field] for field in CONTRACT_FIELDS}: + raise ContractError("Comparison arms do not share the declared input/check contract") + contract_sha256 = digest(contract) + if contract_sha256 in contracts: + raise ContractError("Comparison cases must have unique input contracts; repetitions are not held-out independence") + contracts.add(contract_sha256) + for task in (control, council): + for field in ("base_ref", "target_ref"): + value = task["project"][field] + if not isinstance(value, str) or len(value) not in {40, 64} or any(c not in "0123456789abcdef" for c in value): + raise ContractError("Comparison must pin exact Git commit IDs") + declared.append({"id": case["id"], "split": case["split"], "input_contract_sha256": contract_sha256, + **{arm: _task_version(case[arm]) for arm in ARMS}}) + if splits != {"matched", "heldout"}: + raise ContractError("Comparison must predeclare both matched and held-out cases") + plan = {"schema_version": 1, "created_at": datetime.now(timezone.utc).isoformat(), + "scope": "public_fixture_mechanics", "automatic_enabled": False, + "quality_gate": {"accepted_quality": "unmeasured", "escaped_defects": "unmeasured", "rework": "unmeasured"}, + "cases": declared, "decision_rule": "inconclusive_without_independent_native_quality_evidence"} + path.parent.mkdir(parents=True, exist_ok=True) + descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0), 0o600) + with os.fdopen(descriptor, "wb") as stream: + stream.write(canonical_json(plan).encode()) + stream.flush() + os.fsync(stream.fileno()) + return {"path": str(path), "sha256": _file_sha(path)} + + +def report(store, declaration: dict, pairs: list[dict]) -> dict: + """Validate actual terminal public receipts, then save an immutable report.""" + exact(declaration, {"path", "sha256"}, "comparison declaration reference") + sha(declaration["sha256"]) + path = Path(declaration["path"]) + if path.is_symlink() or _file_sha(path) != declaration["sha256"]: + raise ConflictError("Workflow comparison predeclaration changed") + plan = json.loads(path.read_bytes()) + if plan.get("scope") != "public_fixture_mechanics" or plan.get("automatic_enabled") is not False: + raise ContractError("Comparison declaration is not the bounded public fixture gate") + if len({case["input_contract_sha256"] for case in plan["cases"]}) != len(plan["cases"]): + raise ContractError("Comparison repetitions cannot be relabelled independent matched/held-out cases") + if not isinstance(pairs, list) or len(pairs) != len(plan["cases"]): + raise ContractError("Comparison requires every predeclared case") + assigned = {item["id"]: item for item in pairs} + if len(assigned) != len(pairs) or set(assigned) != {item["id"] for item in plan["cases"]}: + raise ContractError("Comparison pairs differ from the predeclared cases") + created = datetime.fromisoformat(plan["created_at"]) + results, seen = [], set() + for case in plan["cases"]: + pair = assigned[case["id"]] + exact(pair, {"id", "control", "council"}, "comparison pair") + arms = {} + for arm in ARMS: + run_id = pair[arm] + if run_id in seen: + raise ContractError("A comparison run cannot be reused across arms/cases") + seen.add(run_id) + run = store.run(run_id) + if datetime.fromisoformat(run["created_at"]) <= created or run["state"] not in {"succeeded", "failed", "cancelled"}: + raise ConflictError("Comparison run was not terminal after its predeclaration") + snapshot = json.loads(run["mutable_snapshot"]) + task = snapshot["task"] + requested = json.loads(run["submitted_request"])["task"] + if digest(requested) != case[arm]["task_sha256"]: + raise ConflictError("Comparison run task differs from its declared arm") + if digest({field: task[field] for field in CONTRACT_FIELDS}) != case["input_contract_sha256"]: + raise ConflictError("Comparison actual frozen input contract differs") + if (snapshot["routing"]["profile_registry"]["sha256"] != case[arm]["profiles_sha256"] + or snapshot["routing"]["policy"]["sha256"] != case[arm]["policy_sha256"] + or _file_sha(Path(run["package_path"]) / "devsquad" / case[arm]["prompt_module"]) != case[arm]["prompt_module_sha256"]): + raise ConflictError("Comparison frozen profile/policy/prompt version differs") + if (arm == "council" and snapshot.get("council_fixture") is None + or arm == "control" and "internal_review_fixture" not in snapshot): + raise ContractError("This comparison gate accepts actual public fixture mechanics only") + artifact = store.artifact_named(run_id, "receipt.json") + if artifact is None: + raise ConflictError("Comparison run has no terminal workflow receipt") + from .council_runtime import verified_artifact + raw = verified_artifact(store, run_id, "receipt.json", artifact["sha256"]) + receipt = json.loads(raw) + if receipt["run_id"] != run_id or receipt["state"] != run["state"]: + raise ConflictError("Comparison receipt identity/state differs") + attempts = receipt["attempts"] + accounting = receipt.get("accounting", receipt) + arms[arm] = {"run_id": run_id, "receipt": {"artifact_id": artifact["id"], "sha256": artifact["sha256"]}, + "state": run["state"], "lead_disposition": receipt["lead"]["disposition"], + "runtime_package_sha256": run["package_digest"], + "input_contract_sha256": case["input_contract_sha256"], "versions": case[arm], + "actual_prompt_sha256": [item.get("prompt_sha256") for item in attempts], + "candidate": receipt["candidate"], "brief_sha256": receipt.get("brief_sha256"), + "accepted_quality": None, "escaped_defects": None, "rework": None, + "quality_missingness": "fixture mechanics and host acceptance are not independent native quality measurements", + "execution_elapsed_ms": store._execution_elapsed_ms(run_id, run, datetime.now(timezone.utc)), + "worker_invocations": accounting["worker_invocations"], + "native_usage": [item.get("usage") for item in attempts], + "native_model_requests": accounting.get("native_model_requests"), "native_quota": None, + "host_usage": None, "identity_scope": "all_fixture"} + results.append({"id": case["id"], "split": case["split"], "arms": arms}) + value = {"schema_version": 1, "predeclaration_sha256": declaration["sha256"], "cases": results, + "conclusion": "inconclusive", "quality_benefit_supported": False, "automatic_enabled": False, + "limitations": ["Controlled process mechanics only", "No independent native accepted-quality/escaped-defect/rework observations", + "Workflow prompts and invocation count differ; this is not R3 single-binding eligibility"]} + # Terminal workflow ledgers stay immutable. This separately versioned + # comparison artifact references their exact receipts, never rewrites them. + destination = path.parent / (declaration["sha256"] + ".workflow-comparison.json") + encoded = canonical_json(value).encode() + if destination.exists(): + if destination.is_symlink() or destination.read_bytes() != encoded: + raise ConflictError("Workflow comparison report is already frozen differently") + else: + descriptor = os.open(destination, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0), 0o600) + with os.fdopen(descriptor, "wb") as stream: + stream.write(encoded) + stream.flush() + os.fsync(stream.fileno()) + return {"report": value, "path": str(destination), "sha256": digest(value)} diff --git a/plugin/core/src/devsquad/council_isolation.py b/plugin/core/src/devsquad/council_isolation.py new file mode 100644 index 0000000..cf01de9 --- /dev/null +++ b/plugin/core/src/devsquad/council_isolation.py @@ -0,0 +1,141 @@ +"""Versioned macOS default-deny read boundary for untrusted Council participants.""" +from __future__ import annotations + +import hashlib +import json +from pathlib import Path +import platform +import subprocess +import os + +from .contracts import CapabilityUnavailable, ContractError + +BOUNDARY_VERSION = 1 + + +def _literal(path: Path) -> str: + # JSON escaping safely quotes Seatbelt strings without shell interpolation. + return json.dumps(str(path.resolve(strict=True))) + + +def profile(*, executable: Path, evidence: Path, scratch: Path | None = None, + dependencies: tuple[Path, ...] = ()) -> str: + executable = executable.resolve(strict=True) + evidence = evidence.resolve(strict=True) + roots = [evidence, *[p.resolve(strict=True) for p in dependencies]] + system_roots = [Path("/usr/lib"), Path("/System/Library")] + ancestors = {Path("/")} + for root in [executable, *roots, *system_roots, *([scratch.resolve(strict=True)] if scratch else [])]: + ancestors.update(root.parents) + literals = " ".join(f"(literal {_literal(p)})" for p in sorted(ancestors, key=str)) + read_roots = " ".join(f"(subpath {_literal(p)})" for p in [*system_roots, *roots]) + result = ("(version 1)\n(deny default)\n(allow process-fork)\n(allow sysctl-read)\n" + f"(allow process-exec (literal {_literal(executable)}))\n" + f"(allow file-read* (literal {_literal(executable)}) {literals} {read_roots})\n") + if scratch is not None: + result += f"(allow file-read* file-write* (subpath {_literal(scratch)}))\n" + return result + + +def freeze_boundary(*, executable: Path, evidence: Path, dependencies: tuple[Path, ...] = (), + native_codex: bool = False) -> dict: + if platform.system() != "Darwin" or not Path("/usr/bin/sandbox-exec").is_file(): + raise CapabilityUnavailable("Council requires the verified macOS Seatbelt read boundary") + value = profile(executable=executable, evidence=evidence, dependencies=dependencies) + if native_codex: + # Codex probes absent managed config paths during bootstrap. Metadata + # only, not managed-config bytes/MCP configuration, may be inspected. + value += ('(allow file-read-metadata (literal "/etc") (literal "/private/etc") ' + '(subpath "/etc/codex") (subpath "/private/etc/codex"))\n' + '(allow file-read* (literal "/dev/urandom"))\n' + '(allow mach-lookup (global-name "com.apple.cfprefsd.daemon") (global-name "com.apple.cfprefsd.agent"))\n' + f'(allow ipc-posix-shm-read-data (ipc-posix-name "apple.cfprefs.{os.getuid()}v1") (ipc-posix-name "apple.cfprefs.daemonv1"))\n' + '(allow network-outbound (remote tcp "*:443"))\n') + return {"schema_version": BOUNDARY_VERSION, "platform": platform.system(), + "sandbox_binary": "/usr/bin/sandbox-exec", "profile": value, + "profile_sha256": hashlib.sha256(value.encode()).hexdigest(), + "executable_sha256": hashlib.sha256(executable.resolve(strict=True).read_bytes()).hexdigest(), + "dependency_roots": [str(p.resolve(strict=True)) for p in dependencies], + "native_codex": native_codex, "system_read_roots": ["/usr/lib", "/System/Library"], + "uid": os.getuid(), + "network": "outbound_tcp_443" if native_codex else "denied"} + + +def command(boundary: dict, argv: list[str], *, scratch: Path | None = None) -> list[str]: + if (boundary.get("schema_version") != BOUNDARY_VERSION or boundary.get("platform") != "Darwin" + or boundary.get("uid") != os.getuid() + or platform.system() != "Darwin" or boundary.get("sandbox_binary") != "/usr/bin/sandbox-exec" + or hashlib.sha256(boundary.get("profile", "").encode()).hexdigest() != boundary.get("profile_sha256") + or hashlib.sha256(Path(argv[0]).read_bytes()).hexdigest() != boundary.get("executable_sha256")): + raise CapabilityUnavailable("frozen Council read boundary changed or is unavailable") + value = boundary["profile"] + if scratch is not None: + # Provider-owned auth/session cache; never contains peer artifacts or the ledger. + value += f"(allow file-read* file-write* (subpath {_literal(scratch)}))\n" + for ancestor in scratch.resolve(strict=True).parents: + value += f"(allow file-read* (literal {_literal(ancestor)}))\n" + return [boundary["sandbox_binary"], "-p", value, *argv] + + +def probe(boundary: dict, *, executable: Path, own_file: Path, forbidden: tuple[Path, ...]) -> None: + for path, allowed in [(own_file, True), *[(p, False) for p in forbidden]]: + result = subprocess.run(command(boundary, [str(executable), str(path)]), + capture_output=True, timeout=3, check=False) + if (allowed and result.returncode != 0) or (not allowed and result.returncode == 0): + raise CapabilityUnavailable("Council filesystem read isolation probe failed") + + +def verify_read_boundary(boundary: dict, *, own_file: Path, forbidden: tuple[Path, ...]) -> None: + """Probe the frozen file rules; the diagnostic cat executable alone is added.""" + value = boundary["profile"] + '(allow process-exec (literal "/bin/cat"))\n(allow file-read* (literal "/bin/cat") (literal "/bin"))\n' + for path, allowed in [(own_file, True), *[(p, False) for p in forbidden]]: + result = subprocess.run([boundary["sandbox_binary"], "-p", value, "/bin/cat", str(path.resolve(strict=True))], + capture_output=True, timeout=3, check=False) + if (allowed and result.returncode != 0) or (not allowed and result.returncode == 0): + raise CapabilityUnavailable("Council default-deny peer/runtime read probe failed") + + +def verify_native_bootstrap(boundary: dict, *, executable: Path, expected_version: str, evidence: Path) -> None: + """Non-generating app-server bootstrap under the exact frozen boundary.""" + import tempfile + from .codex_protocol import JsonLinePeer, initialize_request, receive_response + from .diagnostics import _probe_output + from .probe_process import capture_probe_identity, close_probe + with tempfile.TemporaryDirectory(prefix="devsquad-council-bootstrap-") as temporary: + scratch = Path(temporary).resolve(strict=True) + environment = {"PATH": "/usr/bin:/bin", "CODEX_HOME": str(scratch), "HOME": str(scratch), "TMPDIR": str(scratch)} + code, output = _probe_output(command(boundary, [str(executable), "--version"], scratch=scratch), + project=evidence, environment=environment) + if code != 0 or output.strip() != expected_version: + raise CapabilityUnavailable("Council native binary cannot start inside frozen read isolation") + process = subprocess.Popen(command(boundary, [str(executable), "--disable", "apps", "--disable", "plugins", + "app-server", "--listen", "stdio://"], scratch=scratch), cwd=evidence, env=environment, + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, text=True, start_new_session=True) + start_identity = capture_probe_identity(process) + try: + if start_identity is None: + raise CapabilityUnavailable("Council native bootstrap ownership identity is unavailable") + peer = JsonLinePeer(process.stdout, process.stdin, max_frame_bytes=16 * 1024) + peer.send(initialize_request(1)) + response = receive_response(peer, 1, timeout_seconds=3) + if "error" in response or not isinstance(response.get("result"), dict): + raise CapabilityUnavailable("Council native isolated bootstrap is unavailable") + except (OSError, EOFError, TimeoutError) as exc: + raise CapabilityUnavailable("Council native isolated bootstrap is unavailable") from exc + finally: + # The shared bounded helper is the sole cleanup authority. Failure + # to prove ownership or absence remains unavailable, never ready. + close_probe(process, start_identity=start_identity) + + +def verify_native_network(boundary: dict) -> None: + """Fail before generation until exact-boundary backend connectivity is proven. + + Initialization and model/list can use local/fallback catalog data. They are + deliberately not a network attestation. No flag or unsafe launch fallback + can override this currently unavailable native capability. + """ + raise CapabilityUnavailable( + "native Council is unavailable: a genuine non-generating HTTPS backend response " + "under the exact frozen sandbox boundary has not been verified" + ) diff --git a/plugin/core/src/devsquad/council_runtime.py b/plugin/core/src/devsquad/council_runtime.py new file mode 100644 index 0000000..24aee86 --- /dev/null +++ b/plugin/core/src/devsquad/council_runtime.py @@ -0,0 +1,452 @@ +"""Council stage projections over the existing durable runner and handoff store.""" +from __future__ import annotations + +from datetime import datetime, timezone +import hashlib +import json +from pathlib import Path +from typing import Any + +from .contracts import CapabilityUnavailable, ContractError +from .council import (ROLES, MAX_PACKET_BYTES, decode, digest, label_mapping, output_schema, + sanitized_proposals, validate_choice, validate_critique, validate_proposal, verify_identities) +from .store import ConflictError, canonical_json, request_hash + + +def frozen_fields(snapshot: dict) -> dict: + return {key: snapshot[key] for key in ("task", "base_oid", "target_oid", "configs", "routing", + "workspace", "check_workspace", "council_brief", "council_directories", "council_adapters", + "council_boundaries", "council_fixture")} + + +def verify_origin(store, run_id: str, snapshot: dict) -> None: + artifact = store.artifact_named(run_id, "council-origin.json") + if (artifact is None or hashlib.sha256(Path(artifact["path"]).read_bytes()).hexdigest() != artifact["sha256"] + or digest(frozen_fields(snapshot)) != snapshot.get("council_origin_sha256") + or artifact["sha256"] != snapshot["council_origin_sha256"]): + raise ConflictError("Council frozen inputs changed or are missing") + + +def verified_artifact(store, run_id: str, name: str, expected_sha256: str | None = None) -> bytes: + artifact = store.artifact_named(run_id, name) + if artifact is None: + raise ConflictError("Council evidence artifact is missing") + path = Path(artifact["path"]) + expected_parent = (store.artifacts / run_id).resolve() + if path.is_symlink() or path.resolve(strict=True).parent != expected_parent: + raise ConflictError("Council evidence artifact escaped its private directory") + raw = path.read_bytes() + if (len(raw) != artifact["byte_size"] or hashlib.sha256(raw).hexdigest() != artifact["sha256"] + or expected_sha256 is not None and artifact["sha256"] != expected_sha256): + raise ConflictError("Council evidence artifact is corrupt") + return raw + + +def validate_saved_handoff(store, run_id: str, handoff, snapshot: dict) -> None: + """Re-derive quorum/check truth from exact imported stage artifacts.""" + verify_origin(store, run_id, snapshot) + state = snapshot["council_state"] + if set(state["documents"]) != set(ROLES) or state["next_role"] != "lead": + raise ConflictError("Council handoff has no complete sealed proposal/critic quorum") + attempts = {a["id"]: a for a in store.attempts_for_run(run_id)} + for role in ROLES: + reference = state["artifacts"][role] + evidence = decode(verified_artifact(store, run_id, reference["name"], reference["sha256"])) + matching = [a for a in attempts.values() if reference["name"] == f"council-{role}-{a['id']}.json"] + if len(matching) != 1 or matching[0]["status"] != "finished" or evidence != state["documents"][role]: + raise ConflictError("Council finalized stage projection differs from its actual imported attempt") + validate_document(evidence, snapshot, matching[0]) + verify_identities(state["documents"], fixture=snapshot["council_fixture"] is not None) + mapping = label_mapping(snapshot["task"]["council"]["seed"], ["proposer_a", "proposer_b"]) + checks = state["documents"]["critic"]["checks"] + failures = [c["id"] for c in checks if c["required_to_pass"] and c["status"] != "passed"] + integrity = [c["id"] for c in checks if c.get("integrity", {}).get("status") in {"violated", "not_run"}] + packet = handoff.packet + if (digest(packet) != handoff.packet_sha256 or packet.get("workflow") != "council-decision" + or packet.get("brief_sha256") != digest(snapshot["council_brief"]) + or packet.get("candidate_sha256") != snapshot["workspace"]["candidate_sha256"] + or packet.get("base_oid") != snapshot["base_oid"] or packet.get("target_oid") != snapshot["target_oid"] + or packet.get("label_mapping") != mapping or packet.get("checks") != checks + or packet.get("critique") != state["documents"]["critic"]["document"] + or packet.get("proposals") != sanitized_proposals(snapshot) + or packet.get("evaluation") != {"accept_allowed": not failures and not integrity, + "required_failures": failures, "integrity_failures": integrity}): + raise ConflictError("Council saved handoff differs from frozen evidence/check truth") + for reference in packet["artifacts"]: + raw = verified_artifact(store, run_id, reference["name"], reference["sha256"]) + if reference["name"].startswith("council-evidence-") and decode(raw) != { + "documents": state["documents"], "label_mapping": mapping, "brief_sha256": digest(snapshot["council_brief"])}: + raise ConflictError("Council raw provenance bundle differs from finalized evidence") + + +def prepare(snapshot: dict, *, store, run_id: str, runtime: Path, fixture: dict | None) -> None: + from .codex_review_worker import freeze_codex_role + from .council_isolation import freeze_boundary, verify_read_boundary, verify_native_bootstrap + from .workspaces import committed_regular_file, _git + + task = snapshot["task"] + if fixture is not None and not isinstance(fixture, dict): + raise ContractError("Council fixture must be an object") + brief = {"schema_version": 1, "goal": task["goal"], "base_oid": snapshot["base_oid"], + "target_oid": snapshot["target_oid"], "scope": task["scope"], "acceptance": task["acceptance"], + "rubric": task["council"]["rubric"], "evidence": [], "source_files": []} + for ref in task["council"]["evidence"]: + row = store.connection.execute( + "SELECT a.*,r.project_id FROM artifacts a JOIN runs r ON r.id=a.run_id WHERE a.id=?", (ref["artifact_id"],) + ).fetchone() + if row is None or row["project_id"] != store.run(run_id)["project_id"] or row["sha256"] != ref["sha256"]: + raise ContractError("Council evidence is missing, belongs to another project, or has changed") + raw = Path(row["path"]).read_bytes() + if len(raw) != row["byte_size"] or hashlib.sha256(raw).hexdigest() != ref["sha256"]: + raise ConflictError("Council evidence artifact is corrupt") + try: + content = raw.decode("utf-8") + except UnicodeDecodeError as exc: + raise ContractError("Council evidence must be UTF-8 text") from exc + brief["evidence"].append({**ref, "content": content}) + repo = Path(task["project"]["repo_path"]) + files = _git(repo, "ls-tree", "-r", "--name-only", "-z", snapshot["target_oid"]).split(b"\0") + for encoded in files: + if not encoded: + continue + path = encoded.decode("utf-8", "surrogateescape") + if not any(scope == "." or path == scope or path.startswith(scope.rstrip("/") + "/") for scope in task["scope"]["read_paths"]): + continue + try: + raw = committed_regular_file(repo, snapshot["target_oid"], path) + content = raw.decode("utf-8") + except UnicodeDecodeError as exc: + raise ContractError("Council scoped source must be UTF-8; narrow the read scope") from exc + source_hash = hashlib.sha256(raw).hexdigest() + brief["source_files"].append({"path": path, "artifact_id": f"source:{path}:{source_hash}", "sha256": source_hash, "content": content}) + if len(canonical_json(brief).encode()) > MAX_PACKET_BYTES: + raise ContractError("Council evidence exceeds its cap; narrow the read scope") + if len(canonical_json(brief).encode()) > MAX_PACKET_BYTES: + raise ContractError("Council evidence exceeds its byte cap") + snapshot.update(council_brief=brief, council_directories={}, council_adapters={}, council_boundaries={}, + council_fixture=fixture, council_state={"documents": {}, "next_role": "proposer_a"}) + for role in (*ROLES, "lead"): + directory = runtime / "council" / run_id / role + directory.mkdir(parents=True, exist_ok=True) + directory.chmod(0o700) + snapshot["council_directories"][role] = str(directory.resolve()) + # A common-brief-only sentinel is also the actual per-role filesystem + # probe target. Peer proposals never enter another author's directory. + (directory / "brief.json").write_bytes(canonical_json(brief).encode()) + private_log = runtime / "private-logs" / f"{run_id}.boundary-probe" + private_log.parent.mkdir(parents=True, exist_ok=True) + private_log.write_bytes(b"Private coordinator boundary probe\n") + private_log.chmod(0o600) + prior_probe = store.artifact_named(run_id, "council-boundary-probe.json") + if prior_probe is None: + store.store_artifact(run_id, "council-boundary-probe.json", b'{"private_coordinator_probe":true}\n') + private_artifact = Path(store.artifact_named(run_id, "council-boundary-probe.json")["path"]) + for role in (*ROLES, "lead"): + directory = Path(snapshot["council_directories"][role]) + if fixture is None and (role != "lead" or task["lead"]["mode"] == "headless"): + routed = snapshot["routing"]["roles"][role] + snapshot["council_adapters"][role] = {} + snapshot["council_boundaries"][role] = {} + for selected in [routed["selected"], *routed["fallbacks"]]: + adapter = freeze_codex_role(selected, role=role, output_schema=output_schema(role)) + snapshot["council_adapters"][role][selected["profile_id"]] = adapter + boundary = freeze_boundary(executable=Path(adapter["binary"]), evidence=directory, native_codex=True) + forbidden = tuple(Path(path) / "brief.json" for peer, path in snapshot["council_directories"].items() if peer != role) + verify_read_boundary(boundary, own_file=directory / "brief.json", + forbidden=(*forbidden, runtime / "state.sqlite3", private_log, private_artifact)) + verify_native_bootstrap(boundary, executable=Path(adapter["binary"]), + expected_version=adapter["harness_version"], evidence=directory) + from .council_isolation import verify_native_network + verify_native_network(boundary) + snapshot["council_boundaries"][role][selected["profile_id"]] = {**boundary, "probe_status": "passed"} + snapshot["council_origin_sha256"] = digest(frozen_fields(snapshot)) + content = canonical_json(frozen_fields(snapshot)).encode() + prior = store.artifact_named(run_id, "council-origin.json") + if prior is None: + store.store_artifact(run_id, "council-origin.json", content) + elif prior["sha256"] != snapshot["council_origin_sha256"]: + raise ConflictError("recovered Council preparation changed frozen inputs") + + +def selected_for(snapshot: dict, role: str, index: int) -> dict: + route = snapshot["routing"]["roles"][role] + return [route["selected"], *route["fallbacks"]][index] + + +def validate_document(evidence: dict, snapshot: dict, attempt: dict) -> dict: + from .council import exact, text, PROMPT_VERSION, output_schema, prompt_digest + from .council_worker import role_packet + exact(evidence, {"schema_version", "role", "brief_sha256", "profile", "profile_sha256", "document", + "observed_identity", "identity_scope", "native_ids", "usage", "checks", "boundary_sha256", + "prompt_version", "prompt_sha256", "role_packet_sha256", "output_schema_sha256"}, "Council attempt") + role = attempt["role"] + selected = selected_for(snapshot, role, attempt["profile_index"]) + packet = role_packet(snapshot, role) + if (evidence["schema_version"] != 1 or evidence["role"] != role or evidence["profile"] != selected["profile"] + or evidence["profile_sha256"] != selected["profile_sha256"] + or attempt["profile_id"] != selected["profile_id"] + or evidence["brief_sha256"] != digest(snapshot["council_brief"]) + or evidence["prompt_version"] != PROMPT_VERSION + or evidence["prompt_sha256"] != prompt_digest(role, packet) + or evidence["role_packet_sha256"] != digest(packet) + or evidence["output_schema_sha256"] != digest(output_schema(role))): + raise ContractError("Council evidence differs from its actual frozen role/profile/brief") + fixture = snapshot["council_fixture"] is not None + identity = evidence["observed_identity"] + if fixture: + if (identity is not None or evidence["identity_scope"] != "all_fixture" or evidence["boundary_sha256"] is not None + or evidence["native_ids"] is not None or evidence["usage"] is not None): + raise ContractError("Council fixture cannot claim native identity/isolation") + else: + adapter = snapshot["council_adapters"][role][selected["profile_id"]] + exact(identity, {"harness", "harness_version", "model_provider", "model_id", "effort", "permission_policy", "verification"}, + "observed Council native identity") + expected_identity = {"harness": adapter["harness"], "harness_version": adapter["harness_version"], + "model_provider": adapter["model_provider"], "model_id": selected["profile"]["model_id"], + "effort": selected["profile"]["effort"]["value"], "permission_policy": "read_only", "verification": "verified"} + if (identity != expected_identity or evidence["identity_scope"] != "native_verified" + or evidence["boundary_sha256"] != snapshot["council_boundaries"][role][selected["profile_id"]]["profile_sha256"]): + raise ContractError("Council requires verified native identity and frozen read isolation") + exact(evidence["native_ids"], {"thread_id", "turn_id"}, "Council native IDs") + for key, value in evidence["native_ids"].items(): + text(value, f"Council native {key}") + if len(value) > 500: + raise ContractError("Council native IDs exceed their bound") + exact(evidence["usage"], {"input_tokens", "output_tokens", "total_tokens", "source"}, "Council native usage") + usage = evidence["usage"] + tokens = [usage[key] for key in ("input_tokens", "output_tokens", "total_tokens")] + if (usage["source"] not in {"native_reported", "unavailable"} + or any(value is not None and (type(value) is not int or value < 0) for value in tokens) + or usage["source"] == "unavailable" and any(value is not None for value in tokens) + or usage["source"] == "native_reported" and any(value is None for value in tokens)): + raise ContractError("Council native token usage is invalid or invents missing observations") + if not isinstance(evidence["checks"], list): + raise ContractError("Council trusted checks must be an array") + spec = {**snapshot["task"]["council"], "source_ids": [item["artifact_id"] for item in snapshot["council_brief"]["source_files"]]} + if role in {"proposer_a", "proposer_b"}: + validate_proposal(evidence["document"], spec) + if evidence["checks"]: + raise ContractError("Council proposers cannot supply trusted check results") + elif role == "critic": + validate_critique(evidence["document"], spec) + documents = {**snapshot["council_state"]["documents"], "critic": evidence} + verify_identities(documents, fixture=fixture) + from .workflows import validate_check_results + validate_check_results(evidence["checks"], snapshot["task"], snapshot["workspace"]) + elif role == "lead": + validate_choice(evidence["document"], spec, snapshot["council_state"]["documents"]["critic"]["document"]) + if evidence["document"]["disposition"] == "accept": + checks = snapshot["council_state"]["documents"]["critic"]["checks"] + if any(c["required_to_pass"] and c["status"] != "passed" or c.get("integrity", {}).get("status") in {"violated", "not_run"} for c in checks): + raise ContractError("Council lead cannot override failed mandatory checks/integrity") + else: + raise ContractError("Council attempt role is invalid") + return evidence + + +def add_artifact(store, run_id: str, name: str, value: Any) -> dict: + path, sha256, size = store.finalize_artifact(run_id, name, canonical_json(value).encode()) + return {"name": name, "path": path, "sha256": sha256, "byte_size": size} + + +def import_stage(store, run_id: str, attempt: dict, artifacts: list, metadata: dict, snapshot: dict, raw: bytes) -> str: + verify_origin(store, run_id, snapshot) + evidence = validate_document(decode(raw), snapshot, attempt) + role = attempt["role"] + name = f"council-{role}-{attempt['id']}.json" + artifacts.append(add_artifact(store, run_id, name, evidence)) + if role == "lead": + # The existing headless import/reopen fence remains the single lead authority. + artifacts.append(add_artifact(store, run_id, f"lead-attempt-{attempt['id']}.json", evidence)) + return store.commit_headless_lead(run_id, attempt["attempt_token"], artifacts, metadata) + updated = json.loads(canonical_json(snapshot)) + state = updated["council_state"] + state["documents"][role] = evidence + state.setdefault("artifacts", {})[role] = {"name": name, "sha256": artifacts[-1]["sha256"]} + if role != "critic": + state["next_role"] = "proposer_b" if role == "proposer_a" else "critic" + return store.commit_council_stage(run_id, attempt["attempt_token"], artifacts, metadata, updated, role) + # Save the critic projection before publishing a handoff in the same transaction. + state["next_role"] = "lead" + mapping = label_mapping(snapshot["task"]["council"]["seed"], ["proposer_a", "proposer_b"]) + checks = evidence["checks"] + failures = [c["id"] for c in checks if c["required_to_pass"] and c["status"] != "passed"] + integrity = [c["id"] for c in checks if c.get("integrity", {}).get("status") in {"violated", "not_run"}] + packet = {"schema_version": 1, "workflow": "council-decision", "candidate_sha256": snapshot["workspace"]["candidate_sha256"], + "base_oid": snapshot["base_oid"], "target_oid": snapshot["target_oid"], "brief_sha256": evidence["brief_sha256"], + "proposals": sanitized_proposals(updated), + "label_mapping": mapping, "critique": evidence["document"], "checks": checks, + "evaluation": {"accept_allowed": not failures and not integrity, "required_failures": failures, + "integrity_failures": integrity}, + "artifacts": [], "instructions": "The sole lead must retain objections and validation; votes cannot override failed checks.", + "identity_scope": "all_fixture" if snapshot["council_fixture"] is not None else "native_verified"} + bundle = {"documents": state["documents"], "label_mapping": mapping, "brief_sha256": evidence["brief_sha256"]} + bundled = add_artifact(store, run_id, f"council-evidence-{attempt['id']}.json", bundle) + artifacts.append(bundled) + packet["artifacts"] = [{"name": a["name"], "sha256": a["sha256"]} for a in (artifacts[-2], bundled)] + return store.commit_durable_handoff(run_id, attempt["attempt_token"], artifacts, metadata, packet, + mutable_snapshot=updated) + + +def reports(store, run_id: str, snapshot: dict | None, state: str, *, choice: dict | None = None, error: dict | None = None) -> dict[str, bytes]: + from .reports import _contents_with_manifest, _event_export + run = store.run(run_id) + snapshot = snapshot or {} + documents = snapshot.get("council_state", {}).get("documents", {}) + attempts = [] + for attempt in store.attempts_for_run(run_id): + role = attempt["role"] + saved = store.artifact_named(run_id, f"council-{role}-{attempt['id']}.json") + evidence = None + if saved is not None: + raw = Path(saved["path"]).read_bytes() + if hashlib.sha256(raw).hexdigest() != saved["sha256"]: + raise ConflictError("Council attempt evidence is corrupt") + evidence = decode(raw) + ended = datetime.fromisoformat(attempt["finished_at"]) if attempt["finished_at"] else datetime.now(timezone.utc) + latency = max(0, int((ended - datetime.fromisoformat(attempt["created_at"])).total_seconds() * 1000)) if attempt["pid"] is not None else None + pending = attempt["status"] in {"running", "cancelling"} and state in {"failed", "cancelled"} + metadata = json.loads(attempt["output_metadata"] or "{}") + attempts.append({"id": attempt["id"], "role": role, + "status": state if pending else "failed" if metadata.get("failure") else attempt["status"], + "profile_id": attempt["profile_id"], "profile_index": attempt["profile_index"], + "requested_profile": selected_for(snapshot, role, attempt["profile_index"])["profile"] if "routing" in snapshot else None, + "observed_identity": evidence.get("observed_identity") if evidence else None, + "usage": evidence.get("usage") if evidence else None, + "native_ids": evidence.get("native_ids") if evidence else None, "native_model_requests": None, + "worker_invocations": 1 if attempt["pid"] is not None else 0, + "policy_sha256": snapshot.get("routing", {}).get("policy", {}).get("sha256"), + "runtime_package_sha256": run.get("package_digest"), "prompt_version": "council-role-packet-v1", + "prompt_sha256": evidence.get("prompt_sha256") if evidence else None, + "role_packet_sha256": evidence.get("role_packet_sha256") if evidence else None, + "output_schema_sha256": evidence.get("output_schema_sha256") if evidence else None, + "output_sha256": saved["sha256"] if saved else None, + "latency_ms": latency, "error": error if pending else metadata.get("failure")}) + artifacts = store.artifacts_for_run(run_id) + receipt = {"schema_version": 1, "run_id": run_id, "workflow": "council-decision", "state": state, + "candidate": {"sha256": snapshot.get("workspace", {}).get("candidate_sha256"), + "base_oid": snapshot.get("base_oid"), "target_oid": snapshot.get("target_oid")}, + "completed_at": datetime.now(timezone.utc).isoformat(), "goal": snapshot.get("task", {}).get("goal"), + "candidate_sha256": snapshot.get("workspace", {}).get("candidate_sha256"), + "brief_sha256": digest(snapshot["council_brief"]) if "council_brief" in snapshot else None, + "attempts": attempts, "criteria": snapshot.get("task", {}).get("acceptance", []), + "lead": {"disposition": choice.get("disposition") if choice else None, "choice": choice, + "work_source": "headless" if snapshot.get("task", {}).get("lead", {}).get("mode") == "headless" else "host", + "usage": None}, "dissent": documents.get("critic", {}).get("document", {}).get("objections", []), + "documents": documents, "error": error, "automatic_enabled": False, + "dispositions": [choice] if choice else [], + "identity_scope": ("all_fixture" if snapshot.get("council_fixture") is not None else + "native_verified" if len(documents) == 3 else "incomplete"), + "worker_invocations": store.worker_invocations(run_id), + "execution_elapsed_ms": store._execution_elapsed_ms(run_id, run, datetime.now(timezone.utc)), + "limitations": ["Partial anonymity", "Fixture evidence establishes mechanics only"] if snapshot.get("council_fixture") is not None else ["Partial anonymity"]} + events, _, _ = _event_export(store.events_for_run(run_id)) + projected = [{key: a[key] for key in ("id", "name", "sha256", "byte_size")} for a in artifacts] + markdown = f"# DevSquad Council\n\nRun: {run_id}\nState: {state}\nAutomatic triggering: off\n\n" + canonical_json(choice or error or {}) + "\n\nDissent:\n" + canonical_json(receipt["dissent"]) + "\n" + return _contents_with_manifest(run_id, receipt, markdown, events, projected) + + +def prepared_reports(store, run_id: str, snapshot: dict | None, state: str, **kwargs) -> list: + result = [] + for name, content in reports(store, run_id, snapshot, state, **kwargs).items(): + path, sha256, size = store.finalize_artifact(run_id, name, content) + result.append({"name": name, "path": path, "sha256": sha256, "byte_size": size}) + return result + + +def decision_gate(store, run_id: str, handoff, snapshot: dict, decision: dict) -> dict: + from .workflows import validate_handoff_decision_evidence + validate_saved_handoff(store, run_id, handoff, snapshot) + validate_handoff_decision_evidence(decision, handoff.packet) + choice = decision.get("council_choice") + validate_choice(choice, snapshot["task"]["council"], handoff.packet["critique"]) + if choice["disposition"] != decision["disposition"] or choice["reason"] != decision["reason"]: + raise ContractError("Council choice contradicts the lead disposition") + if decision["disposition"] == "accept" and handoff.packet["evaluation"]["accept_allowed"] is not True: + raise ContractError("Council acceptance is blocked by mandatory checks/integrity") + if decision["disposition"] == "revise": + raise ContractError("Council additional rounds require a new explicitly capped run") + return choice + + +def finish_recorded(store, run_id: str, handoff, snapshot: dict, decision: dict) -> dict: + choice = decision_gate(store, run_id, handoff, snapshot, decision) + entry = store.recorded_handoff_submission(run_id, handoff.handoff_id) + if entry is None or entry["decision"] != decision: + raise ConflictError("Council recorded decision differs from continuation") + if store.run(run_id)["state"] in {"succeeded", "failed"}: + return {"run_id": run_id, "state": store.run(run_id)["state"], "replayed": True, + "disposition": decision["disposition"], "launched": False} + terminal = "succeeded" if decision["disposition"] == "accept" else "failed" + artifacts = prepared_reports(store, run_id, snapshot, terminal, choice=choice) + result = store.complete_handoff_terminal(run_id, handoff.handoff_id, decision["submission_id"], + decision["submission_hash"], artifacts, terminal, {"disposition": decision["disposition"], "receipt": "result-receipt.json"}) + return {"run_id": run_id, **result, "disposition": decision["disposition"], "launched": False} + + +def complete(service, store, run_id: str, handoff, snapshot: dict, decision: dict) -> dict: + claim = service._decode_claim(decision.pop("_claim")) + decision_gate(store, run_id, handoff, snapshot, decision) + submission = store.record_handoff_submission(run_id, claim, decision) + result = finish_recorded(store, run_id, handoff, snapshot, decision) + result["replayed"] = submission.replayed + return result + + +def saved_claim(store, run_id: str, owner: str): + """Reuse exactly our durable live claim, or reacquire only our expired one.""" + handoff = store.handoff_snapshot(run_id) + prior = handoff.claim + if prior is not None and prior.owner_id != owner: + raise ConflictError("Council continuation cannot take another host's claim") + if prior is not None and datetime.now(timezone.utc) < datetime.fromisoformat(prior.expires_at): + return prior + return store.claim_handoff(run_id, store.run(run_id)["version"], owner) + + +def continue_host_intent(service, store, run_id: str, handoff, snapshot: dict) -> dict: + decision = store.terminal_finish_decision(run_id, handoff.handoff_id) + if decision is None: + raise ConflictError("Council host continuation has no exact current guided-finish claim marker") + decision_gate(store, run_id, handoff, snapshot, decision) + # Shared R6 authority reuses only the latest canonical guided marker. An + # expired identical intent gets a fresh fence; same-name app claims do not. + claim = store.claim_handoff(run_id, store.run(run_id)["version"], "terminal-operator", + initial_only=True, terminal_decision=decision) + return complete(service, store, run_id, handoff, snapshot, + {**decision, "_claim": service._claim_payload(claim)}) + + +def continue_lead(service, store, run_id: str, run: dict, handoff, snapshot: dict) -> dict: + entry = store.recorded_handoff_submission(run_id, handoff.handoff_id) + if entry is not None: + return finish_recorded(store, run_id, handoff, snapshot, entry["decision"]) + if snapshot["task"]["lead"]["mode"] != "headless": + return continue_host_intent(service, store, run_id, handoff, snapshot) + verify_origin(store, run_id, snapshot) + leads = [a for a in store.attempts_for_run(run_id) if a["role"] == "lead"] + for attempt in reversed(leads): + artifact = store.artifact_named(run_id, f"lead-attempt-{attempt['id']}.json") + if artifact is None: + continue + raw = Path(artifact["path"]).read_bytes() + if hashlib.sha256(raw).hexdigest() != artifact["sha256"]: + raise ConflictError("Council lead artifact is corrupt") + evidence = validate_document(decode(raw), snapshot, attempt) + choice = evidence["document"] + claim = saved_claim(store, run_id, f"headless-lead:{attempt['id']}") + body = {"schema_version": 1, "submission_id": f"headless-{attempt['id']}-{claim.fencing_token}", + "disposition": choice["disposition"], "reason": choice["reason"], "council_choice": choice, + "evidence_refs": [{"artifact_id": ref["artifact_id"], "sha256": ref["sha256"]} for ref in handoff.packet["artifacts"]]} + return complete(service, store, run_id, handoff, snapshot, + {**body, "submission_hash": request_hash(body), "_claim": service._claim_payload(claim)}) + queued = store.queue_headless_lead(run_id, run["version"]) + if queued["action"] == "budget_exhausted": + error = {"error": "BUDGET_EXHAUSTED", "message": "Council lead budget exhausted"} + prepared = prepared_reports(store, run_id, snapshot, "failed", error=error) + store.fail_queued_budget(run_id, run["version"], error, prepared) + return {"run_id": run_id, "state": "failed", "launched": False} + current = store.run(run_id) + package, package_digest = service._verified_package(current) + return {"run_id": run_id, "state": current["state"], "launched": True, + "launch": (current["version"], package, package_digest)} diff --git a/plugin/core/src/devsquad/council_task_entry.py b/plugin/core/src/devsquad/council_task_entry.py new file mode 100644 index 0000000..7538154 --- /dev/null +++ b/plugin/core/src/devsquad/council_task_entry.py @@ -0,0 +1,74 @@ +"""Manual, bounded Council command preparation; automatic triggering is off.""" +from __future__ import annotations + +from pathlib import Path + +from .contracts import CapabilityUnavailable, ContractError +from .council import ROLES, digest, validate_spec +from .task_entry import _relative_path, _resolve_commit, resolve_repository +from .validation import validate_task + + +def build_council_task(*, project_dir: Path, goal: str, identities: list[dict], + base_ref: str = "HEAD", target_ref: str = "HEAD", read_paths=(), checks=(), + lead_mode: str = "headless", max_invocations: int = 4, evidence=(), rubric=None) -> tuple[dict, dict]: + if not isinstance(goal, str) or not goal.strip() or len(goal) > 16000: + raise ContractError("Council question must be non-empty and bounded") + if (len(identities) != 3 or any(not isinstance(value, dict) or value.get("harness") != "codex" for value in identities) + or len({value.get("model_id", "").casefold() for value in identities}) != 3): + raise CapabilityUnavailable("Council needs three distinct entitled, verified Codex model IDs") + repo = resolve_repository(project_dir) + base, target = _resolve_commit(repo, base_ref), _resolve_commit(repo, target_ref) + paths = list(dict.fromkeys(_relative_path(path, "Council read path") for path in read_paths)) + profiles = [] + for role, identity in zip(ROLES, identities): + if any(not isinstance(identity.get(key), str) or not identity[key] for key in ("model_id", "model_family", "effort", "harness_version")): + raise ContractError("Council native catalog identity is incomplete") + profile = {"id": f"managed-council-{role}", "harness": "codex", "model_family": identity["model_family"], + "model_id": identity["model_id"], "effort": {"value": identity["effort"], "transport": "native"}, + "required_tools": [], "permission_policy": "read_only", "account_pool_id": identity.get("account_pool_id", "codex-subscription"), + "billing_mode": "subscription", "quality_status": "trial", + "evidence_refs": [f"runtime-catalog:{identity['harness_version']}:{identity['model_id']}", + *([f"runtime-catalog-fingerprint:{identity['catalog_fingerprint']}"] if "catalog_fingerprint" in identity else []), + *([f"runtime-native-scope:{identity['native_scope']}"] if "native_scope" in identity else [])]} + profile["id"] += "-" + digest(profile)[:16] + profiles.append(profile) + rubric = rubric or [{"id": "correctness", "description": "Claims fit the frozen evidence"}, + {"id": "risk", "description": "Important risks and objections are retained"}, + {"id": "validation", "description": "Decision names objective validation and uncertainty"}] + seed = digest({"goal": goal.strip(), "base_oid": base, "target_oid": target, "read_paths": paths, + "evidence": list(evidence), "rubric": rubric, "profiles": profiles}) + council = {"schema_version": 1, "enabled": True, "automatic": False, + "reason": "Explicit manual Council request", "min_valid_proposals": 2, "required_critics": 1, + "max_invocations": max_invocations, "seed": seed, "evidence": list(evidence), "rubric": rubric} + validate_spec(council) + roles = {role: [{"kind": "profile", "id": profile["id"]}] for role, profile in zip(ROLES, profiles)} + if lead_mode == "headless": + roles["lead"] = [{"kind": "profile", "id": profiles[-1]["id"]}] + declared_checks = [{"id": "candidate-integrity", "argv": ["git", "diff", "--check", base, target], + "cwd": ".", "timeout_seconds": 30, "required_to_pass": True}] + declared_checks.extend({"id": f"user-check-{index + 1}", "argv": list(argv), "cwd": ".", + "timeout_seconds": 60, "required_to_pass": True} for index, argv in enumerate(checks)) + task = {"schema_version": 1, "project": {"repo_path": str(repo), "base_ref": base, "target_ref": target}, + "workflow": "council-decision", "goal": goal.strip(), "task_class": "managed-council", + "acceptance": [{"id": item["id"], "description": item["description"], "evidence_kind": "review"} for item in rubric], + "checks": declared_checks, "scope": {"read_paths": paths, "write_paths": []}, "lead": {"mode": lead_mode}, + "routing": {"profiles": {"schema_version": 1, "profiles": profiles, "bindings": {}}, + "policy": {"schema_version": 1, "id": "managed-manual-council", "version": 1, + "roles": roles, "task_classes": {"managed-council": "trial"}, "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, "account_pools": {profile["account_pool_id"]: { + "allowed_billing_modes": ["subscription"], "max_concurrency": 1, "unknown_capacity_policy": "allow_bounded"} for profile in profiles}, + "experiment_budget": {}, "decision_helper": {"schema_version": 1, "mode": "off"}}}, + "budget": {"wall_seconds": 600, "max_worker_invocations": max_invocations, "max_revisions": 0, "max_fallbacks_per_step": 0}, + "origin": {"surface": "cli"}, "council": council} + validate_task(task, require_existing_repo=True) + summary = {"workflow": "council-decision", "project": str(repo), "base_oid": base, "target_oid": target, "goal": task["goal"], + "profiles": {role: {"profile_id": profile["id"], "model_id": profile["model_id"], "quality_status": "trial"} for role, profile in zip(ROLES, profiles)}, + "planned_roles": {role: {"profile_id": profile["id"], "harness": "codex", "model_id": profile["model_id"], + "effort": profile["effort"]["value"], "selection_mode": "catalog_trial", "quality_status": "trial"} for role, profile in zip(ROLES, profiles)}, + "selection_reason": "Three distinct available catalog identities; actual entitlement and identity must be verified before quorum; no quality qualification implied", + "scope": task["scope"], "checks": [item["id"] for item in declared_checks], "check_plan": declared_checks, "task_sha256": digest(task), + "budget": task["budget"], "lead": task["lead"], "automatic_enabled": False, + "native_ready": False, + "limitations": ["Manual only", "One round", "Trial catalog profiles are not quality proof", "Requires verified macOS default-deny isolation"]} + return task, summary diff --git a/plugin/core/src/devsquad/council_worker.py b/plugin/core/src/devsquad/council_worker.py new file mode 100644 index 0000000..99acada --- /dev/null +++ b/plugin/core/src/devsquad/council_worker.py @@ -0,0 +1,97 @@ +"""One Council role, executed inside the existing gated durable attempt.""" +from __future__ import annotations + +import json +from pathlib import Path +import sys +import time + +from .contracts import ContractError +from .council import MAX_PACKET_BYTES, PROMPT_VERSION, digest, sanitized_proposals, output_schema, role_prompt, prompt_digest +from .council_runtime import selected_for +from .store import canonical_json + + +def role_packet(snapshot: dict, role: str) -> dict: + packet = {"brief": snapshot["council_brief"]} + state = snapshot["council_state"] + if role in {"critic", "lead"}: + if not {"proposer_a", "proposer_b"}.issubset(state["documents"]): + raise ContractError("Council proposal barrier has not completed") + packet["proposals"] = sanitized_proposals(snapshot) + if role == "lead": + packet["critique"] = state["documents"]["critic"]["document"] + packet["checks"] = state["documents"]["critic"]["checks"] + return packet + + +def run(snapshot: dict) -> dict: + role = snapshot["council_role"] + selected = selected_for(snapshot, role, snapshot["council_profile_index"]) + packet = role_packet(snapshot, role) + directory = Path(snapshot["council_directories"][role]) + # The role directory contains only its frozen/sanitized packet. Coordinator + # logs, ledger, raw provenance and all peer directories are outside the jail. + packet_path = directory / "evidence.json" + encoded = canonical_json(packet).encode() + if len(encoded) > MAX_PACKET_BYTES: + raise ContractError("Council role packet exceeds its byte cap") + if packet_path.exists() and packet_path.read_bytes() != encoded: + raise ContractError("Council role packet changed after finalization") + packet_path.write_bytes(encoded) + fixture = snapshot["council_fixture"] + boundary_sha256 = None + if fixture is not None: + value = fixture.get(role) + if not isinstance(value, dict): + raise ContractError(f"Council fixture is missing {role}") + if value.get("delay_seconds"): + time.sleep(value["delay_seconds"]) + if value.get("error"): + raise ContractError(value["error"]) + result = {"document": value.get("document"), "observed_identity": None, "native_ids": None, "usage": None} + identity_scope = "all_fixture" + else: + from .codex_review_worker import run as native_turn + adapter = snapshot["council_adapters"][role][selected["profile_id"]] + boundary = snapshot["council_boundaries"][role][selected["profile_id"]] + native = {"task": snapshot["task"], "workspace": {"path": str(directory)}, + "routing": {"roles": {role: {"selected": selected}}}, + "council_adapter": adapter, "council_boundary": boundary} + prompt = role_prompt(role, packet) + result = native_turn(native, council_role=role, council_prompt=prompt) + identity_scope = "native_verified" + boundary_sha256 = boundary["profile_sha256"] + checks = [] + if role == "critic": + from .review_worker import run_review_and_checks + from .workflows import review_mode + # Reuse the trusted, isolated candidate-check path. This temporary + # internal review document is only a check driver, never Council review evidence. + check_snapshot = json.loads(canonical_json(snapshot)) + check_snapshot["task"]["workflow"] = "branch-review" + check_snapshot["task"].pop("council", None) + check_snapshot["routing"]["roles"]["reviewer"] = {"selected": selected} + workspace = snapshot["workspace"] + driver = {"schema_version": 1, "candidate_sha256": workspace["candidate_sha256"], + "base_oid": workspace["base_oid"], "target_oid": workspace["target_oid"], + "review_mode": review_mode(check_snapshot["task"]), "verdict": "clean", "summary": "Trusted checks only", "findings": []} + checks = run_review_and_checks(check_snapshot, driver)["checks"] + return {"schema_version": 1, "role": role, "brief_sha256": digest(snapshot["council_brief"]), + "prompt_version": PROMPT_VERSION, "prompt_sha256": prompt_digest(role, packet), + "role_packet_sha256": digest(packet), "output_schema_sha256": digest(output_schema(role)), + "profile": selected["profile"], "profile_sha256": selected["profile_sha256"], + **result, "identity_scope": identity_scope, "checks": checks, "boundary_sha256": boundary_sha256} + + +def main() -> int: + payload = sys.stdin.buffer.read(4 * MAX_PACKET_BYTES + 1) + if len(payload) > 4 * MAX_PACKET_BYTES: + raise ContractError("Council worker snapshot exceeds its byte cap") + snapshot = json.loads(payload.decode("utf-8")) + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/decision.py b/plugin/core/src/devsquad/decision.py new file mode 100644 index 0000000..40f723a --- /dev/null +++ b/plugin/core/src/devsquad/decision.py @@ -0,0 +1,437 @@ +"""Optional typed decision-helper contracts with no routing authority expansion.""" + +from __future__ import annotations + +import hashlib +import json +import math +from typing import Any + +from .contracts import ContractError +from .store import canonical_json + + +SHA256_LENGTH = 64 +ROLES = {"implementer", "reviewer", "lead", "researcher"} +OFF_FIELDS = {"schema_version", "mode"} +ENABLED_FIELDS = { + "schema_version", "mode", "purpose", "adapter", "language", + "min_confidence", "gate_evidence_sha256", "budget", +} +REQUEST_FIELDS = { + "schema_version", "purpose", "scope_sha256", "evidence", + "candidate_catalog_sha256", "candidates", "pins", "adapter", "language", + "truncation", +} +RESPONSE_FIELDS = { + "schema_version", "request_sha256", "adapter", "language", "truncation", + "recommendations", "usage", "elapsed_ms", +} + + +def _exact(value: Any, fields: set[str], label: str) -> dict[str, Any]: + if not isinstance(value, dict) or set(value) != fields: + raise ContractError(f"{label} fields are invalid") + return value + + +def _identifier(value: Any, field: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ContractError(f"decision helper {field} must be a non-empty string") + return value + + +def _sha256(value: Any, field: str, *, nullable: bool = False) -> str | None: + if value is None and nullable: + return None + if (not isinstance(value, str) or len(value) != SHA256_LENGTH + or any(character not in "0123456789abcdef" for character in value)): + raise ContractError(f"decision helper {field} must be a SHA256 digest") + return value + + +def _finite_probability(value: Any, field: str) -> float: + if (isinstance(value, bool) or not isinstance(value, (int, float)) + or not math.isfinite(value) or not 0.0 <= float(value) <= 1.0): + raise ContractError(f"decision helper {field} must be a finite probability") + return float(value) + + +def _adapter(value: Any) -> dict[str, Any]: + adapter = _exact( + value, {"id", "model", "runtime_revision", "calibration_version"}, + "decision helper adapter", + ) + for field in ("id", "model", "runtime_revision"): + _identifier(adapter[field], f"adapter.{field}") + calibration = adapter["calibration_version"] + if calibration is not None: + _identifier(calibration, "adapter.calibration_version") + return dict(adapter) + + +def validate_decision_policy(value: dict[str, Any] | None) -> dict[str, Any]: + """Validate the optional reviewed policy; omission is exactly mode off.""" + if value is None: + return {"schema_version": 1, "mode": "off"} + if not isinstance(value, dict): + raise ContractError("decision helper policy must be an object") + mode = value.get("mode") + fields = OFF_FIELDS if mode == "off" else ENABLED_FIELDS + _exact(value, fields, "decision helper policy") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("decision helper policy schema_version is invalid") + if mode not in {"off", "shadow", "advisory"}: + raise ContractError("decision helper mode is invalid") + if mode == "off": + return {"schema_version": 1, "mode": "off"} + purpose = _exact( + value["purpose"], + {"id", "version", "question_sha256", "rubric_sha256"}, + "decision helper purpose", + ) + _identifier(purpose["id"], "purpose.id") + if type(purpose["version"]) is not int or purpose["version"] < 1: + raise ContractError("decision helper purpose.version is invalid") + _sha256(purpose["question_sha256"], "purpose.question_sha256") + _sha256(purpose["rubric_sha256"], "purpose.rubric_sha256") + adapter = _adapter(value["adapter"]) + language = _identifier(value["language"], "language") + minimum = _finite_probability(value["min_confidence"], "min_confidence") + gate = _sha256( + value["gate_evidence_sha256"], "gate_evidence_sha256", nullable=True, + ) + if mode == "advisory" and gate is None: + raise ContractError("advisory mode requires reviewed gate evidence") + if mode == "shadow" and gate is not None: + raise ContractError("shadow mode cannot claim advisory gate evidence") + budget = _exact( + value["budget"], + {"max_calls", "max_input_bytes", "wall_seconds", "max_cost_usd"}, + "decision helper budget", + ) + for field in ("max_calls", "max_input_bytes", "wall_seconds"): + if type(budget[field]) is not int or budget[field] < 1: + raise ContractError(f"decision helper budget.{field} is invalid") + if budget["max_calls"] != 1: + raise ContractError("decision helper v1 permits exactly one call") + cost = budget["max_cost_usd"] + if (isinstance(cost, bool) or not isinstance(cost, (int, float)) + or not math.isfinite(cost) or cost < 0): + raise ContractError("decision helper budget.max_cost_usd is invalid") + return json.loads(canonical_json({ + **value, + "purpose": dict(purpose), + "adapter": adapter, + "language": language, + "min_confidence": minimum, + "gate_evidence_sha256": gate, + "budget": dict(budget), + })) + + +def build_decision_request( + task: dict[str, Any], + routing: dict[str, Any], + policy: dict[str, Any], + evidence_payload: bytes | str, + *, + truncated: bool = False, +) -> dict[str, Any] | None: + """Build a cache-complete request containing hashes, never raw task text.""" + config = validate_decision_policy(policy.get("decision_helper")) + if config["mode"] == "off": + return None + if isinstance(evidence_payload, str): + evidence = evidence_payload.encode() + elif isinstance(evidence_payload, bytes): + evidence = evidence_payload + else: + raise ContractError("decision helper evidence payload must be bytes or text") + if len(evidence) > config["budget"]["max_input_bytes"] and not truncated: + raise ContractError("decision helper input exceeds its frozen byte budget") + candidates = {} + pins = {} + for role, routed in routing["roles"].items(): + ordered = [routed["selected"], *routed.get("fallbacks", [])] + identifiers = [candidate["profile_id"] for candidate in ordered] + if len(identifiers) != len(set(identifiers)): + raise ContractError("decision helper candidates must be unique") + candidates[role] = identifiers + if routed["source"] == "override": + pins[role] = routed["selected"]["profile_id"] + request = { + "schema_version": 1, + "purpose": config["purpose"], + "scope_sha256": hashlib.sha256( + canonical_json(task["scope"]).encode(), + ).hexdigest(), + "evidence": { + "sha256": hashlib.sha256(evidence).hexdigest(), + "byte_size": len(evidence), + }, + "candidate_catalog_sha256": routing["profile_registry"]["sha256"], + "candidates": candidates, + "pins": pins, + "adapter": config["adapter"], + "language": config["language"], + "truncation": { + "occurred": bool(truncated), + "limit_bytes": config["budget"]["max_input_bytes"], + }, + } + return validate_decision_request(request) + + +def validate_decision_request(value: dict[str, Any]) -> dict[str, Any]: + _exact(value, REQUEST_FIELDS, "decision helper request") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("decision helper request schema_version is invalid") + purpose = _exact( + value["purpose"], + {"id", "version", "question_sha256", "rubric_sha256"}, + "decision helper request purpose", + ) + _identifier(purpose["id"], "request purpose.id") + if type(purpose["version"]) is not int or purpose["version"] < 1: + raise ContractError("decision helper request purpose.version is invalid") + _sha256(purpose["question_sha256"], "request purpose.question_sha256") + _sha256(purpose["rubric_sha256"], "request purpose.rubric_sha256") + _sha256(value["scope_sha256"], "request scope_sha256") + _sha256(value["candidate_catalog_sha256"], "request candidate_catalog_sha256") + evidence = _exact( + value["evidence"], {"sha256", "byte_size"}, "decision helper evidence", + ) + _sha256(evidence["sha256"], "request evidence.sha256") + if type(evidence["byte_size"]) is not int or evidence["byte_size"] < 0: + raise ContractError("decision helper evidence.byte_size is invalid") + candidates = value["candidates"] + if (not isinstance(candidates, dict) or not candidates + or set(candidates) - ROLES): + raise ContractError("decision helper candidates are invalid") + normalized_candidates = {} + for role, identifiers in candidates.items(): + if (not isinstance(identifiers, list) or not identifiers + or len(identifiers) != len(set(identifiers)) + or any(not isinstance(item, str) or not item for item in identifiers)): + raise ContractError("decision helper candidate IDs are invalid") + normalized_candidates[role] = list(identifiers) + pins = value["pins"] + if not isinstance(pins, dict) or set(pins) - set(normalized_candidates): + raise ContractError("decision helper pins are invalid") + for role, profile_id in pins.items(): + if profile_id not in normalized_candidates[role]: + raise ContractError("decision helper pin is outside eligible candidates") + truncation = _exact( + value["truncation"], {"occurred", "limit_bytes"}, + "decision helper request truncation", + ) + if type(truncation["occurred"]) is not bool: + raise ContractError("decision helper truncation flag is invalid") + if type(truncation["limit_bytes"]) is not int or truncation["limit_bytes"] < 1: + raise ContractError("decision helper truncation limit is invalid") + return json.loads(canonical_json({ + **value, + "purpose": dict(purpose), + "evidence": dict(evidence), + "candidates": normalized_candidates, + "pins": dict(pins), + "adapter": _adapter(value["adapter"]), + "language": _identifier(value["language"], "request language"), + "truncation": dict(truncation), + })) + + +def decision_cache_key(request: dict[str, Any]) -> str: + normalized = validate_decision_request(request) + return hashlib.sha256(canonical_json(normalized).encode()).hexdigest() + + +def validate_decision_response( + request: dict[str, Any], + response: dict[str, Any], + config: dict[str, Any], +) -> dict[str, Any]: + """Validate IDs, distributions, usage and provider identity strictly.""" + request = validate_decision_request(request) + config = validate_decision_policy(config) + _exact(response, RESPONSE_FIELDS, "decision helper response") + if type(response["schema_version"]) is not int or response["schema_version"] != 1: + raise ContractError("decision helper response schema_version is invalid") + expected_sha = hashlib.sha256(canonical_json(request).encode()).hexdigest() + if response["request_sha256"] != expected_sha: + raise ContractError("decision helper response does not bind the request") + adapter = _adapter(response["adapter"]) + if adapter != request["adapter"] or adapter != config["adapter"]: + raise ContractError("decision helper observed adapter identity drifted") + language = _identifier(response["language"], "response language") + if language != request["language"]: + raise ContractError("decision helper response language drifted") + truncation = _exact( + response["truncation"], {"occurred", "detail"}, + "decision helper response truncation", + ) + if type(truncation["occurred"]) is not bool: + raise ContractError("decision helper response truncation flag is invalid") + if truncation["detail"] is not None: + _identifier(truncation["detail"], "response truncation.detail") + recommendations = response["recommendations"] + if not isinstance(recommendations, dict) or set(recommendations) != set(request["candidates"]): + raise ContractError("decision helper recommendation roles are invalid") + normalized_recommendations = {} + for role, candidate_ids in request["candidates"].items(): + item = _exact( + recommendations[role], + {"ranking", "probabilities", "confidence", "abstain_reason"}, + "decision helper recommendation", + ) + ranking = item["ranking"] + if (not isinstance(ranking, list) or len(ranking) != len(candidate_ids) + or len(ranking) != len(set(ranking)) + or set(ranking) != set(candidate_ids)): + raise ContractError("decision helper ranking changes the eligible set") + probabilities = item["probabilities"] + if not isinstance(probabilities, dict) or set(probabilities) != set(candidate_ids): + raise ContractError("decision helper probability IDs are invalid") + normalized_probabilities = { + profile_id: _finite_probability( + probabilities[profile_id], f"probability.{profile_id}", + ) + for profile_id in candidate_ids + } + if not math.isclose( + sum(normalized_probabilities.values()), 1.0, abs_tol=0.02): + raise ContractError("decision helper probabilities do not sum to one") + confidence = _finite_probability(item["confidence"], "confidence") + abstain = item["abstain_reason"] + if abstain is not None: + abstain = _identifier(abstain, "abstain_reason") + normalized_recommendations[role] = { + "ranking": list(ranking), + "probabilities": normalized_probabilities, + "confidence": confidence, + "abstain_reason": abstain, + } + usage = _exact( + response["usage"], + {"source", "billable_requests", "input_tokens", "output_tokens", "cost_usd"}, + "decision helper usage", + ) + if usage["source"] not in {"native_reported", "fake", "unavailable"}: + raise ContractError("decision helper usage source is invalid") + if (type(usage["billable_requests"]) is not int + or not 0 <= usage["billable_requests"] <= config["budget"]["max_calls"]): + raise ContractError("decision helper billable request count is invalid") + for field in ("input_tokens", "output_tokens"): + if usage[field] is not None and ( + type(usage[field]) is not int or usage[field] < 0): + raise ContractError(f"decision helper usage {field} is invalid") + cost = usage["cost_usd"] + if cost is not None and ( + isinstance(cost, bool) or not isinstance(cost, (int, float)) + or not math.isfinite(cost) or cost < 0 + or cost > config["budget"]["max_cost_usd"]): + raise ContractError("decision helper observed cost exceeds its budget") + if usage["source"] == "unavailable" and any( + usage[field] is not None + for field in ("input_tokens", "output_tokens", "cost_usd")): + raise ContractError("decision helper unavailable usage cannot invent values") + if type(response["elapsed_ms"]) is not int or response["elapsed_ms"] < 0: + raise ContractError("decision helper elapsed_ms is invalid") + return json.loads(canonical_json({ + **response, + "adapter": adapter, + "language": language, + "truncation": dict(truncation), + "recommendations": normalized_recommendations, + "usage": dict(usage), + })) + + +def apply_decision_response( + routing: dict[str, Any], + request: dict[str, Any], + response: dict[str, Any], + config: dict[str, Any], +) -> dict[str, Any]: + """Apply advisory ordering only inside the already-frozen eligible set.""" + config = validate_decision_policy(config) + if config["mode"] == "off": + return json.loads(canonical_json(routing)) + request = validate_decision_request(request) + response = validate_decision_response(request, response, config) + result = json.loads(canonical_json(routing)) + applied_roles = [] + role_status = {} + too_late = response["elapsed_ms"] > config["budget"]["wall_seconds"] * 1000 + truncated = request["truncation"]["occurred"] or response["truncation"]["occurred"] + for role, recommendation in response["recommendations"].items(): + routed = result["roles"][role] + reason = None + if config["mode"] == "shadow": + reason = "shadow_mode" + elif routed["source"] == "override" or role in request["pins"]: + reason = "pinned_route" + elif truncated: + reason = "truncated" + elif too_late: + reason = "late" + elif recommendation["abstain_reason"] is not None: + reason = "abstained" + elif recommendation["confidence"] < config["min_confidence"]: + reason = "below_confidence_gate" + if reason is not None: + role_status[role] = reason + continue + candidates = { + item["profile_id"]: item + for item in [routed["selected"], *routed.get("fallbacks", [])] + } + ordered = [candidates[profile_id] for profile_id in recommendation["ranking"]] + routed["selected"] = ordered[0] + routed["fallbacks"] = ordered[1:] + applied_roles.append(role) + role_status[role] = "applied" + result["decision_helper"] = { + "schema_version": 1, + "mode": config["mode"], + "request_sha256": response["request_sha256"], + "response_sha256": hashlib.sha256( + canonical_json(response).encode(), + ).hexdigest(), + "applied_roles": sorted(applied_roles), + "role_status": role_status, + "usage": response["usage"], + "elapsed_ms": response["elapsed_ms"], + } + return result + + +def decision_fallback( + routing: dict[str, Any], mode: str, status: str, request_sha256: str | None, +) -> dict[str, Any]: + """Record an unusable helper result without changing deterministic routing.""" + if mode == "off": + return json.loads(canonical_json(routing)) + if mode not in {"shadow", "advisory"}: + raise ContractError("decision helper fallback mode is invalid") + status = _identifier(status, "fallback status") + if request_sha256 is not None: + _sha256(request_sha256, "fallback request_sha256") + result = json.loads(canonical_json(routing)) + result["decision_helper"] = { + "schema_version": 1, + "mode": mode, + "request_sha256": request_sha256, + "response_sha256": None, + "applied_roles": [], + "role_status": { + role: status for role in sorted(result["roles"]) + }, + "usage": { + "source": "unavailable", "billable_requests": 0, + "input_tokens": None, "output_tokens": None, "cost_usd": None, + }, + "elapsed_ms": None, + } + return result diff --git a/plugin/core/src/devsquad/delivery_worker.py b/plugin/core/src/devsquad/delivery_worker.py new file mode 100644 index 0000000..76d448c --- /dev/null +++ b/plugin/core/src/devsquad/delivery_worker.py @@ -0,0 +1,97 @@ +"""Offline bounded implementation worker used for delivery fault injection.""" + +from __future__ import annotations + +import json +from pathlib import Path, PurePosixPath +import sys +import time +from typing import Any + +from .contracts import ContractError +from .store import canonical_json +from .workflows import make_implementation_evidence + + +MAX_SNAPSHOT_BYTES = 2 * 1024 * 1024 +MAX_FIXTURE_WRITES = 100 +MAX_FIXTURE_CONTENT_BYTES = 1024 * 1024 + + +def _relative_path(value: Any) -> str: + if not isinstance(value, str) or not value or "\\" in value or "\0" in value: + raise ContractError("implementation fixture path is invalid") + path = PurePosixPath(value) + if path.is_absolute() or ".." in path.parts or path.as_posix() != value: + raise ContractError("implementation fixture path must be repository-relative") + return value + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("delivery snapshot must be an object") + fixture = snapshot.get("internal_implementation_fixture") + if isinstance(fixture, dict) and set(fixture) == {"iterations"}: + fixtures = fixture["iterations"] + index = len(snapshot.get("delivery_iterations", [])) + if (not isinstance(fixtures, list) or not fixtures + or index >= len(fixtures)): + raise ContractError("offline implementation fixture iteration is missing") + fixture = fixtures[index] + if (not isinstance(fixture, dict) + or set(fixture) not in ( + {"writes", "delay_seconds"}, + {"writes", "delay_seconds", "fail_profile_ids"}, + )): + raise ContractError("offline implementation fixture is incomplete") + fail_profile_ids = fixture.get("fail_profile_ids", []) + if (not isinstance(fail_profile_ids, list) + or not all(isinstance(item, str) and item for item in fail_profile_ids)): + raise ContractError("implementation fixture fail_profile_ids is invalid") + selected_profile_id = snapshot["routing"]["roles"]["implementer"][ + "selected" + ]["profile_id"] + if selected_profile_id in fail_profile_ids: + raise ContractError("RATE_LIMITED: offline implementation fixture failure") + writes, delay = fixture["writes"], fixture["delay_seconds"] + if (not isinstance(writes, list) or not writes + or len(writes) > MAX_FIXTURE_WRITES): + raise ContractError("implementation fixture writes must be a bounded array") + if not isinstance(delay, (int, float)) or isinstance(delay, bool) or not 0 <= delay <= 60: + raise ContractError("implementation fixture delay is invalid") + workspace = Path(snapshot["delivery_workspace"]["path"]).resolve(strict=True) + if delay: + time.sleep(delay) + for item in writes: + if not isinstance(item, dict) or set(item) != {"path", "content"}: + raise ContractError("implementation fixture write fields are invalid") + relative = _relative_path(item["path"]) + content = item["content"] + if (not isinstance(content, str) + or len(content.encode()) > MAX_FIXTURE_CONTENT_BYTES): + raise ContractError("implementation fixture content is invalid") + destination = workspace / relative + resolved = destination.resolve(strict=False) + if resolved != workspace and workspace not in resolved.parents: + raise ContractError("implementation fixture path escapes its workspace") + destination.parent.mkdir(parents=True, exist_ok=True) + destination.write_text(content) + return make_implementation_evidence( + snapshot, "Applied the bounded offline implementation fixture.", + ) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_SNAPSHOT_BYTES + 1) + if len(payload) > MAX_SNAPSHOT_BYTES: + raise ContractError("delivery snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("delivery snapshot is not valid UTF-8 JSON") from exc + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py new file mode 100644 index 0000000..4bc24a7 --- /dev/null +++ b/plugin/core/src/devsquad/detached.py @@ -0,0 +1,279 @@ +"""Detached M2 supervisor entrypoint; invoked only from the durable service.""" +import argparse +import json +import os +from pathlib import Path +import sys + +from .contracts import BudgetExhausted, ExecutionIdentity, LaunchSpec +from .service import Service +from .store import ConflictError, Store, canonical_json +from .supervisor import Supervisor + + +def _is_saved_fallback_failure(attempt) -> bool: + encoded = attempt.get("output_metadata") + if not encoded: + return False + try: + metadata = json.loads(encoded) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("saved attempt metadata is invalid") from exc + if not isinstance(metadata, dict): + raise ConflictError("saved attempt metadata is invalid") + failure = metadata.get("failure") + if failure is None: + return False + if not isinstance(failure, dict): + raise ConflictError("saved fallback failure is invalid") + return True + + +def _profile_index( + store: Store, + run_id: str, + role: str, + handoff, +) -> int: + attempts = store.attempts_for_run(run_id) + if handoff is None: + return sum( + attempt.get("role") == role and _is_saved_fallback_failure(attempt) + for attempt in attempts + ) + reviewer_id = handoff.packet.get("attempt_id") + reviewer = next( + (attempt for attempt in attempts if attempt["id"] == reviewer_id), None, + ) + if reviewer is None: + raise ConflictError("handoff reviewer attempt is missing") + if role == "implementer": + seen = False + used = 0 + for attempt in attempts: + if attempt["id"] == reviewer_id: + seen = True + elif (seen and attempt.get("role") == "implementer" + and _is_saved_fallback_failure(attempt)): + used += 1 + return used + if role == "reviewer": + seen = False + used = 0 + for attempt in attempts: + if attempt["id"] == reviewer_id: + seen = True + elif (seen and attempt.get("role") == "reviewer" + and _is_saved_fallback_failure(attempt)): + used += 1 + return used + return sum( + attempt.get("role") == "lead" + and attempt["created_at"] >= reviewer["created_at"] + and _is_saved_fallback_failure(attempt) + for attempt in attempts + ) + + +def main(argv=None): + parser = argparse.ArgumentParser() + parser.add_argument("--database", required=True); parser.add_argument("--artifacts", required=True) + parser.add_argument("--run-id", required=True); parser.add_argument("--expected-version", type=int, required=True) + parser.add_argument("--package-digest", required=True) + args = parser.parse_args(argv) + store = Store(Path(args.database), Path(args.artifacts)) + try: + run = store.run(args.run_id) + if run.get("package_digest") != args.package_digest: + raise ConflictError("detached package digest does not match prepared run") + snapshot = json.loads(run["mutable_snapshot"]) + environment = { + "DEVSQUAD_WORKER": "1", + "DEVSQUAD_RUN_ID": args.run_id, + "DEVSQUAD_DELEGATION_DEPTH": "1", + } + stdin_path = None + adapter = None + handoff = store.handoff_snapshot(args.run_id) + workflow = snapshot["task"]["workflow"] + headless_lead = ( + snapshot["task"]["lead"]["mode"] == "headless" + and handoff is not None + and handoff.status == "open" + ) + delivery_implementer = ( + workflow == "issue-delivery" and "candidate" not in snapshot + ) + role = ( + "lead" if headless_lead + else "implementer" if delivery_implementer + else "reviewer" + ) + if workflow == "council-decision": + from .council_runtime import verify_origin + verify_origin(store, args.run_id, snapshot) + role = "lead" if headless_lead else snapshot["council_state"]["next_role"] + workflow_role = ( + "internal_fake_delay" not in snapshot + and workflow in {"branch-review", "issue-delivery", "council-decision"} + ) + profile_index = None + profile_id = None + if workflow_role: + routed_role = snapshot["routing"]["roles"][role] + candidates = [routed_role["selected"], *routed_role["fallbacks"]] + profile_index = (sum(a.get("role") == role and _is_saved_fallback_failure(a) + for a in store.attempts_for_run(args.run_id)) + if workflow == "council-decision" else _profile_index(store, args.run_id, role, handoff)) + if profile_index >= len(candidates): + raise ConflictError("frozen role fallback set is exhausted") + attempt_selection = candidates[profile_index] + profile_id = attempt_selection["profile_id"] + selected = attempt_selection["profile"] + adapter_key = { + "implementer": "implementation_adapter", + "reviewer": "review_adapter", + "lead": "lead_adapter", + "proposer_a": "council_adapter", "proposer_b": "council_adapter", "critic": "council_adapter", + }[role] + adapters_key = { + "implementer": "implementation_adapters", + "reviewer": "review_adapters", + "lead": "lead_adapters", + "proposer_a": "council_adapters", "proposer_b": "council_adapters", "critic": "council_adapters", + }[role] + adapters = snapshot.get(adapters_key) + if workflow == "council-decision": + adapter_key = "council_adapter" + adapters = snapshot["council_adapters"].get(role, {}) + adapter = ( + adapters.get(profile_id) + if isinstance(adapters, dict) + else snapshot.get(adapter_key) if profile_index == 0 else None + ) + identity = ExecutionIdentity( + selected["harness"], + adapter["harness_version"] if adapter else "fixture", + adapter["model_provider"] if adapter else None, + selected["model_family"], + selected["model_id"], + selected["effort"]["value"], + tuple(selected["required_tools"]), + selected["permission_policy"], + selected["account_pool_id"], + "verified" if adapter else "unknown", + ) + if workflow == "council-decision": + module = "devsquad.council_worker" + elif role == "implementer": + module = ( + "devsquad.claude_delivery_worker" + if adapter else "devsquad.delivery_worker" + ) + elif role == "lead": + module = ( + "devsquad.codex_lead_worker" + if adapter else "devsquad.lead_worker" + ) + else: + module = ( + "devsquad.codex_review_worker" + if adapter else "devsquad.review_worker" + ) + command = [sys.executable, "-P", "-m", module] + worker_snapshot = json.loads(canonical_json(snapshot)) + if workflow == "council-decision": + worker_snapshot["council_role"] = role + worker_snapshot["council_profile_index"] = profile_index + else: + worker_snapshot["routing"]["roles"][role]["selected"] = attempt_selection + if adapter is None: + worker_snapshot.pop(adapter_key, None) + else: + worker_snapshot[adapter_key] = adapter + if headless_lead: + worker_snapshot["headless_handoff"] = { + "handoff_id": handoff.handoff_id, + "packet": handoff.packet, + "packet_sha256": handoff.packet_sha256, + } + input_path, _, _ = store.finalize_artifact( + args.run_id, + ( + f"council-input-{role}-{profile_index}.json" if workflow == "council-decision" else + f"lead-workflow-input-{handoff.sequence}-{profile_index}.json" + if headless_lead else + f"implementation-input-{profile_index}.json" + if role == "implementer" else + f"workflow-input-{profile_index}.json" + ), + canonical_json(worker_snapshot).encode(), + ) + stdin_path = str(input_path) + else: + identity = ExecutionIdentity("devsquad-fake-step", "1", None, None, None, None) + command = [sys.executable, "-P", "-m", "devsquad.fake_step"] + if "internal_fake_delay" in snapshot: + command += ["--delay", str(snapshot["internal_fake_delay"])] + remaining_wall = store.remaining_wall_seconds(args.run_id) + if remaining_wall == 0: + Service(Path(args.database).parent).fail_budget_exhausted( + args.run_id, args.expected_version, + ) + return 1 + if workflow == "council-decision": + environment["DEVSQUAD_COUNCIL_ROLE"] = role + spec = LaunchSpec( + 1, + identity.harness, + "native_protocol" if adapter else "cli_exec", + tuple(command), + run["worktree_path"], + stdin_path, + remaining_wall or snapshot["task"]["budget"]["wall_seconds"], + identity, + environment, + ) + supervisor = Supervisor(store) + try: + handle = supervisor.launch_durable( + args.run_id, + args.expected_version, + spec, + f"daemon:{os.getpid()}", + args.package_digest, + role=role if workflow_role else "worker", + profile_id=profile_id, + profile_index=profile_index, + ) + except ConflictError: return 0 + except BudgetExhausted: + Service(Path(args.database).parent).fail_budget_exhausted( + args.run_id, args.expected_version, + ) + return 1 + returncode = supervisor.wait_durable(handle, spec.timeout_seconds) + current = store.run(args.run_id) + finally: + store.close() + if (current["state"] == "awaiting_host" + and snapshot["task"]["lead"]["mode"] == "headless"): + try: + Service(Path(args.database).parent).resume(args.run_id) + except ConflictError: + pass + elif (current["state"] == "queued" and current["phase"] is None + and not ( + workflow == "issue-delivery" + and role == "implementer" + and isinstance( + json.loads(current["mutable_snapshot"]).get("candidate"), dict, + ) + )): + try: + Service(Path(args.database).parent).resume(args.run_id) + except ConflictError: + pass + return 0 if returncode == 0 else 1 + +if __name__ == "__main__": raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/diagnostics.py b/plugin/core/src/devsquad/diagnostics.py new file mode 100644 index 0000000..b447afb --- /dev/null +++ b/plugin/core/src/devsquad/diagnostics.py @@ -0,0 +1,346 @@ +"""Read-only readiness reporting shared by CLI and MCP surfaces.""" + +from __future__ import annotations + +import json +import os +from pathlib import Path +import platform +import re +import selectors +import shutil +import subprocess +import sys +import time +from typing import Any + +from . import __version__ +from .adapters import AdapterManifest +from .codex_protocol import ( + JsonLinePeer, initialize_request, initialized_notification, + receive_response, request, +) +from .contracts import ContractError +from .integrations import ( + LocalIntegrationManager, + load_integrations, +) +from .probe_process import ( + capture_probe_identity, + close_probe as _close_probe, + subscription_environment as _environment, + wait_probe_exit, +) + +SOURCE_ROOT = Path(__file__).resolve().parents[2] +CORE_ROOT = ( + SOURCE_ROOT + if (SOURCE_ROOT / "adapters").is_dir() + else Path(sys.prefix) / "share" / "devsquad" +) +PROBE_TIMEOUT_SECONDS = 3 +AUTH_TIMEOUT_SECONDS = 5 +MAX_PROBE_BYTES = 16 * 1024 +VERSION_PATTERN = re.compile( + r"(?:codex-cli )?\d+\.\d+\.\d+(?:-[A-Za-z0-9.]+)?(?: \(Claude Code\))?" +) + + +def _probe_output( + argv: list[str], *, project: Path, environment: dict[str, str], +) -> tuple[int, str]: + """Bound time and bytes before decoding any provider-controlled output.""" + deadline = time.monotonic() + PROBE_TIMEOUT_SECONDS + process = subprocess.Popen( + argv, cwd=project, env=environment, stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, start_new_session=True, + ) + start_identity = None + try: + start_identity = capture_probe_identity(process) + if start_identity is None: + raise ContractError("diagnostic process ownership is unavailable") + assert process.stdout is not None + chunks = bytearray() + with selectors.DefaultSelector() as selector: + selector.register(process.stdout, selectors.EVENT_READ) + while True: + remaining = deadline - time.monotonic() + if remaining <= 0 or not selector.select(remaining): + raise TimeoutError("diagnostic probe timed out") + chunk = os.read(process.stdout.fileno(), min(4096, MAX_PROBE_BYTES + 1 - len(chunks))) + if not chunk: + break + chunks.extend(chunk) + if len(chunks) > MAX_PROBE_BYTES: + raise ContractError("diagnostic probe exceeds byte bound") + if deadline - time.monotonic() <= 0: + raise TimeoutError("diagnostic probe timed out") + output = chunks.decode("utf-8") + wait_probe_exit(process, start_identity=start_identity, deadline=deadline) + finally: + # Retain the direct child's PID until group cleanup is confirmed; + # waiting here first would discard the exited-parent ownership anchor. + _close_probe(process, start_identity=start_identity) + assert process.returncode is not None + return process.returncode, output + + +def _resolve_adapter( + manifest: AdapterManifest, *, project: Path, environment: dict[str, str], +) -> tuple[str | None, str | None]: + candidates = tuple(dict.fromkeys( + binary for name in manifest.binary_candidates + if (binary := shutil.which(name, path=environment["PATH"])) is not None + )) + first = (None, None) + for binary in candidates: + try: + code, output = _probe_output([binary, "--version"], project=project, environment=environment) + version = output.strip() if code == 0 else None + # A version banner can contain a login error or secrets. Only a + # bounded canonical version is safe to retain in the public report. + if version is not None and VERSION_PATTERN.fullmatch(version) is None: + version = None + except (ContractError, OSError, TimeoutError, UnicodeError, subprocess.TimeoutExpired): + version = None + if first[0] is None: + first = (binary, version) + if version in manifest.verified_versions: + return binary, version + return first + + +def _authentication( + *, authenticated: bool | None = None, method: str | None = None, + subscription_supported: bool = False, check: str | None = None, + reason: str | None = None, next_action: str | None = None, +) -> dict[str, Any]: + return { + "status": ( + "authenticated" if authenticated is True + else "unauthenticated" if authenticated is False else "unknown" + ), + "authenticated": authenticated, + "method": method, + "subscription_supported": subscription_supported, + "check": check, + "reason": reason, + "next_action": next_action, + } + + +def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + value: dict[str, Any] = {} + for key, item in pairs: + if key in value: + raise ValueError("duplicate diagnostic field") + value[key] = item + return value + + +def _claude_auth(binary: str, *, project: Path, environment: dict[str, str]) -> dict[str, Any]: + check = "claude auth status --json" + try: + code, output = _probe_output( + [binary, "auth", "status", "--json"], project=project, environment=environment, + ) + value = json.loads(output, object_pairs_hook=_unique_object) + if not isinstance(value, dict) or type(value.get("loggedIn")) is not bool: + raise ValueError("invalid authentication state") + logged_in, method = value["loggedIn"], value.get("authMethod") + if not isinstance(method, str) or method not in {"none", "claude.ai", "oauth_token", "api_key", "api_key_helper", "third_party"}: + raise ValueError("unknown authentication method") + if (code != (0 if logged_in else 1) + or (logged_in and method == "none") + or (not logged_in and method != "none")): + raise ValueError("inconsistent authentication state") + except (ContractError, OSError, TimeoutError, UnicodeError, ValueError, RecursionError, subprocess.TimeoutExpired): + return _authentication(check=check, reason="auth_check_failed") + subscription_supported = logged_in and method == "claude.ai" + return _authentication( + authenticated=logged_in, method=method, check=check, + subscription_supported=subscription_supported, + reason=(None if subscription_supported else "subscription_login_required"), + next_action=None if subscription_supported else "claude auth login", + ) + + +def _codex_auth(binary: str, *, project: Path, environment: dict[str, str]) -> dict[str, Any]: + check = "codex account/read" + deadline = time.monotonic() + AUTH_TIMEOUT_SECONDS + process = None + start_identity = None + try: + process = subprocess.Popen( + [binary, "app-server", "--listen", "stdio://"], + cwd=project, env=environment, stdin=subprocess.PIPE, + stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, text=True, + bufsize=1, start_new_session=True, + ) + start_identity = capture_probe_identity(process) + if start_identity is None: + raise ContractError("diagnostic process ownership is unavailable") + assert process.stdin is not None and process.stdout is not None + peer = JsonLinePeer(process.stdout, process.stdin, max_frame_bytes=MAX_PROBE_BYTES) + peer.send(initialize_request(1)) + initialized = receive_response(peer, 1, timeout_seconds=max(0.001, deadline - time.monotonic())) + if "error" in initialized or not isinstance(initialized.get("result"), dict): + raise ContractError("native initialization unavailable") + peer.send(initialized_notification()) + peer.send(request(2, "account/read", {"refreshToken": False})) + reply = receive_response(peer, 2, timeout_seconds=max(0.001, deadline - time.monotonic())) + value = reply.get("result") + if ("error" in reply or not isinstance(value, dict) or "account" not in value + or type(value.get("requiresOpenaiAuth")) is not bool): + raise ContractError("native authentication unavailable") + account = value["account"] + if account is None: + return _authentication( + authenticated=False if value["requiresOpenaiAuth"] else None, + check=check, reason="subscription_login_required", + next_action="codex login", + ) + if (not isinstance(account, dict) or not isinstance(account.get("type"), str) + or account["type"] not in {"chatgpt", "apiKey"}): + raise ContractError("native authentication method unknown") + method = account["type"] + subscription_supported = method == "chatgpt" and value["requiresOpenaiAuth"] + return _authentication( + authenticated=True, method=method, check=check, + subscription_supported=subscription_supported, + reason=None if subscription_supported else "subscription_login_required", + next_action=None if subscription_supported else "codex login", + ) + except (ContractError, EOFError, OSError, TimeoutError, ValueError, RecursionError, subprocess.TimeoutExpired): + return _authentication(check=check, reason="auth_check_failed") + finally: + if process is not None: + _close_probe(process, start_identity=start_identity) + + +def _adapter_rows(*, project: Path, home: Path | None = None) -> list[dict[str, Any]]: + environment = _environment(home) + rows = [] + for path in sorted((CORE_ROOT / "adapters").glob("*/adapter.json")): + manifest = AdapterManifest.load(path) + binary, version = _resolve_adapter(manifest, project=project, environment=environment) + supported = binary is not None and version in manifest.verified_versions + status = ( + "unavailable" if not binary + else "supported" if supported + else "unverified" + ) + if not binary: + authentication = _authentication(reason="not_installed") + elif manifest.name not in {"codex", "claude"}: + authentication = _authentication(reason="auth_check_unsupported") + elif not supported: + authentication = _authentication(reason="unverified_version") + else: + probe = _codex_auth if manifest.name == "codex" else _claude_auth + try: + authentication = probe(binary, project=project, environment=environment) + except (ContractError, OSError, subprocess.TimeoutExpired): + authentication = _authentication(reason="auth_cleanup_unconfirmed") + ready = supported and authentication["subscription_supported"] + rows.append({ + "adapter": manifest.name, + "transport": manifest.transport, + "status": status, + "binary": binary, + "version": version, + "manifest": str(path), + "installed": binary is not None, + "supported": supported, + "authenticated": authentication["authenticated"], + "authentication": authentication, + "ready": ready, + # Authentication and MCP registration prove neither worker + # execution nor a compatible model/permission operation receipt. + "operation_verified": None, + "operation_verification": { + "status": "unknown", "reason": "no_compatible_operation_proof", + }, + }) + return rows + + +def build_doctor_report( + *, + project: Path, + home: Path | None = None, + squad_executable: Path | None = None, + manager: LocalIntegrationManager | None = None, +) -> dict[str, Any]: + """Report provider and installed-app readiness without modifying config.""" + + adapters = _adapter_rows(project=project, home=home) + integration_manager = manager or LocalIntegrationManager( + project=project, + home=home, + squad_executable=squad_executable, + ) + local_apps = [] + for template in load_integrations(): + row = integration_manager.inspect(template) + local_apps.append({ + **row, + "registered": row.get("status") == "matching", + "operation_verified": None, + }) + installed_apps = [row for row in local_apps if row["installed"]] + adapter_ready = any(row["ready"] for row in adapters) + by_adapter = {row["adapter"]: row for row in adapters} + supported_workflows = {} + for workflow, required in ( + ("branch-review", ("codex",)), ("issue-delivery", ("claude", "codex")), + ): + blocked = [name for name in required if not by_adapter.get(name, {}).get("ready")] + supported_workflows[workflow] = { + "supported": True, "ready": not blocked, + "required_adapters": list(required), "blocked_adapters": blocked, + } + supported_workflows["council"] = { + "supported": True, "implemented_partial": True, "ready": False, + "native_ready": False, "automatic_enabled": False, + "reason": "native_network_attestation_unavailable", + } + workflow_ready = any(row["ready"] for row in supported_workflows.values()) + local_apps_required = bool(installed_apps) + local_apps_ready = ( + all(row["ready"] for row in installed_apps) + if local_apps_required else True + ) + return { + "core_version": __version__, + "ready": workflow_ready and local_apps_ready, + "adapter_ready": adapter_ready, + "supported_workflows": supported_workflows, + "local_app_access": { + "required": local_apps_required, + "ready": local_apps_ready, + "installed_count": len(installed_apps), + "configured_count": sum(row["ready"] for row in installed_apps), + "mcp_sdk_requirement": "mcp==2.2.0", + "mcp_sdk_available": integration_manager.mcp_sdk_available, + "mcp_sdk_supported": integration_manager.mcp_sdk_supported, + "mcp_sdk_version": integration_manager.mcp_sdk_version, + "squad_executable": ( + str(integration_manager.squad_executable) + if integration_manager.squad_executable else None + ), + "launcher_error": integration_manager.launcher_error, + }, + "adapters": adapters, + "local_apps": local_apps, + "workflows": {"council-decision": { + "implemented": True, "automatic_enabled": False, "ready": False, + "readiness": "native_network_attestation_unavailable", + "native_network_ready": False, + "read_boundary_available": platform.system() == "Darwin" and Path("/usr/bin/sandbox-exec").is_file(), + "requirements": ["three distinct entitled verified read-only model IDs", "frozen default-deny Seatbelt probe", + "subscription/network compatibility", "two valid independent proposals and one distinct critic"], + "supported_rounds": 1, + }}, + } diff --git a/plugin/core/src/devsquad/experiment_eligibility.py b/plugin/core/src/devsquad/experiment_eligibility.py new file mode 100644 index 0000000..6b61c36 --- /dev/null +++ b/plugin/core/src/devsquad/experiment_eligibility.py @@ -0,0 +1,123 @@ +"""One transaction-bound current-evidence gate for all lifecycle consumers.""" + +from __future__ import annotations + +from datetime import datetime +import hashlib +import sqlite3 +from typing import Any + +from .claude_identity import strict_json +from .contracts import ContractError +from .experiment_evidence import read_experiment_chains +from .learning import evaluate_experiment, validate_experiment +from .store import canonical_json + + +def saved_evaluation( + connection: sqlite3.Connection, experiment_id: str, + evaluation_sha256: str | None = None, +) -> dict[str, Any] | None: + """Decode historical bytes without giving them new execution authority.""" + original = connection.execute( + "SELECT * FROM experiments WHERE experiment_id=?", (experiment_id,), + ).fetchone() + if original is None: + return None + record = {**dict(original), "revision_id": None, "previous_evaluation_sha256": None} + if evaluation_sha256 != original["evaluation_sha256"]: + if evaluation_sha256 is None: + revision = connection.execute( + "SELECT * FROM experiment_evaluation_revisions WHERE experiment_id=? ORDER BY id DESC LIMIT 1", + (experiment_id,), + ).fetchone() + else: + revision = connection.execute( + "SELECT * FROM experiment_evaluation_revisions WHERE experiment_id=? AND evaluation_sha256=?", + (experiment_id, evaluation_sha256), + ).fetchone() + if revision is not None: + parent = revision["previous_evaluation_sha256"] + if parent != original["evaluation_sha256"] and connection.execute( + "SELECT 1 FROM experiment_evaluation_revisions WHERE experiment_id=? " + "AND evaluation_sha256=? AND id dict[str, Any]: + """Recompute authority in the caller's transaction, never from a label.""" + if not connection.in_transaction: + raise ContractError("current experiment eligibility requires a ledger transaction") + evaluation = record["evaluation"] + result = { + "schema_version": 1, "experiment_id": record["experiment_id"], + "spec_sha256": record["spec_sha256"], + "evaluation_sha256": record["evaluation_sha256"], + "saved_evidence_sha256": evaluation.get("evidence_sha256"), + "current_evidence_sha256": None, "eligible": False, "reasons": [], + } + if record["spec"].get("schema_version") != 2: + result["reasons"] = ["legacy_unverified_evidence"] + return result + project = connection.execute( + "SELECT git_common_dir FROM projects WHERE id=?", (record["project_id"],), + ).fetchone() + if project is None: + result["reasons"] = ["saved_project_unavailable"] + return result + try: + spec = validate_experiment(record["spec"]) + chains = read_experiment_chains( + connection, spec=spec, project_id=record["project_id"], + project_common_dir=project["git_common_dir"], evaluated_at=now.isoformat(), + ) + fresh = evaluate_experiment( + spec, chains, evaluated_at=now.isoformat(), + project_common_dir=project["git_common_dir"], + ) + except ContractError: + result["reasons"] = ["invalid_saved_run_evidence"] + return result + result["current_evidence_sha256"] = fresh["evidence_sha256"] + if canonical_json({**fresh, "evaluated_at": record["recorded_at"]}) != canonical_json(evaluation): + result["reasons"] = ["saved_evaluation_stale_evidence_changed"] + return result + result["eligible"] = True + return result + + +def require_current_evidence( + connection: sqlite3.Connection, experiment_id: str, evaluation_sha256: str, + *, now: datetime, +) -> tuple[dict[str, Any], dict[str, Any]]: + record = saved_evaluation(connection, experiment_id, evaluation_sha256) + if record is None: + raise ContractError("experiment evaluation evidence is unavailable") + eligibility = current_evidence(connection, record, now=now) + if not eligibility["eligible"]: + raise ContractError("experiment evidence is not current: " + ", ".join(eligibility["reasons"])) + return record, eligibility diff --git a/plugin/core/src/devsquad/experiment_evidence.py b/plugin/core/src/devsquad/experiment_evidence.py new file mode 100644 index 0000000..16b2fcc --- /dev/null +++ b/plugin/core/src/devsquad/experiment_evidence.py @@ -0,0 +1,502 @@ +"""Authoritative v2 experiment inputs from one consistent saved-ledger read. + +The pure evaluator accepts normalized witnesses, but callers cannot supply +them here. Assignment, execution and outcome identity come from the ledger. +The caller owns the transaction, including any subsequent decision mutation. +""" + +from __future__ import annotations + +from datetime import datetime +import hashlib +from pathlib import Path +import sqlite3 +from typing import Any + +from .contracts import ContractError +from .claude_identity import strict_json +from .experiment_provenance import ( + assignment_for, paired_input_identity, selected_execution_fingerprint, + validate_arm_chain, validate_assignment, +) +from .learning import validate_outcome +from .store import canonical_json, request_hash + + +def _digest(value: Any) -> str: + return hashlib.sha256(canonical_json(value).encode()).hexdigest() + + +def _object(payload: str, label: str) -> dict[str, Any]: + try: + value = strict_json(payload) + except (ContractError, TypeError, ValueError) as exc: + raise ContractError(f"experiment saved {label} is not valid JSON") from exc + if not isinstance(value, dict): + raise ContractError(f"experiment saved {label} must be an object") + return value + + +def _outcome(row: sqlite3.Row, run: sqlite3.Row, now: datetime) -> dict[str, Any]: + value = validate_outcome(_object(row["payload_json"], "outcome"), now=now) + if (row["run_id"] != run["id"] or row["payload_sha256"] != _digest(value) + or any(row[key] != value[key] for key in ( + "outcome_id", "kind", "verdict", "selection_mode", + "observed_at", "corrects_outcome_id", + )) or value["selection_mode"] != "experimental"): + raise ContractError("experiment saved outcome identity/hash is inconsistent") + if value["kind"] == "final" and value["verdict"] != run["state"]: + raise ContractError("experiment final outcome does not match terminal run") + return value + + +def _artifact( + connection: sqlite3.Connection, artifact_id: str, run_id: str, +) -> tuple[dict[str, Any], bytes]: + row = connection.execute( + "SELECT id,run_id,name,path,sha256,byte_size FROM artifacts WHERE id=?", + (artifact_id,), + ).fetchone() + if row is None or row["run_id"] != run_id: + raise ContractError("experiment attempt artifact belongs to another run or is missing") + try: + content = Path(row["path"]).read_bytes() + except OSError as exc: + raise ContractError("experiment attempt artifact is unavailable") from exc + if len(content) != row["byte_size"] or hashlib.sha256(content).hexdigest() != row["sha256"]: + raise ContractError("experiment attempt artifact hash is inconsistent") + return {key: row[key] for key in ("id", "name", "sha256", "byte_size")}, content + + +def verify_prelaunch_snapshot( + connection: sqlite3.Connection, *, run_id: str, + snapshot: Any, package_digest: str, +) -> None: + """Fence continued trial execution to its original controlled inputs.""" + frozen = connection.execute( + "SELECT a.*,r.package_digest AS run_package,p.git_common_dir " + "FROM experiment_assignments a JOIN runs r ON r.id=a.run_id " + "JOIN projects p ON p.id=r.project_id WHERE a.run_id=?", (run_id,), + ).fetchone() + if frozen is None: + if isinstance(snapshot, dict) and ( + snapshot.get("experiment_spec") is not None + or snapshot.get("experiment_assignment") is not None): + raise ContractError("experiment launch has no immutable prelaunch assignment") + return + if not isinstance(snapshot, dict): + raise ContractError("experiment launch snapshot is missing") + original = _object(frozen["snapshot_json"], "preparation snapshot") + spec = original.get("experiment_spec") + assignment = validate_assignment( + _object(frozen["assignment_json"], "assignment"), spec=spec, + project_common_dir=frozen["git_common_dir"], + ) + if (package_digest != frozen["package_digest"] + or package_digest != frozen["run_package"] + or _digest(assignment) != frozen["assignment_sha256"] + or original.get("experiment_assignment") != assignment + or snapshot.get("experiment_spec") != spec + or snapshot.get("experiment_assignment") != assignment): + raise ContractError("experiment launch differs from its frozen assignment/package") + role = assignment["role"] + inputs = paired_input_identity(snapshot, role=role, package_digest=package_digest) + selected = snapshot["routing"]["roles"][role]["selected"] + if (any(inputs[key] != assignment[key] for key in inputs) + or any(selected[key] != assignment[key] for key in ("profile_id", "profile_sha256")) + or selected_execution_fingerprint(snapshot, role=role) != assignment["execution_sha256"]): + raise ContractError("experiment launch changed its controlled input/profile/execution") + + +def _named_artifact(connection, run_id, name, artifacts): + row = connection.execute("SELECT id FROM artifacts WHERE run_id=? AND name=?", (run_id, name)).fetchone() + if row is None: + raise ContractError(f"experiment missing imported artifact: {name}") + artifact, content = _artifact(connection, row["id"], run_id) + artifacts.append(artifact) + return content + + +def _delivery_revision(connection, run_id, events, claim_version, previous_candidate): + revisions = [event for event in events if event["type"] == "delivery.revision_queued" + and event["run_version"] < claim_version] + if not revisions: + if previous_candidate is not None: + raise ContractError("experiment delivery continuation has no revision fence") + return None + event = revisions[-1] + payload = _object(event["payload"], "delivery revision event") + row = connection.execute( + "SELECT h.*,s.submission_id,s.submission_hash,s.decision_json,s.disposition,s.recorded_run_version " + "FROM handoffs h JOIN handoff_submissions s ON s.handoff_id=h.id " + "WHERE h.run_id=? AND h.id=? AND s.submission_id=? AND s.outcome='recorded'", + (run_id, payload.get("handoff_id"), payload.get("submission_id")), + ).fetchone() + if row is None: + raise ContractError("experiment delivery revision has no recorded disposition") + packet = _object(row["packet_json"], "revision handoff") + decision = _object(row["decision_json"], "revision disposition") + body = {key: value for key, value in decision.items() if key != "submission_hash"} + if (row["packet_sha256"] != _digest(packet) or row["submission_hash"] != request_hash(body) + or decision.get("submission_hash") != row["submission_hash"] + or decision.get("submission_id") != row["submission_id"] + or row["disposition"] != "revise" or decision.get("disposition") != "revise" + or row["recorded_run_version"] >= event["run_version"] + or previous_candidate is None + or packet.get("candidate_sha256") != previous_candidate["candidate_sha256"] + or payload.get("previous_candidate_sha256") != previous_candidate["candidate_sha256"]): + raise ContractError("experiment delivery revision identity/hash is inconsistent") + return { + "handoff_id": row["id"], "sequence": row["sequence"], + "submission_id": row["submission_id"], "submission_hash": row["submission_hash"], + "previous_candidate_sha256": packet["candidate_sha256"], "reason": decision["reason"], + "review": packet["review"], "checks": packet["checks"], "evidence_refs": decision["evidence_refs"], + } + + +def _delivery_candidate(connection, run_id, attempt_id, events, artifacts, snapshot, iteration): + candidates = connection.execute( + "SELECT id FROM artifacts WHERE run_id=? AND name GLOB 'candidate-*.json'", (run_id,), + ).fetchall() + matches = [] + for row in candidates: + artifact, content = _artifact(connection, row["id"], run_id) + candidate = _object(content, "delivery candidate") + if candidate.get("implementation_artifact") == f"implementation-attempt-{attempt_id}.json": + matches.append((artifact, candidate)) + ready = [event for event in events if event["type"] == "delivery.candidate_ready" + and _object(event["payload"], "candidate event").get("attempt_id") == attempt_id] + if len(matches) != 1 or len(ready) != 1: + raise ContractError("experiment delivery implementation has no unique saved candidate fence") + artifact, candidate = matches[0] + fields = ("schema_version", "baseline_oid", "commit_oid", "tree_oid", "patch_sha256", "changed_paths") + if any(key not in candidate for key in fields): + raise ContractError("experiment saved candidate identity is incomplete") + identity = {key: candidate[key] for key in fields} + event = _object(ready[0]["payload"], "candidate event") + if (candidate.get("candidate_sha256") != _digest(identity) + or candidate["baseline_oid"] != snapshot["delivery_workspace"]["baseline_oid"] + or candidate.get("iteration") != iteration + or any(event.get(key) != candidate[key] for key in ("candidate_sha256", "commit_oid", "patch_sha256")) + or artifact["name"] != f"candidate-{iteration}.json" + or candidate.get("patch_artifact") != f"candidate-{iteration}.patch"): + raise ContractError("experiment saved candidate differs from its implementation fence") + artifacts.append(artifact) + patch = _named_artifact(connection, run_id, candidate["patch_artifact"], artifacts) + if hashlib.sha256(patch).hexdigest() != candidate["patch_sha256"] or len(patch) != candidate.get("patch_bytes"): + raise ContractError("experiment candidate patch differs from its saved identity") + return candidate, ready[0]["run_version"] + + +def _attempts( + connection: sqlite3.Connection, run: sqlite3.Row, frozen: sqlite3.Row, + snapshot: dict[str, Any], assignment: dict[str, Any], +) -> tuple[list[dict[str, Any]], bool]: + """Validate *all* reservations/history, not only a final successful arm. + + Unstarted reservations remain in the digest, but do not become exposure. + A launched attempt without reconciled output keeps the arm unavailable. + """ + events = connection.execute( + "SELECT run_version,type,payload FROM events WHERE run_id=? ORDER BY run_version", + (run["id"],), + ).fetchall() + prepared_version = frozen["frozen_run_version"] + if (type(prepared_version) is not int or prepared_version < 2 + or type(frozen["preparation_fencing_token"]) is not int + or frozen["preparation_fencing_token"] < 1 + or run["version"] < prepared_version): + raise ContractError("experiment preparation fence/version is invalid") + queued = [event for event in events if event["run_version"] == prepared_version] + if len(queued) != 1 or queued[0]["type"] != "run.queued": + raise ContractError("experiment preparation has no saved queued fence") + if not events or events[0]["type"] != "run.preparing": + raise ContractError("experiment preparation history is unavailable") + preparation_token = 1 + claimed, launched, unstarted = {}, {}, set() + for event in events: + if event["type"] == "run.preparation_reclaimed" and event["run_version"] < prepared_version: + preparation_token = _object(event["payload"], "preparation event").get("fencing_token") + if event["type"] not in {"supervisor.claimed", "run.running", "run.unstarted_attempt_recovered"}: + continue + payload = _object(event["payload"], "attempt event") + attempt_id = payload.get("attempt_id") + if not isinstance(attempt_id, str) or not attempt_id: + raise ContractError("experiment attempt event identity is missing") + if event["type"] == "run.unstarted_attempt_recovered": + unstarted.add(attempt_id) + continue + target = claimed if event["type"] == "supervisor.claimed" else launched + if attempt_id in target or event["run_version"] <= prepared_version: + raise ContractError("experiment assignment must precede distinct attempt events") + target[attempt_id] = {"run_version": event["run_version"], **payload} + if preparation_token != frozen["preparation_fencing_token"]: + raise ContractError("experiment preparation fencing token does not match history") + rows = connection.execute("SELECT * FROM attempts WHERE run_id=?", (run["id"],)).fetchall() + if set(claimed) != {row["id"] for row in rows} or not set(launched) <= set(claimed): + raise ContractError("experiment attempt reservations do not match saved history") + history = [] + exposed = False + incomplete = False + delivery_snapshot = dict(snapshot) + delivery_iterations = [] + candidate_version = None + for row in sorted(rows, key=lambda item: claimed[item["id"]]["run_version"]): + role = row["role"] + route = snapshot["routing"]["roles"].get(role) + if (row["project_id"] != run["project_id"] + or row["package_digest"] != frozen["package_digest"] + or not isinstance(route, dict)): + raise ContractError("experiment actual attempt project/package/role is inconsistent") + slots = [route["selected"], *route.get("fallbacks", [])] + index = row["profile_index"] + if (type(index) is not int or not 0 <= index < len(slots) + or row["profile_id"] != slots[index]["profile_id"] + or row["account_pool_id"] != slots[index]["profile"]["account_pool_id"]): + raise ContractError("experiment actual attempt profile does not match its frozen slot") + selected = slots[index] + started = launched.get(row["id"]) + if (row["pid"] is None) != (started is None): + raise ContractError("experiment actual attempt launch identity is inconsistent") + if started is not None: + if (started["run_version"] <= claimed[row["id"]]["run_version"] + or any(started.get(key) != row[key] for key in ("pid", "pgid", "process_start_id"))): + raise ContractError("experiment actual attempt launch fence is inconsistent") + if row["status"] not in {"finished", "recovery_required"}: + incomplete = True + # A repaired pre-gate crash never became a worker exposure. All other + # launched tested-role attempts, including failed fallbacks, must be + # the explicitly declared execution, not merely an eligible profile. + was_worker = started is not None and row["id"] not in unstarted + if was_worker and role == assignment["role"]: + actual = {**snapshot, "routing": {**snapshot["routing"], "roles": { + **snapshot["routing"]["roles"], role: {**route, "selected": selected}, + }}} + if (selected["profile_id"] != assignment["profile_id"] + or selected["profile_sha256"] != assignment["profile_sha256"] + or selected_execution_fingerprint(actual, role=role) != assignment["execution_sha256"]): + raise ContractError("experiment actual attempt does not execute the declared arm") + metadata = None + artifacts = [] + captures = {} + if row["output_metadata"] is not None: + metadata = _object(row["output_metadata"], "attempt output") + if started is None or row["id"] in unstarted: + raise ContractError("experiment unstarted attempt cannot have worker output") + for stream in ("stdout", "stderr"): + artifact, content = _artifact(connection, row[f"{stream}_artifact_id"], run["id"]) + capture = metadata.get(stream) + if (not isinstance(capture, dict) + or capture.get("captured_sha256") != artifact["sha256"] + or capture.get("captured_bytes") != artifact["byte_size"]): + raise ContractError("experiment attempt output does not match its saved capture") + artifacts.append(artifact) + captures[stream] = content + if was_worker and role == assignment["role"] and row["status"] == "finished": + exposed = True + elif was_worker: + incomplete = True + # These immutable artifacts retain imported observed identity and + # native usage; failure diagnostics are retained in output_metadata. + imports = {} + for prefix in ("review", "implementation", "lead"): + artifact_row = connection.execute( + "SELECT id FROM artifacts WHERE run_id=? AND name=?", + (run["id"], f"{prefix}-attempt-{row['id']}.json"), + ).fetchone() + if artifact_row is not None: + artifact, content = _artifact(connection, artifact_row["id"], run["id"]) + document = _object(content, "imported attempt evidence") + evidence = document.get("attempt", document) + if (not isinstance(evidence, dict) or evidence.get("role") != role + or canonical_json(evidence.get("selected_profile")) != canonical_json(selected)): + raise ContractError("experiment imported execution differs from its frozen attempt profile") + artifacts.append(artifact) + imports[prefix] = document + delivery = snapshot["task"]["workflow"] == "issue-delivery" + if captures and (role == "reviewer" or delivery and role == "implementer"): + prefix = "implementation" if role == "implementer" else "review" + if prefix in imports: + from .workflows import ( + validate_branch_review_evidence, validate_implementation_evidence, + require_check_integrity, require_independent_delivery_review, + ) + + context = snapshot + if delivery and role == "implementer": + context = dict(snapshot) + previous = delivery_iterations[-1]["candidate"] if delivery_iterations else None + revision = _delivery_revision(connection, run["id"], events, claimed[row["id"]]["run_version"], previous) + if revision is not None: + context["revision_request"] = revision + document = validate_implementation_evidence(strict_json(captures["stdout"]), context) + if canonical_json(imports[prefix]) != canonical_json(document): + raise ContractError("experiment imported implementation differs from captured evidence") + candidate, candidate_version = _delivery_candidate( + connection, run["id"], row["id"], events, artifacts, snapshot, len(delivery_iterations) + 1, + ) + if candidate_version <= claimed[row["id"]]["run_version"]: + raise ContractError("experiment candidate precedes its implementation reservation") + iteration = {"candidate": candidate, "implementation": document} + if revision is not None: + iteration["revision_request"] = revision + delivery_iterations.append(iteration) + delivery_snapshot = {**snapshot, "delivery_iterations": delivery_iterations, "workspace": { + "candidate_sha256": candidate["candidate_sha256"], "base_oid": candidate["baseline_oid"], + "target_oid": candidate["commit_oid"], + }} + if "pending_review_fixture" in snapshot: + delivery_snapshot["internal_review_fixture"] = snapshot["pending_review_fixture"] + else: + if delivery: + if candidate_version is None or candidate_version >= claimed[row["id"]]["run_version"]: + raise ContractError("experiment delivery reviewer has no preceding implementation candidate") + context = delivery_snapshot + document = validate_branch_review_evidence(strict_json(captures["stdout"]), context) + if canonical_json(imports[prefix]) != canonical_json(document["attempt"]): + raise ContractError("experiment imported review attempt differs from captured evidence") + for name, expected in ( + (f"review-{row['id']}.json", document["review"]), + (f"checks-{row['id']}.json", {"schema_version": 1, "candidate_sha256": document["candidate_sha256"], + "target_oid": document["target_oid"], "results": document["checks"]}), + (f"evaluation-{row['id']}.json", document["evaluation"]), + ): + imported = _object(_named_artifact(connection, run["id"], name, artifacts), "review import") + if canonical_json(imported) != canonical_json(expected): + raise ContractError("experiment imported review/check/evaluation differs from captured evidence") + require_check_integrity(document["checks"]) + require_independent_delivery_review(context, document) + else: + # Terminal failed/cancelled workers do not publish successful + # review evidence. Bind their opaque output to the hashed + # early-terminal receipt, not an absent metadata.failure key. + receipt_row = connection.execute( + "SELECT id FROM artifacts WHERE run_id=? AND name='result-receipt.json'", + (run["id"],), + ).fetchone() + if receipt_row is None: + raise ContractError(f"experiment {role} has no imported evidence or failure receipt") + artifact, content = _artifact(connection, receipt_row["id"], run["id"]) + receipt = _object(content, "terminal failure receipt") + receipt_attempts = receipt.get("attempts") + if not isinstance(receipt_attempts, list): + raise ContractError(f"experiment failed {role} receipt attempts are invalid") + projections = [item for item in receipt_attempts + if isinstance(item, dict) and item.get("id") == row["id"]] + if (receipt.get("run_id") != run["id"] + or receipt.get("state") != run["state"] + or len(projections) != 1 + or projections[0].get("status") not in {"failed", "cancelled"} + or projections[0].get("role") != role + or canonical_json(projections[0].get("selected_profile")) != canonical_json(selected)): + raise ContractError(f"experiment failed {role} receipt is inconsistent") + artifacts.append(artifact) + history.append({ + **{key: row[key] for key in ( + "id", "run_id", "project_id", "role", "status", "profile_id", + "profile_index", "account_pool_id", "package_digest", "pid", + "pgid", "process_start_id", "created_at", "finished_at", + )}, + "reservation": claimed[row["id"]], "launch": started, + "unstarted_recovered": row["id"] in unstarted, + "profile_sha256": selected["profile_sha256"], + "output_metadata": metadata, "artifacts": artifacts, + }) + return history, exposed and not incomplete + + +def read_experiment_chains( + connection: sqlite3.Connection, *, spec: dict[str, Any], project_id: str | None, + project_common_dir: str, evaluated_at: str, +) -> dict[str, dict[str, Any]]: + """Build v2 witnesses solely from immutable preparation and durable runs.""" + if not connection.in_transaction: + raise ContractError("experiment evidence requires a consistent ledger transaction") + if spec["schema_version"] != 2: + raise ContractError("saved-run provenance requires a v2 experiment") + declared = connection.execute( + "SELECT project_id,spec_json,spec_sha256 FROM experiment_specs WHERE experiment_id=?", + (spec["experiment_id"],), + ).fetchone() + if (declared is None or declared["project_id"] != project_id + or declared["spec_json"] != canonical_json(spec) + or declared["spec_sha256"] != _digest(spec)): + raise ContractError("experiment specification is not predeclared for this saved project") + records = connection.execute( + "SELECT * FROM experiment_assignments WHERE experiment_id=?", (spec["experiment_id"],), + ).fetchall() + expected_keys = {(case["case_id"], arm) for case in spec["cases"] for arm in ("control", "candidate")} + frozen_by_arm = {(row["case_id"], row["arm"]): row for row in records} + if len(frozen_by_arm) != len(records) or not set(frozen_by_arm) <= expected_keys: + raise ContractError("experiment saved assignments have undeclared or duplicate arms") + chains = {} + for case in spec["cases"]: + for arm in ("control", "candidate"): + expected = assignment_for(spec, case["case_id"], arm, project_common_dir=project_common_dir) + final_row = connection.execute( + "SELECT * FROM outcomes WHERE outcome_id=?", (expected["outcome_id"],), + ).fetchone() + frozen = frozen_by_arm.get((case["case_id"], arm)) + if frozen is None: + if final_row is not None: + raise ContractError("experiment outcome has no prelaunch assignment") + continue + assignment = validate_assignment( + _object(frozen["assignment_json"], "assignment"), spec=spec, + project_common_dir=project_common_dir, + ) + if (assignment != expected or frozen["assignment_sha256"] != _digest(assignment) + or frozen["outcome_id"] != expected["outcome_id"]): + raise ContractError("experiment saved assignment identity/hash is inconsistent") + run = connection.execute( + "SELECT r.*,p.git_common_dir FROM runs r JOIN projects p ON p.id=r.project_id WHERE r.id=?", + (frozen["run_id"],), + ).fetchone() + if (run is None or run["project_id"] != project_id + or run["git_common_dir"] != project_common_dir + or run["package_digest"] != frozen["package_digest"]): + raise ContractError("experiment assigned run project/package is inconsistent") + snapshot = _object(frozen["snapshot_json"], "preparation snapshot") + if (snapshot.get("experiment_spec") != spec + or snapshot.get("experiment_assignment") != assignment): + raise ContractError("experiment immutable preparation witness is inconsistent") + inputs = paired_input_identity(snapshot, role=assignment["role"], package_digest=frozen["package_digest"]) + selected = snapshot["routing"]["roles"][assignment["role"]]["selected"] + execution = selected_execution_fingerprint(snapshot, role=assignment["role"]) + if (any(inputs[key] != assignment[key] for key in inputs) + or any(selected[key] != assignment[key] for key in ("profile_id", "profile_sha256")) + or execution != assignment["execution_sha256"]): + raise ContractError("experiment frozen profile/execution/input differs from declaration") + history, exposed = _attempts(connection, run, frozen, snapshot, assignment) + outcomes = connection.execute( + "SELECT * FROM outcomes WHERE run_id=? ORDER BY observed_at,id", (run["id"],), + ).fetchall() + if any(row["kind"] == "final" and row["outcome_id"] != expected["outcome_id"] for row in outcomes): + raise ContractError("experiment run final outcome differs from its assigned outcome") + if final_row is None: + continue + if run["state"] not in {"succeeded", "failed", "cancelled"} or final_row["kind"] != "final": + raise ContractError("experiment outcome requires a terminal assigned run") + final = _outcome(final_row, run, datetime.fromisoformat(evaluated_at)) + corrections = [ + _outcome(row, run, datetime.fromisoformat(evaluated_at)) + for row in outcomes if row["kind"] == "late_correction" + ] + if not exposed: + continue + chain = { + "final": final, "late_corrections": corrections, + "provenance": { + "run_id": run["id"], "assignment": assignment, + "profile_sha256": selected["profile_sha256"], + "execution_sha256": execution, **inputs, + "attempt_ids": [attempt["id"] for attempt in history], + "attempts_sha256": _digest(history), + "final_outcome_sha256": _digest(final), + "correction_sha256": [_digest(value) for value in corrections], + }, + } + validate_arm_chain(chain, spec=spec, case=case, arm=arm, + project_common_dir=project_common_dir, evaluated_at=evaluated_at) + chains[expected["outcome_id"]] = chain + return chains diff --git a/plugin/core/src/devsquad/experiment_provenance.py b/plugin/core/src/devsquad/experiment_provenance.py new file mode 100644 index 0000000..167ff40 --- /dev/null +++ b/plugin/core/src/devsquad/experiment_provenance.py @@ -0,0 +1,290 @@ +"""Pure prelaunch experiment contracts; saved-run authority lives in Store. + +An assignment dictionary alone is not execution evidence. It must be frozen +under the preparation fence and checked against actual saved attempts before +it can authorize evaluation or a lifecycle decision. +""" + +from __future__ import annotations + +import hashlib +from pathlib import Path +import re +from typing import Any + +from .contracts import ContractError +from .store import canonical_json +from .validation import validate_profile, validate_task + + +def require_sha256(value: Any, label: str) -> str: + if not isinstance(value, str) or re.fullmatch(r"[0-9a-f]{64}", value) is None: + raise ContractError(f"experiment {label} SHA-256 is invalid") + return value + + +def _digest(value: Any) -> str: + return hashlib.sha256(canonical_json(value).encode()).hexdigest() + + +def assignment_for( + spec: dict[str, Any], case_id: str, arm: str, *, project_common_dir: str, +) -> dict[str, Any]: + """Construct the exact assignment a fenced prelaunch must persist.""" + from .learning import validate_experiment + + spec = validate_experiment(spec) + if spec["schema_version"] != 2: + raise ContractError("experiment assignment requires a v2 specification") + if not isinstance(arm, str) or arm not in {"control", "candidate"}: + raise ContractError("experiment assignment arm is invalid") + if (not isinstance(project_common_dir, str) or not project_common_dir + or not Path(project_common_dir).is_absolute() + or ".." in Path(project_common_dir).parts): + raise ContractError("experiment project common directory is invalid") + case = next((row for row in spec["cases"] if row["case_id"] == case_id), None) + if case is None: + raise ContractError("experiment assignment case is not declared") + variable = spec["variable"] + return { + "schema_version": 1, + "experiment_id": spec["experiment_id"], + "spec_sha256": _digest(spec), + "project_common_dir": project_common_dir, + "case_id": case["case_id"], + "split": case["split"], + "arm": arm, + "role": variable["role"], + "profile_id": variable[f"{arm}_profile_id"], + "profile_sha256": variable[f"{arm}_profile_sha256"], + "execution_sha256": variable[f"{arm}_execution_sha256"], + "input_sha256": case["input_sha256"], + "case_sha256": case["case_sha256"], + "outcome_id": case[f"{arm}_outcome_id"], + } + + +def validate_assignment( + value: dict[str, Any], *, spec: dict[str, Any], project_common_dir: str, +) -> dict[str, Any]: + if not isinstance(value, dict) or type(value.get("schema_version")) is not int: + raise ContractError("experiment assignment fields are invalid") + expected = assignment_for( + spec, value.get("case_id"), value.get("arm"), + project_common_dir=project_common_dir, + ) + if value != expected: + raise ContractError("experiment assignment does not match its frozen contract") + return expected + + +def _profile_identity(selected: Any) -> dict[str, Any]: + if not isinstance(selected, dict) or not isinstance(selected.get("profile"), dict): + raise ContractError("experiment frozen profile is missing") + profile = selected["profile"] + validate_profile(profile) + if (selected.get("profile_id") != profile["id"] + or selected.get("profile_sha256") != _digest(profile)): + raise ContractError("experiment frozen profile fingerprint is inconsistent") + return {"profile_id": profile["id"], "profile_sha256": selected["profile_sha256"]} + + +def _adapter_identity(snapshot: dict[str, Any], role: str, selected: dict[str, Any]) -> Any: + """Preserve supporting native execution context, never auth paths.""" + if selected["profile"]["harness"] == "fixture": + return {"harness": "fixture"} + prefix = {"implementer": "implementation", "reviewer": "review", "lead": "lead"}[role] + adapters = snapshot.get(f"{prefix}_adapters") + if not isinstance(adapters, dict): + raise ContractError("experiment supporting native adapters are missing") + adapter = adapters.get(selected["profile_id"]) + if not isinstance(adapter, dict): + raise ContractError("experiment supporting native adapter is missing") + require_sha256(adapter.get("binary_sha256"), "native binary") + for key in ("harness", "harness_version", "model_provider", "transport"): + if not isinstance(adapter.get(key), str) or not adapter[key]: + raise ContractError("experiment supporting native adapter identity is invalid") + if adapter["harness"] != selected["profile"]["harness"]: + raise ContractError("experiment supporting native adapter harness is inconsistent") + # Frozen adapter schemas already delimit native settings. Exclude only + # resolved machine-local executable and credential locations, not argv, + # tool/permission arguments, protocol/output-schema or binary fingerprints. + return {key: value for key, value in adapter.items() if key not in {"binary", "auth_file"}} + + +def paired_input_identity( + snapshot: dict[str, Any], *, role: str, package_digest: str, +) -> dict[str, str]: + """Hash corpus identity separately from the full controlled pair context. + + No assignment, outcome, produced implementation or mutable runtime field + participates. Thus predeclaring case hashes cannot create a spec/hash cycle. + """ + if not isinstance(snapshot, dict): + raise ContractError("experiment frozen snapshot is missing") + task = snapshot.get("task") + validate_task(task) + if not isinstance(role, str) or role not in {"implementer", "reviewer"}: + raise ContractError("experiment tested role is invalid") + if role == "implementer" and task["workflow"] != "issue-delivery": + raise ContractError("experiment implementer requires issue-delivery") + require_sha256(package_digest, "runtime package") + for key in ("base_oid", "target_oid"): + if (not isinstance(snapshot.get(key), str) + or re.fullmatch(r"(?:[0-9a-f]{40}|[0-9a-f]{64})", snapshot[key]) is None): + raise ContractError("experiment frozen commit identity is invalid") + source = {key: snapshot[key] for key in ("base_oid", "target_oid")} + if role == "reviewer": + workspace = snapshot.get("workspace") + if not isinstance(workspace, dict): + raise ContractError("experiment frozen review candidate is missing") + source["candidate_sha256"] = require_sha256( + workspace.get("candidate_sha256"), "review candidate", + ) + # Delivery review workspaces can have a new candidate target. Record + # their exact commits in addition to the task's original baseline. + for key in ("base_oid", "target_oid"): + value = workspace.get(key) + if not isinstance(value, str) or re.fullmatch(r"(?:[0-9a-f]{40}|[0-9a-f]{64})", value) is None: + raise ContractError("experiment review candidate commit is invalid") + source[f"candidate_{key}"] = value + else: + delivery = snapshot.get("delivery_workspace") + if not isinstance(delivery, dict) or delivery.get("baseline_oid") != snapshot["target_oid"]: + raise ContractError("experiment frozen implementation baseline is inconsistent") + routing = snapshot.get("routing") + if not isinstance(routing, dict) or not isinstance(routing.get("roles"), dict): + raise ContractError("experiment frozen routing is missing") + expected_roles = {"reviewer"} + if task["workflow"] == "issue-delivery": + expected_roles.add("implementer") + if task["lead"]["mode"] == "headless": + expected_roles.add("lead") + if set(routing["roles"]) != expected_roles or role not in expected_roles: + raise ContractError("experiment frozen roles do not match the task") + selected_execution_fingerprint(snapshot, role=role) + try: + policy_sha256 = require_sha256(snapshot["configs"]["policy_file"]["sha256"], "policy") + if policy_sha256 != routing["policy"]["sha256"]: + raise ContractError("experiment frozen policy fingerprint is inconsistent") + except (KeyError, TypeError) as exc: + raise ContractError("experiment frozen policy is missing") from exc + supporting = {} + tested_fallbacks = [] + for name, route in routing["roles"].items(): + if (not isinstance(route, dict) or not isinstance(route.get("selected"), dict) + or not isinstance(route.get("fallbacks"), list) + or not isinstance(route.get("fallback_mode"), str) + or route["fallback_mode"] not in {"none", "policy"} + or (route["fallback_mode"] == "none" and route["fallbacks"])): + raise ContractError("experiment frozen role selection is invalid") + identities = [] + for index, selected in enumerate([route["selected"], *route["fallbacks"]]): + identity = _profile_identity(selected) + if name != role or index > 0: + context_identity = {**identity, "adapter": _adapter_identity(snapshot, name, selected)} + identities.append(context_identity) + if name == role: + tested_fallbacks.append(context_identity) + if name != role: + supporting[name] = {"fallback_mode": route["fallback_mode"], "profiles": identities} + corpus = { + "schema_version": 1, + "source": source, + "task": {key: task[key] for key in ( + "workflow", "goal", "task_class", "acceptance", "checks", "scope", + )}, + "review": task.get("review", {"mode": "standard"}), + } + context = { + "schema_version": 1, "case_sha256": _digest(corpus), + "tested_role": role, "lead": task["lead"], "budget": task["budget"], + "policy_sha256": policy_sha256, "package_digest": package_digest, + "supporting_roles": supporting, + "tested_fallback_mode": routing["roles"][role]["fallback_mode"], + "tested_fallbacks": tested_fallbacks, + } + # Canonical hashing rejects non-finite/non-serializable selected context; + # only these digests cross the evidence boundary. + return {"case_sha256": context["case_sha256"], "input_sha256": _digest(context)} + + +def selected_execution_fingerprint(snapshot: dict[str, Any], *, role: str) -> str: + """Per-arm concrete execution variable, including native version/binary. + + The experiment explicitly declares one fingerprint for each arm. This + permits an intentional harness change without silently permitting drift. + """ + if not isinstance(role, str) or role not in {"implementer", "reviewer"}: + raise ContractError("experiment tested execution role is invalid") + try: + selected = snapshot["routing"]["roles"][role]["selected"] + except (KeyError, TypeError) as exc: + raise ContractError("experiment tested execution selection is missing") from exc + profile = _profile_identity(selected) + return _digest({"profile_sha256": profile["profile_sha256"], + "adapter": _adapter_identity(snapshot, role, selected)}) + + +def validate_arm_chain( + chain: Any, *, spec: dict[str, Any], case: dict[str, Any], arm: str, + project_common_dir: str, evaluated_at: str, +) -> dict[str, Any]: + """Check a normalized chain supplied by the saved-run evidence reader. + + These shape/hash checks do not authorize importing arbitrary caller + provenance. Store must derive the witness from fenced prelaunch records, + actual attempts and append-only outcomes, not accept a submitted witness. + """ + from .learning import _timestamp, validate_outcome + + if not isinstance(chain, dict) or set(chain) != {"final", "late_corrections", "provenance"}: + raise ContractError("experiment arm requires saved-run provenance") + provenance = chain["provenance"] + fields = { + "run_id", "assignment", "profile_sha256", "input_sha256", "case_sha256", + "attempt_ids", "attempts_sha256", "final_outcome_sha256", "correction_sha256", + "execution_sha256", + } + if not isinstance(provenance, dict) or set(provenance) != fields: + raise ContractError("experiment arm provenance fields are invalid") + expected = assignment_for(spec, case["case_id"], arm, project_common_dir=project_common_dir) + if validate_assignment(provenance["assignment"], spec=spec, project_common_dir=project_common_dir) != expected: + raise ContractError("experiment arm provenance has a crossed assignment") + if any(provenance[key] != expected[key] for key in ( + "profile_sha256", "execution_sha256", "input_sha256", "case_sha256", + )): + raise ContractError("experiment arm provenance does not match actual profile/input") + run_id = provenance["run_id"] + if not isinstance(run_id, str) or not run_id.strip(): + raise ContractError("experiment provenance run id is missing") + attempt_ids = provenance["attempt_ids"] + if (not isinstance(attempt_ids, list) or not attempt_ids + or any(not isinstance(item, str) or not item.strip() for item in attempt_ids) + or len(set(attempt_ids)) != len(attempt_ids)): + raise ContractError("experiment provenance requires distinct actual attempt ids") + require_sha256(provenance["attempts_sha256"], "actual attempts") + final = validate_outcome(chain["final"], now=_timestamp(evaluated_at, "evaluated_at")) + if (final["kind"] != "final" or final["selection_mode"] != "experimental" + or final["outcome_id"] != expected["outcome_id"]): + raise ContractError("experiment provenance final outcome is inconsistent") + if provenance["final_outcome_sha256"] != _digest(final): + raise ContractError("experiment provenance final outcome hash is inconsistent") + corrections = chain["late_corrections"] + if not isinstance(corrections, list): + raise ContractError("experiment provenance corrections must be an array") + correction_ids = set() + hashes = [] + for correction in corrections: + normalized = validate_outcome(correction, now=_timestamp(evaluated_at, "evaluated_at")) + if (normalized["kind"] != "late_correction" or normalized["selection_mode"] != "experimental" + or normalized["corrects_outcome_id"] != final["outcome_id"] + or _timestamp(normalized["observed_at"], "observed_at") + < _timestamp(final["observed_at"], "observed_at") + or normalized["outcome_id"] in correction_ids): + raise ContractError("experiment provenance correction chain is inconsistent") + correction_ids.add(normalized["outcome_id"]) + hashes.append(_digest(normalized)) + if provenance["correction_sha256"] != hashes: + raise ContractError("experiment provenance correction hashes are inconsistent") + return provenance diff --git a/plugin/core/src/devsquad/fake_step.py b/plugin/core/src/devsquad/fake_step.py new file mode 100644 index 0000000..a2f5dab --- /dev/null +++ b/plugin/core/src/devsquad/fake_step.py @@ -0,0 +1,10 @@ +"""Internal deterministic lifecycle fixture. It is not a public task command.""" +import argparse +import sys +import time + +parser = argparse.ArgumentParser() +parser.add_argument("--delay", type=float, default=0.05) +delay = parser.parse_args().delay +time.sleep(delay) +sys.stdout.write("M2_FAKE_STEP_OK\n") diff --git a/plugin/core/src/devsquad/integrations.py b/plugin/core/src/devsquad/integrations.py new file mode 100644 index 0000000..630a34d --- /dev/null +++ b/plugin/core/src/devsquad/integrations.py @@ -0,0 +1,686 @@ +"""Strict local MCP host-registration templates.""" + +from __future__ import annotations + +from dataclasses import dataclass +from importlib import metadata +import importlib.util +import hashlib +import json +import os +from pathlib import Path +import re +import shlex +import shutil +import string +import subprocess +import sys +import tomllib +from typing import Any, Callable + +from .contracts import ContractError + +SOURCE_ROOT = Path(__file__).resolve().parents[2] +CORE_ROOT = ( + SOURCE_ROOT + if (SOURCE_ROOT / "integrations").is_dir() + else Path(sys.prefix) / "share" / "devsquad" +) +TEMPLATE_FIELDS = { + "schema_version", + "id", + "display_name", + "executable_paths", + "executable_names", + "server_name", + "surface", + "register_argv", + "remove_argv", + "inspect_argv", + "inspect_format", +} +PLACEHOLDERS = { + "host_executable", "squad_executable", "server_name", "surface", +} + + +def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise ContractError(f"duplicate integration template key: {key}") + result[key] = value + return result + + +@dataclass(frozen=True) +class IntegrationTemplate: + id: str + display_name: str + executable_paths: tuple[str, ...] + executable_names: tuple[str, ...] + server_name: str + surface: str + register_argv: tuple[str, ...] + remove_argv: tuple[str, ...] | None + inspect_argv: tuple[str, ...] + inspect_format: str + path: Path + + @classmethod + def load(cls, path: Path) -> "IntegrationTemplate": + try: + value = json.loads(path.read_text(), object_pairs_hook=_unique_object) + except (OSError, UnicodeError, json.JSONDecodeError) as exc: + raise ContractError(f"cannot read integration template {path}: {exc}") from exc + if not isinstance(value, dict) or set(value) != TEMPLATE_FIELDS: + fields = set(value) if isinstance(value, dict) else set() + raise ContractError( + f"integration template fields differ: {sorted(fields ^ TEMPLATE_FIELDS)}" + ) + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("unsupported integration template schema_version") + for field in ("id", "display_name", "server_name", "surface"): + if not isinstance(value[field], str) or not value[field]: + raise ContractError(f"integration {field} must be non-empty") + for field in ( + "executable_paths", "executable_names", "register_argv", "inspect_argv", + ): + sequence = value[field] + allow_empty = field == "executable_paths" + if (not isinstance(sequence, list) or (not allow_empty and not sequence) + or not all(isinstance(item, str) and item for item in sequence)): + qualifier = "a string array" if allow_empty else "a non-empty string array" + raise ContractError(f"integration {field} must be {qualifier}") + if field in {"executable_paths", "executable_names"}: + if len(set(sequence)) != len(sequence): + raise ContractError(f"integration {field} must be unique") + if not all(Path(item).is_absolute() for item in value["executable_paths"]): + raise ContractError("integration executable_paths must be absolute") + remove_argv = value["remove_argv"] + if (remove_argv is not None + and (not isinstance(remove_argv, list) or not remove_argv + or not all(isinstance(item, str) and item for item in remove_argv))): + raise ContractError("integration remove_argv must be null or a non-empty string array") + if value["inspect_format"] not in {"json", "text"}: + raise ContractError("integration inspect_format is invalid") + cls._validate_placeholders(value["register_argv"]) + cls._validate_placeholders(value["inspect_argv"]) + if remove_argv is not None: + cls._validate_placeholders(remove_argv) + return cls( + id=value["id"], + display_name=value["display_name"], + executable_paths=tuple(value["executable_paths"]), + executable_names=tuple(value["executable_names"]), + server_name=value["server_name"], + surface=value["surface"], + register_argv=tuple(value["register_argv"]), + remove_argv=tuple(remove_argv) if remove_argv is not None else None, + inspect_argv=tuple(value["inspect_argv"]), + inspect_format=value["inspect_format"], + path=path, + ) + + @staticmethod + def _validate_placeholders(argv: list[str]) -> None: + formatter = string.Formatter() + referenced = { + name + for argument in argv + for _, name, _, _ in formatter.parse(argument) + if name is not None + } + if referenced - PLACEHOLDERS: + raise ContractError( + f"unknown integration placeholders: {sorted(referenced - PLACEHOLDERS)}" + ) + + def _render( + self, + argv: tuple[str, ...], + *, + host_executable: Path, + squad_executable: Path | None, + ) -> tuple[str, ...]: + try: + host = host_executable.resolve(strict=True) + except OSError as exc: + raise ContractError("host executable must exist") from exc + if not host.is_file() or not os.access(host, os.X_OK): + raise ContractError("host executable must be an executable file") + referenced = { + name + for argument in argv + for _, name, _, _ in string.Formatter().parse(argument) + if name is not None + } + squad = None + if "squad_executable" in referenced: + if squad_executable is None: + raise ContractError("squad executable is not resolved") + try: + squad = squad_executable.resolve(strict=True) + except OSError as exc: + raise ContractError("squad executable must exist") from exc + if not squad.is_file() or not os.access(squad, os.X_OK): + raise ContractError("squad executable must be an executable file") + values = { + "host_executable": str(host), + "squad_executable": str(squad) if squad else "", + "server_name": self.server_name, + "surface": self.surface, + } + return tuple(argument.format_map(values) for argument in argv) + + def registration_command( + self, host_executable: Path, squad_executable: Path, + ) -> tuple[str, ...]: + return self._render( + self.register_argv, + host_executable=host_executable, + squad_executable=squad_executable, + ) + + def inspection_command( + self, host_executable: Path, squad_executable: Path | None = None, + ) -> tuple[str, ...]: + return self._render( + self.inspect_argv, + host_executable=host_executable, + squad_executable=squad_executable, + ) + + def removal_command(self, host_executable: Path) -> tuple[str, ...] | None: + if self.remove_argv is None: + return None + return self._render( + self.remove_argv, + host_executable=host_executable, + squad_executable=None, + ) + + +def load_integrations(root: Path | None = None) -> tuple[IntegrationTemplate, ...]: + integration_root = root or (CORE_ROOT / "integrations") + templates = tuple( + IntegrationTemplate.load(path) + for path in sorted(integration_root.glob("*/registration.json")) + ) + if not templates: + raise ContractError("no local MCP integration templates are installed") + ids = [template.id for template in templates] + if len(set(ids)) != len(ids): + raise ContractError("integration template ids must be unique") + return templates + + +def resolve_squad_executable(explicit: Path | None = None) -> Path: + """Resolve one executable that remains valid outside the current shell.""" + + candidates = ( + [explicit] + if explicit is not None + else [ + Path(found) if (found := shutil.which("squad")) else None, + CORE_ROOT / "bin" / "squad", + ] + ) + for candidate in candidates: + if candidate is None: + continue + try: + resolved = candidate.resolve(strict=True) + except OSError: + continue + if resolved.is_file() and os.access(resolved, os.X_OK): + return resolved + raise ContractError( + "cannot resolve a stable squad executable; install devsquad-core or " + "pass --squad-executable" + ) + + +def _registration( + *, scope: str, path: Path, value: Any, disabled_key: str | None = None, +) -> dict[str, Any]: + valid = isinstance(value, dict) + command = value.get("command") if valid else None + args = value.get("args", []) if valid else None + if not isinstance(command, str) or not command: + valid = False + if not isinstance(args, list) or not all(isinstance(item, str) for item in args): + valid = False + enabled = True + if isinstance(value, dict): + if disabled_key is not None: + enabled = value.get(disabled_key) is not True + elif "enabled" in value: + enabled = value.get("enabled") is True + return { + "scope": scope, + "path": str(path), + "command": command if isinstance(command, str) else None, + "args": args if isinstance(args, list) else None, + "enabled": enabled, + "valid": valid, + } + + +def _safe_parse_error(path: Path, format_name: str) -> str: + return f"cannot parse {format_name} configuration at {path}" + + +def _json_document(path: Path) -> Any: + return json.loads(path.read_text(), object_pairs_hook=_unique_object) + + +def _toml_document(path: Path) -> Any: + return tomllib.loads(path.read_text()) + + +def registration_sources( + template: IntegrationTemplate, + *, + home: Path, + project: Path, +) -> tuple[list[dict[str, Any]], list[dict[str, str]]]: + """Find direct and inherited registrations without returning env/secrets.""" + + project = project.resolve() + candidates: list[tuple[str, Path, str, tuple[str, ...], str | None]] = [] + if template.id == "codex": + candidates = [ + ("user", home / ".codex/config.toml", "toml", ("mcp_servers",), None), + ("project", project / ".codex/config.toml", "toml", ("mcp_servers",), None), + ] + elif template.id == "grok": + candidates = [ + ("user", home / ".grok/config.toml", "toml", ("mcp_servers",), None), + ("project", project / ".grok/config.toml", "toml", ("mcp_servers",), None), + ] + elif template.id == "antigravity": + candidates = [ + ("user", home / ".gemini/config/mcp_config.json", "json", ("mcpServers",), "disabled"), + ("project", project / ".gemini/config/mcp_config.json", "json", ("mcpServers",), "disabled"), + ("project", project / ".gemini/settings.json", "json", ("mcpServers",), "disabled"), + ] + elif template.id == "claude-code": + candidates = [ + ("user", home / ".claude.json", "json", ("mcpServers",), None), + ("project", project / ".mcp.json", "json", ("mcpServers",), None), + ( + "local", + home / ".claude.json", + "json", + ("projects", str(project), "mcpServers"), + None, + ), + ] + registrations: list[dict[str, Any]] = [] + errors: list[dict[str, str]] = [] + documents: dict[tuple[Path, str], Any] = {} + invalid_documents: set[tuple[Path, str]] = set() + for scope, path, format_name, key_path, disabled_key in candidates: + if not path.is_file(): + continue + cache_key = (path, format_name) + if cache_key in invalid_documents: + continue + if cache_key not in documents: + try: + documents[cache_key] = ( + _json_document(path) if format_name == "json" else _toml_document(path) + ) + except (OSError, UnicodeError, json.JSONDecodeError, tomllib.TOMLDecodeError, ContractError): + invalid_documents.add(cache_key) + errors.append({ + "scope": scope, + "path": str(path), + "error": _safe_parse_error(path, format_name), + }) + continue + value = documents[cache_key] + for key in key_path: + if not isinstance(value, dict) or key not in value: + value = None + break + value = value[key] + if isinstance(value, dict) and template.server_name in value: + registrations.append(_registration( + scope=scope, + path=path, + value=value[template.server_name], + disabled_key=disabled_key, + )) + return registrations, errors + + +def _actual_from_json(template: IntegrationTemplate, stdout: str) -> dict[str, Any] | None: + try: + value = json.loads(stdout) + except json.JSONDecodeError: + return None + if template.id == "codex" and isinstance(value, dict): + transport = value.get("transport") + if value.get("name") == template.server_name and isinstance(transport, dict): + return { + "command": transport.get("command"), + "args": transport.get("args"), + "enabled": value.get("enabled") is True, + "scope": None, + } + if template.id == "grok" and isinstance(value, list): + for item in value: + if isinstance(item, dict) and item.get("name") == template.server_name: + return { + "command": item.get("command"), + "args": item.get("args"), + "enabled": item.get("enabled") is True, + "scope": item.get("scope"), + } + return None + + +def _actual_from_text(template: IntegrationTemplate, stdout: str) -> dict[str, Any] | None: + if template.id == "claude-code": + fields: dict[str, str] = {} + for line in stdout.splitlines(): + if ":" in line: + key, value = line.split(":", 1) + fields[key.strip().lower()] = value.strip() + if template.server_name + ":" not in stdout.lower() or "command" not in fields: + return None + try: + args = shlex.split(fields.get("args", "")) + except ValueError: + return None + return { + "command": fields["command"], + "args": args, + "enabled": not fields.get("status", "").lower().startswith("disabled"), + "scope": fields.get("scope"), + "connection": ( + "failed" if "failed" in fields.get("status", "").lower() + else "connected" if "connected" in fields.get("status", "").lower() + else "unknown" + ), + } + if template.id == "antigravity": + for line in stdout.splitlines(): + columns = re.split(r"\s{2,}", line.strip(), maxsplit=3) + if len(columns) == 4 and columns[0] == template.server_name: + try: + command = shlex.split(columns[3]) + except ValueError: + return None + if not command: + return None + expected_args = ["mcp", "serve", "--surface", template.surface] + if len(command) > len(expected_args) and command[-len(expected_args):] == expected_args: + executable = " ".join(command[:-len(expected_args)]) + arguments = expected_args + else: + executable = command[0] + arguments = command[1:] + return { + "command": executable, + "args": arguments, + "enabled": columns[2].lower() == "enabled", + "scope": None, + } + return None + + +def _sdk_version() -> str | None: + if importlib.util.find_spec("mcp") is None: + return None + try: + return metadata.version("mcp") + except metadata.PackageNotFoundError: + return None + + +def _redacted_registration( + registration: dict[str, Any], expected: dict[str, Any], +) -> dict[str, Any]: + """Expose matching argv, but never echo arbitrary drifted arguments.""" + + arguments = registration.get("args") + args_match = arguments == expected["args"] + public = { + key: value + for key, value in registration.items() + if key not in {"args"} + } + public["args"] = arguments if args_match else None + public["args_match"] = args_match + if isinstance(arguments, list): + canonical = json.dumps(arguments, ensure_ascii=True, separators=(",", ":")) + public["args_sha256"] = hashlib.sha256(canonical.encode()).hexdigest() + public["args_count"] = len(arguments) + else: + public["args_sha256"] = None + public["args_count"] = None + return public + + +class LocalIntegrationManager: + """Inspect and idempotently register the local stdio server via host CLIs.""" + + def __init__( + self, + *, + project: Path, + home: Path | None = None, + squad_executable: Path | None = None, + which: Callable[[str], str | None] = shutil.which, + runner: Callable[..., subprocess.CompletedProcess[str]] = subprocess.run, + timeout_seconds: float = 8, + mcp_sdk_available: bool | None = None, + mcp_sdk_version: str | None = None, + ): + self.project = project.resolve(strict=True) + self.home = (home or Path.home()).resolve(strict=True) + self.launcher_error = None + try: + self.squad_executable = resolve_squad_executable(squad_executable) + except ContractError as exc: + self.squad_executable = None + self.launcher_error = str(exc) + self.which = which + self.runner = runner + self.timeout_seconds = timeout_seconds + detected_version = _sdk_version() + self.mcp_sdk_version = ( + detected_version if mcp_sdk_version is None else mcp_sdk_version + ) + self.mcp_sdk_available = ( + detected_version is not None + if mcp_sdk_available is None else mcp_sdk_available + ) + self.mcp_sdk_supported = ( + self.mcp_sdk_available and self.mcp_sdk_version == "2.2.0" + ) + + def _host_executable(self, template: IntegrationTemplate) -> Path | None: + candidates: list[str] = list(template.executable_paths) + candidates.extend( + found + for name in template.executable_names + if (found := self.which(name)) is not None + ) + for found in candidates: + try: + resolved = Path(found).resolve(strict=True) + except OSError: + continue + if resolved.is_file() and os.access(resolved, os.X_OK): + return resolved + return None + + def _run(self, command: tuple[str, ...]) -> subprocess.CompletedProcess[str] | None: + environment = os.environ.copy() + environment["HOME"] = str(self.home) + try: + return self.runner( + command, + cwd=self.project, + env=environment, + text=True, + capture_output=True, + timeout=self.timeout_seconds, + ) + except (OSError, subprocess.TimeoutExpired): + return None + + def inspect(self, template: IntegrationTemplate) -> dict[str, Any]: + host = self._host_executable(template) + expected = { + "command": str(self.squad_executable) if self.squad_executable else None, + "args": ["mcp", "serve", "--surface", template.surface], + } + sources, source_errors = registration_sources( + template, home=self.home, project=self.project, + ) + actual = None + inspection = {"completed": False, "exit_code": None} + if host is not None: + command = template.inspection_command(host) + completed = self._run(command) + if completed is not None: + inspection = { + "completed": True, + "exit_code": completed.returncode, + } + if completed.returncode == 0: + actual = ( + _actual_from_json(template, completed.stdout) + if template.inspect_format == "json" + else _actual_from_text(template, completed.stdout) + ) + matches = bool( + actual + and actual.get("command") == expected["command"] + and actual.get("args") == expected["args"] + and actual.get("enabled") is True + ) + loaded_matches_source = bool( + actual + and len(sources) == 1 + and actual.get("command") == sources[0].get("command") + and actual.get("args") == sources[0].get("args") + and actual.get("enabled") == sources[0].get("enabled") + ) + if host is None: + status = "unavailable" + elif source_errors: + status = "invalid_config" + elif len(sources) > 1: + status = "duplicate" + elif any(not source["valid"] for source in sources): + status = "invalid_config" + elif self.squad_executable is None: + status = "unstable_launcher" + elif actual is None and sources: + status = "not_loaded" + elif actual is None: + status = "missing" + elif not sources: + status = "inherited" + elif not loaded_matches_source: + status = "duplicate" + elif matches: + status = "matching" + else: + status = "drifted" + public_sources = [ + _redacted_registration(source, expected) for source in sources + ] + public_actual = ( + _redacted_registration(actual, expected) if actual is not None else None + ) + return { + "id": template.id, + "display_name": template.display_name, + "installed": host is not None, + "host_executable": str(host) if host else None, + "server_name": template.server_name, + "status": status, + "ready": status == "matching" and self.mcp_sdk_supported, + "mcp_sdk_available": self.mcp_sdk_available, + "mcp_sdk_supported": self.mcp_sdk_supported, + "mcp_sdk_version": self.mcp_sdk_version, + "expected": expected, + "loaded": public_actual, + "sources": public_sources, + "source_errors": source_errors, + "inspection": inspection, + "template": str(template.path), + "launcher_error": self.launcher_error, + } + + def setup(self, template: IntegrationTemplate, *, dry_run: bool = False) -> dict[str, Any]: + before = self.inspect(template) + status = before["status"] + if not self.mcp_sdk_available: + return {**before, "action": "blocked_missing_mcp_sdk", "changed": False} + if not self.mcp_sdk_supported: + return {**before, "action": "blocked_unsupported_mcp_sdk", "changed": False} + if status == "unavailable": + return {**before, "action": "blocked_host_unavailable", "changed": False} + if status == "unstable_launcher": + return {**before, "action": "blocked_unstable_launcher", "changed": False} + if status == "matching": + return {**before, "action": "unchanged", "changed": False} + if status in {"duplicate", "inherited", "not_loaded", "invalid_config"}: + return {**before, "action": f"blocked_{status}", "changed": False} + if status == "drifted" and before["sources"][0]["scope"] != "user": + return {**before, "action": "blocked_inherited", "changed": False} + action = "add" if status == "missing" else "update" + if dry_run: + return {**before, "action": f"would_{action}", "changed": False} + host = Path(before["host_executable"]) + removed = False + removal_command = template.removal_command(host) if action == "update" else None + if removal_command is not None: + removal = self._run(removal_command) + if removal is None or removal.returncode != 0: + return { + **before, + "action": "update_failed", + "changed": None, + "removal_exit_code": ( + removal.returncode if removal is not None else None + ), + } + removed = True + completed = self._run( + template.registration_command(host, self.squad_executable) + ) + if completed is None or completed.returncode != 0: + return { + **before, + "action": f"{action}_failed", + "changed": True if removed else None, + "removal_exit_code": 0 if removed else None, + "registration_exit_code": ( + completed.returncode if completed is not None else None + ), + } + after = self.inspect(template) + if not after["ready"]: + return { + **after, + "action": f"{action}_incomplete", + "changed": True, + "removal_exit_code": 0 if removed else None, + "registration_exit_code": completed.returncode, + } + return { + **after, + "action": "added" if action == "add" else "updated", + "changed": True, + "removal_exit_code": 0 if removed else None, + "registration_exit_code": completed.returncode, + } diff --git a/plugin/core/src/devsquad/lead_worker.py b/plugin/core/src/devsquad/lead_worker.py new file mode 100644 index 0000000..bb9b914 --- /dev/null +++ b/plugin/core/src/devsquad/lead_worker.py @@ -0,0 +1,45 @@ +"""Run one frozen offline headless-lead fixture for branch-review tests.""" + +from __future__ import annotations + +import json +import sys +from typing import Any + +from .contracts import ContractError +from .store import canonical_json +from .workflows import MAX_EVIDENCE_BYTES, make_headless_lead_evidence + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("workflow snapshot must be an object") + fixture = snapshot.get("internal_lead_fixture") + handoff = snapshot.get("headless_handoff") + if not isinstance(fixture, dict) or not isinstance(handoff, dict): + raise ContractError("offline headless lead snapshot is incomplete") + selected = snapshot["routing"]["roles"]["lead"]["selected"] + if selected.get("profile_id", "").endswith("-fixture-fail"): + raise ContractError("offline headless lead fixture requested a failed attempt") + choice = { + "schema_version": 1, + "candidate_sha256": handoff["packet"]["candidate_sha256"], + **fixture, + } + return make_headless_lead_evidence(snapshot, handoff, choice) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_EVIDENCE_BYTES + 1) + if len(payload) > MAX_EVIDENCE_BYTES: + raise ContractError("workflow snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("workflow snapshot is not valid UTF-8 JSON") from exc + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/learning.py b/plugin/core/src/devsquad/learning.py new file mode 100644 index 0000000..07d494e --- /dev/null +++ b/plugin/core/src/devsquad/learning.py @@ -0,0 +1,871 @@ +"""Strict outcome evidence and comparison primitives for M6 learning.""" + +from __future__ import annotations + +from datetime import datetime, timedelta, timezone +import hashlib +import json +from pathlib import Path +from typing import Any + +from .contracts import ContractError +from .store import canonical_json + + +OUTCOME_FIELDS = { + "schema_version", "outcome_id", "kind", "verdict", "selection_mode", + "observed_at", "corrects_outcome_id", "summary", "criteria", + "contributions", "lead_repairs", "evidence_refs", +} +FINAL_VERDICTS = {"succeeded", "failed", "cancelled"} +LATE_VERDICTS = {"escaped_defect", "corrected"} +SELECTION_MODES = {"automatic", "pinned", "experimental"} +CRITERION_STATES = {"passed", "failed", "unknown"} +CONTRIBUTION_RESULTS = {"failed", "successful", "repair", "finding", "neutral"} +ROLES = {"worker", "implementer", "reviewer", "lead", "researcher", "proposer_a", "proposer_b", "critic"} +MAX_CLOCK_SKEW = timedelta(minutes=5) +EXPERIMENT_FIELDS = { + "schema_version", "experiment_id", "project_path", "question", "hypothesis", + "evidence_availability", "variable", "cases", "gate", "budget", + "rollback_target", +} +EXPERIMENT_EVALUATION_FIELDS = { + "schema_version", "experiment_id", "spec_sha256", "evaluated_at", + "verdict", "active_policy_changed", "reasons", "metrics", "cases", + "failures", "variable", "rollback_target", "evidence_availability", +} +EXPERIMENT_RECORD_FIELDS = { + "experiment", "spec_sha256", "evaluation", "evaluation_sha256", + "recorded_at", +} + + +def _now(value: datetime | None) -> datetime: + current = datetime.now(timezone.utc) if value is None else value + if not isinstance(current, datetime) or current.tzinfo is None or current.utcoffset() is None: + raise ContractError("outcome evaluation time must include a timezone") + return current.astimezone(timezone.utc) + + +def _timestamp(value: Any, field: str) -> datetime: + if not isinstance(value, str) or not value: + raise ContractError(f"outcome {field} must be an ISO timestamp") + try: + parsed = datetime.fromisoformat(value) + except ValueError as exc: + raise ContractError(f"outcome {field} must be an ISO timestamp") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ContractError(f"outcome {field} must include a timezone") + return parsed.astimezone(timezone.utc) + + +def _identifier(value: Any, field: str, *, nullable: bool = False) -> str | None: + if value is None and nullable: + return None + if not isinstance(value, str) or not value.strip(): + suffix = " or null" if nullable else "" + raise ContractError(f"outcome {field} must be a non-empty string{suffix}") + return value + + +def _evidence_refs(value: Any, field: str, *, required: bool = False) -> list[str]: + if not isinstance(value, list) or any( + not isinstance(item, str) or not item for item in value + ): + raise ContractError(f"outcome {field} must be a string array") + if len(set(value)) != len(value): + raise ContractError(f"outcome {field} must be unique") + if required and not value: + raise ContractError(f"outcome {field} must not be empty") + return list(value) + + +def validate_outcome( + value: dict[str, Any], *, now: datetime | None = None, +) -> dict[str, Any]: + """Validate and detach one final or late-correction outcome record.""" + if not isinstance(value, dict) or set(value) != OUTCOME_FIELDS: + unknown = sorted(set(value) - OUTCOME_FIELDS) if isinstance(value, dict) else [] + missing = sorted(OUTCOME_FIELDS - set(value)) if isinstance(value, dict) else [] + raise ContractError( + f"outcome fields invalid: unknown={unknown} missing={missing}", + ) + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("outcome schema_version is invalid") + outcome_id = _identifier(value["outcome_id"], "outcome_id") + kind = value["kind"] + if not isinstance(kind, str) or kind not in {"final", "late_correction"}: + raise ContractError("outcome kind is invalid") + verdict = value["verdict"] + allowed_verdicts = FINAL_VERDICTS if kind == "final" else LATE_VERDICTS + if not isinstance(verdict, str) or verdict not in allowed_verdicts: + raise ContractError("outcome verdict is invalid for its kind") + selection_mode = value["selection_mode"] + if not isinstance(selection_mode, str) or selection_mode not in SELECTION_MODES: + raise ContractError("outcome selection_mode is invalid") + corrects = _identifier( + value["corrects_outcome_id"], "corrects_outcome_id", nullable=True, + ) + if (kind == "final" and corrects is not None) or ( + kind == "late_correction" and corrects is None + ): + raise ContractError("outcome correction reference is invalid") + if not isinstance(value["summary"], str) or not value["summary"].strip(): + raise ContractError("outcome summary must be a non-empty string") + observed = _timestamp(value["observed_at"], "observed_at") + if observed > _now(now) + MAX_CLOCK_SKEW: + raise ContractError("outcome observed_at exceeds allowed clock skew") + + if not isinstance(value["criteria"], list): + raise ContractError("outcome criteria must be an array") + criteria = [] + criterion_ids = set() + for item in value["criteria"]: + if not isinstance(item, dict) or set(item) != { + "criterion_id", "status", "evidence_refs", + }: + raise ContractError("outcome criterion fields are invalid") + criterion_id = _identifier(item["criterion_id"], "criterion_id") + if criterion_id in criterion_ids: + raise ContractError("outcome criterion ids must be unique") + criterion_ids.add(criterion_id) + if not isinstance(item["status"], str) or item["status"] not in CRITERION_STATES: + raise ContractError("outcome criterion status is invalid") + criteria.append({ + "criterion_id": criterion_id, + "status": item["status"], + "evidence_refs": _evidence_refs( + item["evidence_refs"], "criterion evidence_refs", + ), + }) + if kind == "final" and verdict == "succeeded" and any( + item["status"] != "passed" for item in criteria + ): + raise ContractError("successful outcome criteria must all pass") + + if not isinstance(value["contributions"], list): + raise ContractError("outcome contributions must be an array") + contributions = [] + contribution_attempts = set() + for item in value["contributions"]: + if not isinstance(item, dict) or set(item) != { + "attempt_id", "role", "result", "independent_success", "evidence_refs", + }: + raise ContractError("outcome contribution fields are invalid") + attempt_id = _identifier(item["attempt_id"], "attempt_id") + if attempt_id in contribution_attempts: + raise ContractError("outcome contribution attempt ids must be unique") + contribution_attempts.add(attempt_id) + if not isinstance(item["role"], str) or item["role"] not in ROLES: + raise ContractError("outcome contribution role is invalid") + if not isinstance(item["result"], str) or item["result"] not in CONTRIBUTION_RESULTS: + raise ContractError("outcome contribution result is invalid") + if type(item["independent_success"]) is not bool: + raise ContractError("outcome independent_success must be boolean") + if item["independent_success"] and item["result"] != "successful": + raise ContractError("only a successful contribution can be independently successful") + contributions.append({ + "attempt_id": attempt_id, + "role": item["role"], + "result": item["result"], + "independent_success": item["independent_success"], + "evidence_refs": _evidence_refs( + item["evidence_refs"], "contribution evidence_refs", + ), + }) + + if not isinstance(value["lead_repairs"], list): + raise ContractError("outcome lead_repairs must be an array") + lead_repairs = [] + for item in value["lead_repairs"]: + if not isinstance(item, dict) or set(item) != { + "lead_attempt_id", "description", "evidence_refs", + }: + raise ContractError("outcome lead repair fields are invalid") + lead_attempt_id = _identifier( + item["lead_attempt_id"], "lead_attempt_id", nullable=True, + ) + if not isinstance(item["description"], str) or not item["description"].strip(): + raise ContractError("outcome lead repair description is invalid") + lead_repairs.append({ + "lead_attempt_id": lead_attempt_id, + "description": item["description"], + "evidence_refs": _evidence_refs( + item["evidence_refs"], "lead repair evidence_refs", + ), + }) + + normalized = { + **value, + "outcome_id": outcome_id, + "corrects_outcome_id": corrects, + "criteria": criteria, + "contributions": contributions, + "lead_repairs": lead_repairs, + "evidence_refs": _evidence_refs( + value["evidence_refs"], "evidence_refs", required=True, + ), + } + return json.loads(canonical_json(normalized)) + + +def build_comparison_report( + *, + project_id: str | None, + project_path: str, + terminal_runs: list[dict[str, Any]], + outcome_records: list[dict[str, Any]], + attempt_profiles: dict[str, str | None], + generated_at: str, +) -> dict[str, Any]: + """Aggregate outcomes without conflating final success and worker quality.""" + if project_id is not None and (not isinstance(project_id, str) or not project_id): + raise ContractError("learning report project_id is invalid") + if not isinstance(project_path, str) or not project_path: + raise ContractError("learning report project_path is invalid") + _timestamp(generated_at, "generated_at") + if not isinstance(terminal_runs, list) or not isinstance(outcome_records, list): + raise ContractError("learning report inputs are invalid") + if not isinstance(attempt_profiles, dict): + raise ContractError("learning report attempt profiles are invalid") + + terminal_by_id = {} + for run in terminal_runs: + if (not isinstance(run, dict) or set(run) != {"run_id", "state"} + or not isinstance(run["run_id"], str) + or run["state"] not in FINAL_VERDICTS): + raise ContractError("learning report terminal run is invalid") + terminal_by_id[run["run_id"]] = run["state"] + + finals: dict[str, dict[str, Any]] = {} + corrections: dict[str, list[dict[str, Any]]] = {} + for record in outcome_records: + if (not isinstance(record, dict) or set(record) != {"run_id", "outcome"} + or record["run_id"] not in terminal_by_id + or not isinstance(record["outcome"], dict)): + raise ContractError("learning report outcome record is invalid") + outcome = record["outcome"] + if outcome.get("kind") == "final": + if record["run_id"] in finals: + raise ContractError("learning report has duplicate final outcomes") + finals[record["run_id"]] = outcome + elif outcome.get("kind") == "late_correction": + corrections.setdefault(record["run_id"], []).append(outcome) + else: + raise ContractError("learning report outcome kind is invalid") + + modes = { + mode: { + "sample_size": 0, + "succeeded": 0, + "failed": 0, + "cancelled": 0, + "escaped_defects": 0, + "success_rate": None, + } + for mode in sorted(SELECTION_MODES) + } + profiles: dict[str, dict[str, Any]] = {} + lead_repairs = 0 + missing_profile_contributions = 0 + finals_without_contributions = 0 + for run_id, final in finals.items(): + mode = final["selection_mode"] + mode_row = modes[mode] + mode_row["sample_size"] += 1 + mode_row[final["verdict"]] += 1 + escaped = sum( + correction["verdict"] == "escaped_defect" + for correction in corrections.get(run_id, []) + ) + mode_row["escaped_defects"] += escaped + lead_repairs += len(final["lead_repairs"]) + if not final["contributions"]: + finals_without_contributions += 1 + for contribution in final["contributions"]: + profile_id = attempt_profiles.get(contribution["attempt_id"]) + if profile_id is None: + missing_profile_contributions += 1 + continue + row = profiles.setdefault(profile_id, { + "contributions": 0, + "independent_successes": 0, + "failed": 0, + "repairs": 0, + "findings": 0, + }) + row["contributions"] += 1 + row["independent_successes"] += int( + contribution["independent_success"], + ) + if contribution["result"] == "failed": + row["failed"] += 1 + elif contribution["result"] == "repair": + row["repairs"] += 1 + elif contribution["result"] == "finding": + row["findings"] += 1 + for row in modes.values(): + if row["sample_size"]: + row["success_rate"] = row["succeeded"] / row["sample_size"] + + missing_run_ids = sorted(set(terminal_by_id) - set(finals)) + return { + "schema_version": 1, + "project_id": project_id, + "project_path": project_path, + "generated_at": generated_at, + "sample_size": len(finals), + "terminal_run_count": len(terminal_by_id), + "final_successes": sum( + outcome["verdict"] == "succeeded" for outcome in finals.values() + ), + "escaped_defects": sum( + correction["verdict"] == "escaped_defect" + for history in corrections.values() + for correction in history + ), + "lead_repairs": lead_repairs, + "selection_modes": modes, + "profiles": {profile_id: profiles[profile_id] for profile_id in sorted(profiles)}, + "missingness": { + "terminal_runs_without_final_outcome": len(missing_run_ids), + "terminal_run_ids_without_final_outcome": missing_run_ids, + "finals_without_contributions": finals_without_contributions, + "contributions_without_profile_id": missing_profile_contributions, + }, + "interpretation": { + "final_task_success_is_not_profile_success": True, + "selection_modes_are_not_pooled": True, + }, + } + + +def validate_experiment(value: dict[str, Any]) -> dict[str, Any]: + """Validate a predeclared one-variable, paired outcome experiment.""" + if not isinstance(value, dict) or set(value) != EXPERIMENT_FIELDS: + raise ContractError("experiment fields are invalid") + if type(value["schema_version"]) is not int or value["schema_version"] not in {1, 2}: + raise ContractError("experiment schema_version is invalid") + experiment_id = _identifier(value["experiment_id"], "experiment_id") + project_path = value["project_path"] + if (not isinstance(project_path, str) or not project_path + or not Path(project_path).is_absolute()): + raise ContractError("experiment project_path must be absolute") + for field in ("question", "hypothesis"): + if not isinstance(value[field], str) or not value[field].strip(): + raise ContractError(f"experiment {field} must be a non-empty string") + if (not isinstance(value["evidence_availability"], str) + or value["evidence_availability"] not in { + "local", "tracked_fixture", "unavailable", + }): + raise ContractError("experiment evidence_availability is invalid") + variable = value["variable"] + variable_fields = { + "kind", "alias", "control_profile_id", "candidate_profile_id", + } + if value["schema_version"] == 2: + variable_fields.update({ + "role", "control_profile_sha256", "candidate_profile_sha256", + "control_execution_sha256", "candidate_execution_sha256", + }) + if not isinstance(variable, dict) or set(variable) != variable_fields: + raise ContractError("experiment variable fields are invalid") + if variable["kind"] != "profile_binding": + raise ContractError("experiment variable kind is invalid") + for field in ("alias", "control_profile_id", "candidate_profile_id"): + _identifier(variable[field], f"variable.{field}") + if variable["control_profile_id"] == variable["candidate_profile_id"]: + raise ContractError("experiment control and candidate must differ") + if value["schema_version"] == 2: + from .experiment_provenance import require_sha256 + + if (not isinstance(variable["role"], str) + or variable["role"] not in {"implementer", "reviewer"}): + raise ContractError("experiment tested role is invalid") + for arm in ("control", "candidate"): + require_sha256(variable[f"{arm}_profile_sha256"], f"{arm} profile") + require_sha256(variable[f"{arm}_execution_sha256"], f"{arm} execution") + + budget = value["budget"] + if not isinstance(budget, dict) or set(budget) != { + "max_cases", "max_worker_invocations", "wall_seconds", + }: + raise ContractError("experiment budget fields are invalid") + for field in ("max_cases", "wall_seconds"): + if type(budget[field]) is not int or budget[field] < 1: + raise ContractError(f"experiment budget {field} must be positive") + if type(budget["max_worker_invocations"]) is not int or budget["max_worker_invocations"] < 0: + raise ContractError("experiment max_worker_invocations must be non-negative") + + cases = value["cases"] + if not isinstance(cases, list) or not cases or len(cases) > budget["max_cases"]: + raise ContractError("experiment cases exceed the bounded case budget") + case_ids = set() + outcome_ids = set() + case_hashes = set() + normalized_cases = [] + split_counts = {"evaluation": 0, "held_out": 0} + for case in cases: + case_fields = { + "case_id", "split", "control_outcome_id", "candidate_outcome_id", + } + if value["schema_version"] == 2: + case_fields.update({"input_sha256", "case_sha256"}) + if not isinstance(case, dict) or set(case) != case_fields: + raise ContractError("experiment case fields are invalid") + case_id = _identifier(case["case_id"], "case_id") + if case_id in case_ids: + raise ContractError("experiment case ids must be unique") + case_ids.add(case_id) + if not isinstance(case["split"], str) or case["split"] not in split_counts: + raise ContractError("experiment case split is invalid") + split_counts[case["split"]] += 1 + control_id = _identifier(case["control_outcome_id"], "control_outcome_id") + candidate_id = _identifier( + case["candidate_outcome_id"], "candidate_outcome_id", + ) + if control_id == candidate_id: + raise ContractError("experiment paired outcomes must differ") + if control_id in outcome_ids or candidate_id in outcome_ids: + raise ContractError("experiment outcome ids must be globally unique") + outcome_ids.update((control_id, candidate_id)) + if value["schema_version"] == 2: + require_sha256(case["input_sha256"], "paired input") + require_sha256(case["case_sha256"], "corpus case") + if case["case_sha256"] in case_hashes: + raise ContractError("experiment corpus case identities must be unique") + case_hashes.add(case["case_sha256"]) + normalized_cases.append({ + **case, + "case_id": case_id, + "split": case["split"], + "control_outcome_id": control_id, + "candidate_outcome_id": candidate_id, + }) + + gate = value["gate"] + if not isinstance(gate, dict) or set(gate) != { + "min_evaluation_pairs", "min_held_out_pairs", "noninferiority_margin", + "minimum_success_gain", "max_candidate_escaped_defects", + }: + raise ContractError("experiment gate fields are invalid") + for field, split in ( + ("min_evaluation_pairs", "evaluation"), + ("min_held_out_pairs", "held_out"), + ): + if (type(gate[field]) is not int or gate[field] < 1 + or gate[field] > split_counts[split]): + raise ContractError(f"experiment gate {field} is invalid") + for field in ("noninferiority_margin", "minimum_success_gain"): + number = gate[field] + if (isinstance(number, bool) or not isinstance(number, (int, float)) + or not 0 <= number <= 1): + raise ContractError(f"experiment gate {field} must be between zero and one") + if (type(gate["max_candidate_escaped_defects"]) is not int + or gate["max_candidate_escaped_defects"] < 0): + raise ContractError("experiment escaped-defect gate is invalid") + + rollback = value["rollback_target"] + if not isinstance(rollback, dict) or set(rollback) != { + "profile_id", "binding_version", + }: + raise ContractError("experiment rollback target fields are invalid") + if rollback["profile_id"] != variable["control_profile_id"]: + raise ContractError("experiment rollback target must be the control profile") + if type(rollback["binding_version"]) is not int or rollback["binding_version"] < 1: + raise ContractError("experiment rollback binding_version is invalid") + return json.loads(canonical_json({ + **value, + "experiment_id": experiment_id, + "variable": dict(variable), + "cases": normalized_cases, + "gate": dict(gate), + "budget": dict(budget), + "rollback_target": dict(rollback), + })) + + +def evaluate_experiment( + experiment: dict[str, Any], + outcome_chains: dict[str, dict[str, Any]], + *, + evaluated_at: str, + project_common_dir: str | None = None, +) -> dict[str, Any]: + """Evaluate a frozen paired experiment without changing active policy.""" + spec = validate_experiment(experiment) + _timestamp(evaluated_at, "evaluated_at") + if not isinstance(outcome_chains, dict): + raise ContractError("experiment outcome chains are invalid") + provenance = {} + if spec["schema_version"] == 2: + from .experiment_provenance import validate_arm_chain + + if not isinstance(project_common_dir, str) or not Path(project_common_dir).is_absolute(): + raise ContractError("experiment provenance requires the saved project identity") + run_ids = set() + attempt_ids = set() + # Validate every available arm, even when its partner is missing. A + # missing partner must not hide reused or crossed evidence. + for case in spec["cases"]: + for arm in ("control", "candidate"): + outcome_id = case[f"{arm}_outcome_id"] + chain = outcome_chains.get(outcome_id) + if chain is None: + provenance[outcome_id] = None + continue + witness = validate_arm_chain( + chain, spec=spec, case=case, arm=arm, + project_common_dir=project_common_dir, evaluated_at=evaluated_at, + ) + if witness["run_id"] in run_ids: + raise ContractError("experiment provenance run ids must be globally unique") + if attempt_ids.intersection(witness["attempt_ids"]): + raise ContractError("experiment provenance attempt ids must be globally unique") + run_ids.add(witness["run_id"]) + attempt_ids.update(witness["attempt_ids"]) + provenance[outcome_id] = witness + rows = [] + metrics = { + split: { + "declared_pairs": 0, + "available_pairs": 0, + "control_successes": 0, + "candidate_successes": 0, + "candidate_escaped_defects": 0, + "control_success_rate": None, + "candidate_success_rate": None, + "success_gain": None, + } + for split in ("evaluation", "held_out") + } + failures = [] + for case in spec["cases"]: + split = case["split"] + metrics[split]["declared_pairs"] += 1 + control = outcome_chains.get(case["control_outcome_id"]) + candidate = outcome_chains.get(case["candidate_outcome_id"]) + missing = [] + if control is None: + missing.append("control") + if candidate is None: + missing.append("candidate") + row = {**case, "status": "missing" if missing else "available", "missing": missing} + if missing: + failures.append({"case_id": case["case_id"], "reason": "missing_outcome"}) + rows.append(row) + continue + for arm, chain in (("control", control), ("candidate", candidate)): + chain_fields = {"final", "late_corrections"} + if spec["schema_version"] == 2: + chain_fields.add("provenance") + if (not isinstance(chain, dict) or set(chain) != chain_fields + or not isinstance(chain["final"], dict) + or not isinstance(chain["late_corrections"], list)): + raise ContractError("experiment outcome chain is invalid") + if chain["final"].get("kind") != "final": + raise ContractError("experiment arm must reference a final outcome") + if chain["final"].get("selection_mode") != "experimental": + raise ContractError("experiment outcomes must be explicitly experimental") + if any(not isinstance(correction, dict) + for correction in chain["late_corrections"]): + raise ContractError("experiment late corrections are invalid") + row[f"{arm}_verdict"] = chain["final"]["verdict"] + row[f"{arm}_escaped_defects"] = sum( + correction.get("verdict") == "escaped_defect" + for correction in chain["late_corrections"] + ) + metrics[split]["available_pairs"] += 1 + metrics[split]["control_successes"] += int(row["control_verdict"] == "succeeded") + metrics[split]["candidate_successes"] += int( + row["candidate_verdict"] == "succeeded", + ) + metrics[split]["candidate_escaped_defects"] += row[ + "candidate_escaped_defects" + ] + if row["candidate_verdict"] != "succeeded": + failures.append({ + "case_id": case["case_id"], "reason": "candidate_not_successful", + }) + if row["candidate_escaped_defects"]: + failures.append({ + "case_id": case["case_id"], "reason": "candidate_escaped_defect", + }) + rows.append(row) + + reasons = [] + total_candidate_escaped = 0 + for split, row in metrics.items(): + minimum = spec["gate"][ + "min_evaluation_pairs" if split == "evaluation" else "min_held_out_pairs" + ] + if row["available_pairs"] < minimum: + reasons.append(f"insufficient_{split}_pairs") + continue + row["control_success_rate"] = row["control_successes"] / row["available_pairs"] + row["candidate_success_rate"] = ( + row["candidate_successes"] / row["available_pairs"] + ) + row["success_gain"] = row["candidate_success_rate"] - row["control_success_rate"] + if (row["candidate_success_rate"] + spec["gate"]["noninferiority_margin"] + < row["control_success_rate"]): + reasons.append(f"{split}_noninferiority_failed") + if row["success_gain"] < spec["gate"]["minimum_success_gain"]: + reasons.append(f"{split}_minimum_gain_failed") + total_candidate_escaped += row["candidate_escaped_defects"] + if total_candidate_escaped > spec["gate"]["max_candidate_escaped_defects"]: + reasons.append("candidate_escaped_defect_limit_exceeded") + + verdict = "promotion_proposal" if not reasons else "no_change" + evaluation = { + "schema_version": spec["schema_version"], + "experiment_id": spec["experiment_id"], + "spec_sha256": hashlib.sha256( + canonical_json(spec).encode(), + ).hexdigest(), + "evaluated_at": evaluated_at, + "verdict": verdict, + "active_policy_changed": False, + "reasons": sorted(set(reasons)), + "metrics": metrics, + "cases": rows, + "failures": failures, + "variable": spec["variable"], + "rollback_target": spec["rollback_target"], + "evidence_availability": spec["evidence_availability"], + } + if spec["schema_version"] == 2: + evaluation["evidence_sha256"] = hashlib.sha256(canonical_json(provenance).encode()).hexdigest() + return evaluation + + +def build_learning_proposal( + report: dict[str, Any], + experiment_record: dict[str, Any] | None, + *, + generated_at: str, +) -> dict[str, Any]: + """Distill saved evidence into a reviewable draft without changing policy.""" + _timestamp(generated_at, "generated_at") + required_report_fields = { + "schema_version", "project_id", "project_path", "generated_at", + "sample_size", "terminal_run_count", "final_successes", + "escaped_defects", "lead_repairs", "selection_modes", "profiles", + "missingness", "interpretation", + } + if not isinstance(report, dict) or set(report) != required_report_fields: + raise ContractError("learning proposal report is invalid") + if (report["schema_version"] != 1 + or not isinstance(report["project_path"], str) + or not isinstance(report["selection_modes"], dict) + or not isinstance(report["missingness"], dict)): + raise ContractError("learning proposal report values are invalid") + for field in ("sample_size", "terminal_run_count"): + if type(report[field]) is not int or report[field] < 0: + raise ContractError("learning proposal report counts are invalid") + mode_samples = {} + for mode in sorted(SELECTION_MODES): + row = report["selection_modes"].get(mode) + if (not isinstance(row, dict) or type(row.get("sample_size")) is not int + or row["sample_size"] < 0): + raise ContractError("learning proposal selection samples are invalid") + mode_samples[mode] = row["sample_size"] + + report_sha256 = hashlib.sha256(canonical_json(report).encode()).hexdigest() + experiment_id = None + question = None + hypothesis = None + variable = None + rollback_target = None + failures = [] + reasons = ["no_evaluated_experiment"] + verdict = "no_change" + evidence_availability = "unavailable" + experiment_samples = None + experiment_evidence = None + if experiment_record is not None: + if (not isinstance(experiment_record, dict) + or set(experiment_record) not in ( + EXPERIMENT_RECORD_FIELDS, EXPERIMENT_RECORD_FIELDS | {"eligibility"}, + )): + raise ContractError("learning proposal experiment record is invalid") + raw_spec = experiment_record["experiment"] + if isinstance(raw_spec, dict) and raw_spec.get("schema_version") == 1: + # Audit history as saved, including cases whose old labels reused + # outcomes. Strict new-spec validation is not historical decoding. + if set(raw_spec) != EXPERIMENT_FIELDS: + raise ContractError("historical experiment fields are invalid") + spec = json.loads(canonical_json(raw_spec)) + else: + spec = validate_experiment(raw_spec) + evaluation = experiment_record["evaluation"] + expected_fields = EXPERIMENT_EVALUATION_FIELDS | ( + {"evidence_sha256"} if spec["schema_version"] == 2 else set() + ) + if (not isinstance(evaluation, dict) + or set(evaluation) != expected_fields + or type(evaluation.get("schema_version")) is not int + or evaluation["schema_version"] != spec["schema_version"]): + raise ContractError("learning proposal evaluation is invalid") + if spec["schema_version"] == 2: + from .experiment_provenance import require_sha256 + require_sha256(evaluation["evidence_sha256"], "evaluated evidence") + spec_sha256 = hashlib.sha256(canonical_json(spec).encode()).hexdigest() + evaluation_sha256 = hashlib.sha256( + canonical_json(evaluation).encode(), + ).hexdigest() + if (experiment_record["spec_sha256"] != spec_sha256 + or evaluation.get("spec_sha256") != spec_sha256 + or experiment_record["evaluation_sha256"] != evaluation_sha256): + raise ContractError("learning proposal evidence hash is invalid") + if (evaluation.get("experiment_id") != spec["experiment_id"] + or evaluation.get("verdict") not in { + "no_change", "promotion_proposal", + } + or evaluation.get("active_policy_changed") is not False + or evaluation.get("variable") != spec["variable"] + or evaluation.get("rollback_target") != spec["rollback_target"] + or not isinstance(evaluation.get("reasons"), list) + or not isinstance(evaluation.get("failures"), list) + or not isinstance(evaluation.get("metrics"), dict)): + raise ContractError("learning proposal evaluation values are invalid") + _timestamp(experiment_record["recorded_at"], "recorded_at") + metrics = evaluation["metrics"] + if set(metrics) != {"evaluation", "held_out"} or any( + not isinstance(metrics.get(split), dict) + or type(metrics[split].get("declared_pairs")) is not int + or type(metrics[split].get("available_pairs")) is not int + for split in ("evaluation", "held_out")): + raise ContractError("learning proposal experiment samples are invalid") + experiment_id = spec["experiment_id"] + question = spec["question"] + hypothesis = spec["hypothesis"] + variable = spec["variable"] + rollback_target = spec["rollback_target"] + failures = list(evaluation["failures"]) + reasons = list(evaluation["reasons"]) + verdict = evaluation["verdict"] + eligibility = experiment_record.get("eligibility") + if spec["schema_version"] == 1: + verdict = "no_change" + reasons = sorted(set([*reasons, "legacy_unverified_evidence"])) + elif eligibility is None: + verdict = "no_change" + reasons = sorted(set([*reasons, "current_evidence_not_checked"])) + else: + if (not isinstance(eligibility, dict) + or eligibility.get("experiment_id") != spec["experiment_id"] + or eligibility.get("spec_sha256") != spec_sha256 + or eligibility.get("evaluation_sha256") != evaluation_sha256 + or eligibility.get("saved_evidence_sha256") != evaluation["evidence_sha256"] + or type(eligibility.get("eligible")) is not bool + or not isinstance(eligibility.get("reasons"), list) + or (eligibility["eligible"] and ( + eligibility.get("current_evidence_sha256") != evaluation["evidence_sha256"] + or eligibility["reasons"]))): + raise ContractError("learning proposal current eligibility is invalid") + if not eligibility["eligible"]: + verdict = "no_change" + reasons = sorted(set([*reasons, *eligibility["reasons"]])) + evidence_availability = spec["evidence_availability"] + experiment_samples = { + split: { + "declared_pairs": metrics[split]["declared_pairs"], + "available_pairs": metrics[split]["available_pairs"], + } + for split in ("evaluation", "held_out") + } + experiment_evidence = { + "experiment_id": experiment_id, + "spec_sha256": spec_sha256, + "evaluation_sha256": evaluation_sha256, + "recorded_at": experiment_record["recorded_at"], + "eligibility": eligibility, + } + + identity = { + "project_path": report["project_path"], + "report_sha256": report_sha256, + "experiment": experiment_evidence, + "verdict": verdict, + } + proposal_id = "proposal-" + hashlib.sha256( + canonical_json(identity).encode(), + ).hexdigest()[:24] + return { + "schema_version": 1, + "proposal_id": proposal_id, + "project_id": report["project_id"], + "project_path": report["project_path"], + "generated_at": generated_at, + "verdict": verdict, + "active_policy_changed": False, + "question": question, + "hypothesis": hypothesis, + "variable": variable, + "rollback_target": rollback_target, + "sample_sizes": { + "terminal_runs": report["terminal_run_count"], + "final_outcomes": report["sample_size"], + "selection_modes": mode_samples, + "experiment": experiment_samples, + }, + "missingness": json.loads(canonical_json(report["missingness"])), + "reasons": reasons, + "failures": failures, + "evidence": { + "availability": evidence_availability, + "report_sha256": report_sha256, + "experiment": experiment_evidence, + }, + "decision": { + "action": ( + "review_policy_change" + if verdict == "promotion_proposal" + else "retain_current_policy" + ), + "review_required": verdict == "promotion_proposal", + }, + } + + +def render_learning_proposal_markdown(proposal: dict[str, Any]) -> str: + """Render a compact local review record for a validated proposal.""" + if not isinstance(proposal, dict) or proposal.get("schema_version") != 1: + raise ContractError("learning proposal is invalid") + lines = [ + f"# Learning proposal {proposal['proposal_id']}", + "", + f"- Verdict: `{proposal['verdict']}`", + f"- Project: `{proposal['project_path']}`", + f"- Generated: `{proposal['generated_at']}`", + "- Active policy changed: `false`", + f"- Next action: `{proposal['decision']['action']}`", + "", + "## Evidence", + "", + f"- Report SHA256: `{proposal['evidence']['report_sha256']}`", + f"- Final outcomes: {proposal['sample_sizes']['final_outcomes']}", + f"- Terminal runs: {proposal['sample_sizes']['terminal_runs']}", + ] + experiment = proposal["evidence"]["experiment"] + if experiment is None: + lines.append("- Experiment: none") + else: + lines.extend([ + f"- Experiment: `{experiment['experiment_id']}`", + f"- Evaluation SHA256: `{experiment['evaluation_sha256']}`", + f"- Rollback target: `{proposal['rollback_target']['profile_id']}` " + f"binding version {proposal['rollback_target']['binding_version']}", + ]) + lines.extend(["", "## Reasons", ""]) + lines.extend( + [f"- `{reason}`" for reason in proposal["reasons"]] + or ["- No gate failures were recorded."] + ) + lines.extend(["", "## Recorded failures", ""]) + lines.extend( + [f"- `{canonical_json(failure)}`" for failure in proposal["failures"]] + or ["- None."] + ) + return "\n".join(lines) + "\n" diff --git a/plugin/core/src/devsquad/lifecycle.py b/plugin/core/src/devsquad/lifecycle.py new file mode 100644 index 0000000..0050b55 --- /dev/null +++ b/plugin/core/src/devsquad/lifecycle.py @@ -0,0 +1,430 @@ +"""Strict M6 profile lifecycle contracts and qualification gates.""" + +from __future__ import annotations + +from datetime import datetime +import hashlib +import json +from typing import Any + +from .contracts import ContractError +from .store import canonical_json +from .validation import validate_profile + + +TEMPLATE_FIELDS = { + "schema_version", "template_id", "alias", "update_mode", "policy", + "allowed_harnesses", "allowed_model_families", "allowed_account_pools", + "allowed_task_classes", "permission_policy", "allowed_tools", + "allowed_billing_modes", "gate", +} +QUALIFICATION_FIELDS = { + "schema_version", "qualification_id", "alias", "template_id", + "candidate_profile", "task_class", "experiment_id", "evaluation_sha256", + "source", "budget", "measured", "verdict", "evidence_refs", +} +BINDING_CHANGE_FIELDS = { + "schema_version", "decision_id", "action", "alias", + "expected_binding_version", "qualification_id", "rollback_target", + "experiment_id", "evaluation_sha256", "actor", "reason", "evidence_refs", +} +CATALOG_FALLBACK_FIELDS = { + "schema_version", "decision_id", "action", "alias", + "expected_binding_version", "catalog_change", "actor", "reason", + "evidence_refs", +} +SHA256_LENGTH = 64 + + +def _exact(value: Any, fields: set[str], label: str) -> dict[str, Any]: + if not isinstance(value, dict) or set(value) != fields: + raise ContractError(f"{label} fields are invalid") + return value + + +def _identifier(value: Any, field: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ContractError(f"lifecycle {field} must be a non-empty string") + return value + + +def _strings(value: Any, field: str, *, required: bool = True) -> list[str]: + if (not isinstance(value, list) + or (required and not value) + or any(not isinstance(item, str) or not item for item in value) + or len(set(value)) != len(value)): + raise ContractError(f"lifecycle {field} must contain unique strings") + return list(value) + + +def _sha256(value: Any, field: str, *, nullable: bool = False) -> str | None: + if value is None and nullable: + return None + if (not isinstance(value, str) or len(value) != SHA256_LENGTH + or any(character not in "0123456789abcdef" for character in value)): + raise ContractError(f"lifecycle {field} must be a SHA256 digest") + return value + + +def _timestamp(value: Any, field: str) -> str: + if not isinstance(value, str) or not value: + raise ContractError(f"lifecycle {field} must be a timestamp") + try: + parsed = datetime.fromisoformat(value) + except ValueError as exc: + raise ContractError(f"lifecycle {field} must be an ISO timestamp") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ContractError(f"lifecycle {field} must include a timezone") + return value + + +def validate_profile_template(value: dict[str, Any]) -> dict[str, Any]: + """Validate one reviewed, versioned alias qualification policy.""" + _exact(value, TEMPLATE_FIELDS, "profile template") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("profile template schema_version is invalid") + template_id = _identifier(value["template_id"], "template_id") + alias = _identifier(value["alias"], "alias") + if value["update_mode"] not in {"reviewed", "guarded_auto"}: + raise ContractError("profile template update_mode is invalid") + policy = _exact(value["policy"], {"id", "version"}, "profile template policy") + _identifier(policy["id"], "policy.id") + if type(policy["version"]) is not int or policy["version"] < 1: + raise ContractError("profile template policy version is invalid") + permission = value["permission_policy"] + if permission not in {"read_only", "workspace_write"}: + raise ContractError("profile template permission policy is invalid") + billing = _strings(value["allowed_billing_modes"], "allowed_billing_modes") + if any(mode not in {"subscription", "paid_api"} for mode in billing): + raise ContractError("profile template billing mode is invalid") + gate = _exact(value["gate"], { + "min_evaluation_pairs", "min_held_out_pairs", "max_critical_defects", + "max_latency_ratio", "max_usage_ratio", + }, "profile template gate") + for field in ( + "min_evaluation_pairs", "min_held_out_pairs", "max_critical_defects", + ): + if type(gate[field]) is not int or gate[field] < (0 if field.startswith("max_") else 1): + raise ContractError(f"profile template gate {field} is invalid") + for field in ("max_latency_ratio", "max_usage_ratio"): + number = gate[field] + if (number is not None and (isinstance(number, bool) + or not isinstance(number, (int, float)) or number <= 0)): + raise ContractError(f"profile template gate {field} is invalid") + normalized = { + **value, + "template_id": template_id, + "alias": alias, + "policy": dict(policy), + "allowed_harnesses": _strings( + value["allowed_harnesses"], "allowed_harnesses", + ), + "allowed_model_families": _strings( + value["allowed_model_families"], "allowed_model_families", + ), + "allowed_account_pools": _strings( + value["allowed_account_pools"], "allowed_account_pools", + ), + "allowed_task_classes": _strings( + value["allowed_task_classes"], "allowed_task_classes", + ), + "allowed_tools": _strings( + value["allowed_tools"], "allowed_tools", required=False, + ), + "allowed_billing_modes": billing, + "gate": dict(gate), + } + return json.loads(canonical_json(normalized)) + + +def profile_fingerprint(profile: dict[str, Any]) -> str: + validate_profile(profile) + return hashlib.sha256(canonical_json(profile).encode()).hexdigest() + + +def profile_template_violation( + profile: dict[str, Any], template: dict[str, Any], +) -> str | None: + """Return the first authority-boundary violation, if any.""" + validate_profile(profile) + normalized = validate_profile_template(template) + checks = ( + (profile["harness"] in normalized["allowed_harnesses"], "harness_not_allowed"), + ( + profile["model_family"] in normalized["allowed_model_families"], + "model_family_not_allowed", + ), + ( + profile["account_pool_id"] in normalized["allowed_account_pools"], + "account_pool_not_allowed", + ), + ( + profile["permission_policy"] == normalized["permission_policy"], + "permission_change_not_allowed", + ), + ( + set(profile["required_tools"]) <= set(normalized["allowed_tools"]), + "tool_not_allowed", + ), + ( + profile["billing_mode"] in normalized["allowed_billing_modes"], + "billing_mode_not_allowed", + ), + ) + return next((reason for allowed, reason in checks if not allowed), None) + + +def guarded_change_violation( + incumbent: dict[str, Any], candidate: dict[str, Any], +) -> str | None: + """Prevent guarded automation from widening account/tool authority.""" + validate_profile(incumbent) + validate_profile(candidate) + if incumbent["permission_policy"] != candidate["permission_policy"]: + return "guarded_permission_change" + if incumbent["billing_mode"] != candidate["billing_mode"]: + return "guarded_billing_change" + if incumbent["account_pool_id"] != candidate["account_pool_id"]: + return "guarded_account_route_change" + if not set(candidate["required_tools"]) <= set(incumbent["required_tools"]): + return "guarded_tool_expansion" + return None + + +def validate_qualification(value: dict[str, Any]) -> dict[str, Any]: + """Validate a bounded candidate qualification record.""" + _exact(value, QUALIFICATION_FIELDS, "qualification") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("qualification schema_version is invalid") + qualification_id = _identifier(value["qualification_id"], "qualification_id") + alias = _identifier(value["alias"], "qualification.alias") + template_id = _identifier(value["template_id"], "qualification.template_id") + task_class = _identifier(value["task_class"], "qualification.task_class") + candidate = json.loads(canonical_json(value["candidate_profile"])) + validate_profile(candidate) + experiment_id = value["experiment_id"] + evaluation_sha256 = value["evaluation_sha256"] + if (experiment_id is None) != (evaluation_sha256 is None): + raise ContractError("qualification experiment evidence must be paired") + if experiment_id is not None: + _identifier(experiment_id, "qualification.experiment_id") + _sha256(evaluation_sha256, "qualification.evaluation_sha256") + source = _exact(value["source"], { + "harness_version", "catalog_sha256", "model_revision", + }, "qualification source") + if source["harness_version"] is not None: + _identifier(source["harness_version"], "source.harness_version") + if source["model_revision"] is not None: + _identifier(source["model_revision"], "source.model_revision") + _sha256(source["catalog_sha256"], "source.catalog_sha256", nullable=True) + budget = _exact(value["budget"], { + "max_cases", "used_cases", "max_worker_invocations", + "worker_invocations", "max_wall_seconds", "wall_seconds", + }, "qualification budget") + for field in budget: + if type(budget[field]) is not int or budget[field] < 0: + raise ContractError(f"qualification budget {field} is invalid") + if (budget["used_cases"] > budget["max_cases"] + or budget["worker_invocations"] > budget["max_worker_invocations"] + or budget["wall_seconds"] > budget["max_wall_seconds"]): + raise ContractError("qualification exceeded its frozen budget") + measured = _exact(value["measured"], { + "evaluation_pairs", "held_out_pairs", "critical_defects", + "latency_ratio", "usage_ratio", + }, "qualification measurements") + for field in ("evaluation_pairs", "held_out_pairs", "critical_defects"): + if type(measured[field]) is not int or measured[field] < 0: + raise ContractError(f"qualification measurement {field} is invalid") + for field in ("latency_ratio", "usage_ratio"): + number = measured[field] + if (number is not None and (isinstance(number, bool) + or not isinstance(number, (int, float)) or number < 0)): + raise ContractError(f"qualification measurement {field} is invalid") + if value["verdict"] not in {"qualified", "rejected", "incomplete"}: + raise ContractError("qualification verdict is invalid") + normalized = { + **value, + "qualification_id": qualification_id, + "alias": alias, + "template_id": template_id, + "task_class": task_class, + "candidate_profile": candidate, + "source": dict(source), + "budget": dict(budget), + "measured": dict(measured), + "evidence_refs": _strings( + value["evidence_refs"], "qualification.evidence_refs", + required=value["verdict"] == "qualified", + ), + } + return json.loads(canonical_json(normalized)) + + +def qualification_gate_failures( + qualification: dict[str, Any], template: dict[str, Any], +) -> list[str]: + """Evaluate the static preauthorized qualification gate.""" + record = validate_qualification(qualification) + policy = validate_profile_template(template) + reasons = [] + if record["alias"] != policy["alias"] or record["template_id"] != policy["template_id"]: + reasons.append("template_identity_mismatch") + if record["task_class"] not in policy["allowed_task_classes"]: + reasons.append("task_class_not_allowed") + violation = profile_template_violation(record["candidate_profile"], policy) + if violation: + reasons.append(violation) + if record["candidate_profile"]["quality_status"] not in {"trial", "proven"}: + reasons.append("candidate_quality_not_eligible") + if record["experiment_id"] is None: + reasons.append("experiment_evidence_missing") + measured = record["measured"] + gate = policy["gate"] + if measured["evaluation_pairs"] < gate["min_evaluation_pairs"]: + reasons.append("insufficient_evaluation_pairs") + if measured["held_out_pairs"] < gate["min_held_out_pairs"]: + reasons.append("insufficient_held_out_pairs") + if measured["critical_defects"] > gate["max_critical_defects"]: + reasons.append("critical_defect_limit_exceeded") + for measurement, limit in ( + ("latency_ratio", "max_latency_ratio"), + ("usage_ratio", "max_usage_ratio"), + ): + if gate[limit] is not None and measured[measurement] is None: + reasons.append(f"{measurement}_missing") + elif (gate[limit] is not None + and measured[measurement] > gate[limit]): + reasons.append(f"{measurement}_limit_exceeded") + return sorted(set(reasons)) + + +def validate_binding_change(value: dict[str, Any]) -> dict[str, Any]: + """Validate a replay-safe promotion or rollback request.""" + _exact(value, BINDING_CHANGE_FIELDS, "binding change") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("binding change schema_version is invalid") + decision_id = _identifier(value["decision_id"], "decision_id") + alias = _identifier(value["alias"], "binding change alias") + if value["action"] not in {"promote", "rollback"}: + raise ContractError("binding change action is invalid") + if (type(value["expected_binding_version"]) is not int + or value["expected_binding_version"] < 1): + raise ContractError("binding change expected version is invalid") + qualification_id = value["qualification_id"] + rollback_target = value["rollback_target"] + if value["action"] == "promote": + _identifier(qualification_id, "binding change qualification_id") + if (rollback_target is not None or value["experiment_id"] is not None + or value["evaluation_sha256"] is not None): + raise ContractError( + "promotion cannot specify rollback experiment evidence", + ) + else: + if qualification_id is not None: + raise ContractError("rollback cannot specify a qualification") + _exact(rollback_target, {"profile_id", "binding_version"}, "rollback target") + _identifier(rollback_target["profile_id"], "rollback profile_id") + if type(rollback_target["binding_version"]) is not int or rollback_target["binding_version"] < 1: + raise ContractError("rollback binding_version is invalid") + _identifier(value["experiment_id"], "rollback experiment_id") + _sha256(value["evaluation_sha256"], "rollback evaluation_sha256") + if value["actor"] not in {"human", "guarded_auto"}: + raise ContractError("binding change actor is invalid") + reason = _identifier(value["reason"], "binding change reason") + return json.loads(canonical_json({ + **value, + "decision_id": decision_id, + "alias": alias, + "reason": reason, + "evidence_refs": _strings( + value["evidence_refs"], "binding change evidence_refs", required=True, + ), + "rollback_target": ( + dict(rollback_target) if rollback_target is not None else None + ), + })) + + +def validate_catalog_fallback(value: dict[str, Any]) -> dict[str, Any]: + """Validate a CAS rollback driven by complete model-removal evidence.""" + from .catalog import validate_catalog_change + + _exact(value, CATALOG_FALLBACK_FIELDS, "catalog fallback") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("catalog fallback schema_version is invalid") + decision_id = _identifier(value["decision_id"], "decision_id") + alias = _identifier(value["alias"], "catalog fallback alias") + if value["action"] != "rollback": + raise ContractError("catalog fallback action must be rollback") + if (type(value["expected_binding_version"]) is not int + or value["expected_binding_version"] < 1): + raise ContractError("catalog fallback expected version is invalid") + if value["actor"] not in {"human", "guarded_auto"}: + raise ContractError("catalog fallback actor is invalid") + reason = _identifier(value["reason"], "catalog fallback reason") + catalog_change = validate_catalog_change(value["catalog_change"]) + if catalog_change["profile_scope"] != "provided": + raise ContractError("catalog fallback requires profile-scoped evidence") + if not catalog_change["removed_model_ids"]: + raise ContractError("catalog fallback requires a removed model") + return json.loads(canonical_json({ + **value, + "decision_id": decision_id, + "alias": alias, + "reason": reason, + "catalog_change": catalog_change, + "evidence_refs": _strings( + value["evidence_refs"], "catalog fallback evidence_refs", + required=True, + ), + })) + + +def validate_recorded_at(value: Any) -> str: + return _timestamp(value, "recorded_at") + + +def render_binding_decision_markdown(receipt: dict[str, Any]) -> str: + """Render an inspectable local promotion/rollback decision receipt.""" + required = { + "schema_version", "decision_id", "action", "alias", "actor", "reason", + "evidence_refs", "from", "to", "qualification_id", "qualification", + "policy", "rollback_target", "requested_rollback_target", "effective_at", + "rollback_evaluation", "affects_new_runs_only", + } + if not isinstance(receipt, dict) or set(receipt) != required: + raise ContractError("binding decision receipt is invalid") + if (receipt["schema_version"] != 1 + or receipt["action"] not in {"promote", "rollback"} + or receipt["actor"] not in {"human", "guarded_auto"} + or receipt["affects_new_runs_only"] is not True): + raise ContractError("binding decision receipt values are invalid") + _timestamp(receipt["effective_at"], "effective_at") + lines = [ + f"# Profile binding decision {receipt['decision_id']}", + "", + f"- Action: `{receipt['action']}`", + f"- Alias: `{receipt['alias']}`", + f"- Actor: `{receipt['actor']}`", + f"- Effective: `{receipt['effective_at']}`", + "- Scope: new runs only", + f"- Reason: {receipt['reason']}", + "", + "## Binding change", + "", + f"- From: `{receipt['from']['profile_id']}` at version " + f"{receipt['from']['binding_version']}", + f"- To: `{receipt['to']['profile_id']}` at version " + f"{receipt['to']['binding_version']}", + f"- Rollback: `{receipt['rollback_target']['profile_id']}` at version " + f"{receipt['rollback_target']['binding_version']}", + f"- Policy: `{receipt['policy']['id']}` version {receipt['policy']['version']}", + "", + "## Evidence", + "", + ] + lines.extend( + [f"- `{reference}`" for reference in receipt["evidence_refs"]] + or ["- None."] + ) + return "\n".join(lines) + "\n" diff --git a/plugin/core/src/devsquad/mcp_server.py b/plugin/core/src/devsquad/mcp_server.py new file mode 100644 index 0000000..4b138c6 --- /dev/null +++ b/plugin/core/src/devsquad/mcp_server.py @@ -0,0 +1,375 @@ +"""Optional local MCP transport for the durable DevSquad service. + +This module must remain importable without the MCP SDK. Ordinary CLI and +worker processes never pay for or depend on the optional transport package. +""" + +from __future__ import annotations + +import copy +import os +from pathlib import Path +from typing import Any, Callable, Mapping + +from . import __version__ +from .contracts import ContractError, PolicyDenied, envelope, error_payload +from .diagnostics import build_doctor_report +from .service import Service +from .store import ConflictError + +MCP_SDK_REQUIREMENT = "mcp==2.2.0" +MAX_ARTIFACT_PREVIEW_BYTES = 16 * 1024 + + +class MCPDependencyUnavailable(RuntimeError): + """Raised when the explicitly requested MCP transport is not installed.""" + + +def _server_type() -> Any: + try: + from mcp.server import MCPServer + except ModuleNotFoundError as exc: + if exc.name != "mcp": + raise + raise MCPDependencyUnavailable( + "DevSquad MCP support is not installed; install the optional " + f"dependency with: python3 -m pip install 'devsquad-core[mcp]' " + f"(requires {MCP_SDK_REQUIREMENT})" + ) from exc + return MCPServer + + +class MCPBridge: + """Strict MCP-facing application functions over one saved runtime.""" + + def __init__( + self, + runtime: Path, + service: Service | None = None, + *, + caller_surface: str | None = None, + caller_session_ref: str | None = None, + environment: Mapping[str, str] | None = None, + project: Path | None = None, + ): + for label, value in ( + ("caller_surface", caller_surface), + ("caller_session_ref", caller_session_ref), + ): + if value is not None and (not isinstance(value, str) or not value): + raise ContractError(f"{label} must be a non-empty string or null") + self.runtime = runtime.resolve() + self.service = service or Service(runtime) + self.caller_surface = caller_surface + self.caller_session_ref = caller_session_ref + self.environment = os.environ if environment is None else environment + self.project = (project or Path.cwd()).resolve() + + @staticmethod + def _response(operation: Callable[[], dict[str, Any]]) -> dict[str, Any]: + try: + return envelope(data=operation()) + except ContractError as exc: + return envelope(error=error_payload(exc.code, str(exc))) + except Exception as exc: + return envelope(error=error_payload("INTERNAL_ERROR", str(exc))) + + def _mutation_response( + self, operation: Callable[[], dict[str, Any]], + ) -> dict[str, Any]: + def guarded() -> dict[str, Any]: + worker = self.environment.get("DEVSQUAD_WORKER", "0") + depth = self.environment.get("DEVSQUAD_DELEGATION_DEPTH", "0") + if worker not in {"", "0"} or depth not in {"", "0"}: + raise PolicyDenied( + "DevSquad worker sessions cannot start or mutate team workflows" + ) + return operation() + + return self._response(guarded) + + def _task_with_bound_origin(self, task: dict[str, Any]) -> dict[str, Any]: + if self.caller_surface is None and self.caller_session_ref is None: + return task + bound = copy.deepcopy(task) + if not isinstance(bound, dict): + raise ContractError("task must be an object") + saved_origin = bound.get("origin", {}) + if not isinstance(saved_origin, dict): + raise ContractError("task origin must be an object") + origin = dict(saved_origin) + if self.caller_surface is not None: + origin["surface"] = self.caller_surface + if self.caller_session_ref is not None: + origin["session_ref"] = self.caller_session_ref + bound["origin"] = origin + return bound + + def _saved_read_response(self, operation: Callable[[], dict[str, Any]]) -> dict[str, Any]: + def guarded(): + if self.environment.get("DEVSQUAD_COUNCIL_ROLE", ""): + raise PolicyDenied("Council worker sessions cannot read saved team artifacts, events or handoff provenance") + return operation() + return self._response(guarded) + + def start( + self, + task: dict[str, Any], + idempotency_key: str, + supersedes_run_id: str | None = None, + ) -> dict[str, Any]: + """Validate and save a run, returning without observing its worker.""" + + return self._mutation_response( + lambda: self.service.start( + self._task_with_bound_origin(task), + idempotency_key, + supersedes_run_id, + ) + ) + + def doctor(self) -> dict[str, Any]: + """Inspect local provider and application readiness without mutation.""" + + return self._response(lambda: build_doctor_report(project=self.project)) + + def status(self, run_id: str) -> dict[str, Any]: + """Inspect the current projection for a saved run.""" + + return self._saved_read_response(lambda: self.service.status(run_id)) + + def events( + self, run_id: str, after: int = 0, limit: int = 100, + ) -> dict[str, Any]: + """Read one bounded event page using its durable integer cursor.""" + + return self._saved_read_response(lambda: self.service.events(run_id, after, limit)) + + def result( + self, run_id: str, preview_bytes: int = 4096, + ) -> dict[str, Any]: + """Return receipt references and bounded UTF-8 artifact previews.""" + + def operation() -> dict[str, Any]: + if (type(preview_bytes) is not int + or not 0 <= preview_bytes <= MAX_ARTIFACT_PREVIEW_BYTES): + raise ContractError( + "preview_bytes must be between 0 and " + f"{MAX_ARTIFACT_PREVIEW_BYTES}" + ) + result = self.service.result(run_id) + if not result.get("ready") or preview_bytes == 0: + return result + remaining = preview_bytes + artifact_root = (self.runtime / "artifacts").resolve(strict=True) + artifacts = [] + for saved in result.get("artifacts", []): + artifact = dict(saved) + artifact["preview_text"] = None + artifact["preview_truncated"] = artifact.get("byte_size", 0) > 0 + if remaining > 0: + path = Path(artifact["path"]).resolve(strict=True) + try: + path.relative_to(artifact_root) + except ValueError as exc: + raise ConflictError( + "result artifact path escapes the saved runtime" + ) from exc + with path.open("rb") as stream: + raw = stream.read(remaining + 1) + consumed = min(len(raw), remaining) + try: + artifact["preview_text"] = raw[:consumed].decode("utf-8") + except UnicodeDecodeError: + artifact["preview_text"] = None + artifact["preview_truncated"] = len(raw) > consumed + remaining -= consumed + artifacts.append(artifact) + return {**result, "artifacts": artifacts, "preview_bytes": preview_bytes} + + return self._saved_read_response(operation) + + def cancel(self, run_id: str) -> dict[str, Any]: + """Persist cancellation intent without observing worker completion.""" + + return self._mutation_response(lambda: self.service.cancel(run_id)) + + def resume( + self, run_id: str, recovery: dict[str, Any] | None = None, + ) -> dict[str, Any]: + """Reconcile and safely resume a saved run.""" + + return self._mutation_response(lambda: self.service.resume(run_id, recovery)) + + def handoff_claim( + self, + run_id: str, + expected_version: int, + owner: str, + prior_claim: dict[str, Any] | None = None, + ) -> dict[str, Any]: + """Claim or renew one fenced host handoff.""" + + return self._mutation_response( + lambda: self.service.handoff_claim( + run_id, expected_version, owner, prior_claim, + ) + ) + + def handoff_complete( + self, + run_id: str, + claim: dict[str, Any], + decision: dict[str, Any], + ) -> dict[str, Any]: + """Submit a decision against a current fenced host claim.""" + + return self._mutation_response( + lambda: self.service.handoff_complete(run_id, claim, decision) + ) + + +def build_server( + runtime: Path, + service: Service | None = None, + *, + caller_surface: str | None = None, + caller_session_ref: str | None = None, + environment: Mapping[str, str] | None = None, + project: Path | None = None, +) -> Any: + """Build the local stdio server without starting it.""" + + server_type = _server_type() + from mcp.types import ToolAnnotations + + read_only = ToolAnnotations( + readOnlyHint=True, + destructiveHint=False, + idempotentHint=True, + openWorldHint=False, + ) + durable_write = ToolAnnotations( + readOnlyHint=False, + destructiveHint=False, + idempotentHint=True, + openWorldHint=False, + ) + state_change = ToolAnnotations( + readOnlyHint=False, + destructiveHint=False, + idempotentHint=False, + openWorldHint=False, + ) + destructive_change = ToolAnnotations( + readOnlyHint=False, + destructiveHint=True, + idempotentHint=True, + openWorldHint=False, + ) + bridge = MCPBridge( + runtime, + service, + caller_surface=caller_surface, + caller_session_ref=caller_session_ref, + environment=environment, + project=project, + ) + server = server_type( + "DevSquad", + version=__version__, + instructions=( + "Operate on durable local DevSquad runs. Submit a task once with " + "an idempotency key, then inspect status/events/result by run ID." + ), + ) + + @server.tool(name="squad_doctor", annotations=read_only) + def squad_doctor() -> dict[str, Any]: + """Inspect versions, capabilities and local host registration drift.""" + + return bridge.doctor() + + @server.tool(name="squad_start", annotations=durable_write) + def squad_start( + task: dict[str, Any], + idempotency_key: str, + supersedes_run_id: str | None = None, + ) -> dict[str, Any]: + """Validate, snapshot and persist one durable DevSquad run.""" + + return bridge.start(task, idempotency_key, supersedes_run_id) + + @server.tool(name="squad_status", annotations=read_only) + def squad_status(run_id: str) -> dict[str, Any]: + """Inspect state, version, active work and the next action.""" + + return bridge.status(run_id) + + @server.tool(name="squad_events", annotations=read_only) + def squad_events( + run_id: str, after: int = 0, limit: int = 100, + ) -> dict[str, Any]: + """Read a bounded durable event page after an integer cursor.""" + + return bridge.events(run_id, after, limit) + + @server.tool(name="squad_result", annotations=read_only) + def squad_result( + run_id: str, preview_bytes: int = 4096, + ) -> dict[str, Any]: + """Read result references and capped local artifact previews.""" + + return bridge.result(run_id, preview_bytes) + + @server.tool(name="squad_cancel", annotations=destructive_change) + def squad_cancel(run_id: str) -> dict[str, Any]: + """Persist cancellation intent for a saved run.""" + + return bridge.cancel(run_id) + + @server.tool(name="squad_resume", annotations=state_change) + def squad_resume( + run_id: str, recovery: dict[str, Any] | None = None, + ) -> dict[str, Any]: + """Reconcile ownership and safely resume a saved run.""" + + return bridge.resume(run_id, recovery) + + @server.tool(name="squad_handoff_claim", annotations=state_change) + def squad_handoff_claim( + run_id: str, + expected_version: int, + owner: str, + prior_claim: dict[str, Any] | None = None, + ) -> dict[str, Any]: + """Obtain or renew a fenced host claim and its input packet.""" + + return bridge.handoff_claim(run_id, expected_version, owner, prior_claim) + + @server.tool(name="squad_handoff_complete", annotations=destructive_change) + def squad_handoff_complete( + run_id: str, + claim: dict[str, Any], + decision: dict[str, Any], + ) -> dict[str, Any]: + """Submit a host disposition against a current fenced claim.""" + + return bridge.handoff_complete(run_id, claim, decision) + + return server + + +def serve_stdio( + runtime: Path, + *, + caller_surface: str | None = None, + caller_session_ref: str | None = None, +) -> None: + """Run the local MCP server; stdout is owned exclusively by the SDK.""" + + build_server( + runtime, + caller_surface=caller_surface, + caller_session_ref=caller_session_ref, + ).run() diff --git a/plugin/core/src/devsquad/migrations/001_initial.sql b/plugin/core/src/devsquad/migrations/001_initial.sql new file mode 100644 index 0000000..0dead90 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/001_initial.sql @@ -0,0 +1,57 @@ +CREATE TABLE schema_migrations ( + version INTEGER PRIMARY KEY, + applied_at TEXT NOT NULL +); + +CREATE TABLE projects ( + id TEXT PRIMARY KEY, + git_common_dir TEXT NOT NULL UNIQUE, + created_at TEXT NOT NULL +); + +CREATE TABLE runs ( + id TEXT PRIMARY KEY, + project_id TEXT NOT NULL REFERENCES projects(id), + idempotency_key TEXT NOT NULL, + request_hash TEXT NOT NULL, + submitted_request TEXT NOT NULL, + mutable_snapshot TEXT, + state TEXT NOT NULL, + phase TEXT, + version INTEGER NOT NULL, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + UNIQUE(project_id, idempotency_key) +); + +CREATE TABLE claims ( + run_id TEXT PRIMARY KEY REFERENCES runs(id), + kind TEXT NOT NULL, + fencing_token INTEGER NOT NULL, + owner_id TEXT NOT NULL, + active INTEGER NOT NULL CHECK(active IN (0, 1)), + claimed_at TEXT NOT NULL +); + +CREATE TABLE events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + run_id TEXT NOT NULL REFERENCES runs(id), + run_version INTEGER NOT NULL, + type TEXT NOT NULL, + payload TEXT NOT NULL, + created_at TEXT NOT NULL, + UNIQUE(run_id, run_version) +); + +CREATE TABLE artifacts ( + id TEXT PRIMARY KEY, + run_id TEXT NOT NULL REFERENCES runs(id), + name TEXT NOT NULL, + path TEXT NOT NULL UNIQUE, + sha256 TEXT NOT NULL, + byte_size INTEGER NOT NULL, + created_at TEXT NOT NULL, + UNIQUE(run_id, name) +); + +CREATE INDEX events_run_cursor ON events(run_id, id); diff --git a/plugin/core/src/devsquad/migrations/002_supervisor.sql b/plugin/core/src/devsquad/migrations/002_supervisor.sql new file mode 100644 index 0000000..b9fe86f --- /dev/null +++ b/plugin/core/src/devsquad/migrations/002_supervisor.sql @@ -0,0 +1,35 @@ +ALTER TABLE runs ADD COLUMN worktree_path TEXT; + +CREATE TABLE supervisor_claims ( + run_id TEXT PRIMARY KEY REFERENCES runs(id), + owner_id TEXT NOT NULL, + fencing_token INTEGER NOT NULL, + package_digest TEXT NOT NULL, + heartbeat_at TEXT NOT NULL, + active INTEGER NOT NULL CHECK(active IN (0, 1)) +); + +CREATE TABLE attempts ( + id TEXT PRIMARY KEY, + run_id TEXT NOT NULL REFERENCES runs(id), + project_id TEXT NOT NULL REFERENCES projects(id), + worktree_path TEXT NOT NULL, + attempt_token TEXT NOT NULL UNIQUE, + status TEXT NOT NULL, + pid INTEGER, + pgid INTEGER, + process_start_id TEXT, + heartbeat_at TEXT NOT NULL, + package_digest TEXT NOT NULL, + stdout_artifact_id TEXT REFERENCES artifacts(id), + stderr_artifact_id TEXT REFERENCES artifacts(id), + output_metadata TEXT, + created_at TEXT NOT NULL, + finished_at TEXT +); + +CREATE UNIQUE INDEX one_active_writer_per_worktree +ON attempts(worktree_path) +WHERE status IN ('reserved', 'running', 'cancelling', 'ownership_ambiguous'); + +CREATE INDEX attempts_run ON attempts(run_id, created_at); diff --git a/plugin/core/src/devsquad/migrations/003_durable_io.sql b/plugin/core/src/devsquad/migrations/003_durable_io.sql new file mode 100644 index 0000000..0002d01 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/003_durable_io.sql @@ -0,0 +1,6 @@ +ALTER TABLE attempts ADD COLUMN stdout_spool TEXT; +ALTER TABLE attempts ADD COLUMN stderr_spool TEXT; +ALTER TABLE attempts ADD COLUMN stdout_meta TEXT; +ALTER TABLE attempts ADD COLUMN stderr_meta TEXT; +ALTER TABLE attempts ADD COLUMN exit_record TEXT; +ALTER TABLE attempts ADD COLUMN child_record TEXT; diff --git a/plugin/core/src/devsquad/migrations/004_run_snapshot.sql b/plugin/core/src/devsquad/migrations/004_run_snapshot.sql new file mode 100644 index 0000000..a607d75 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/004_run_snapshot.sql @@ -0,0 +1,3 @@ +ALTER TABLE runs ADD COLUMN package_path TEXT; +ALTER TABLE runs ADD COLUMN package_digest TEXT; +ALTER TABLE runs ADD COLUMN supersedes_run_id TEXT REFERENCES runs(id); diff --git a/plugin/core/src/devsquad/migrations/005_host_handoffs.sql b/plugin/core/src/devsquad/migrations/005_host_handoffs.sql new file mode 100644 index 0000000..9396afb --- /dev/null +++ b/plugin/core/src/devsquad/migrations/005_host_handoffs.sql @@ -0,0 +1,55 @@ +CREATE TABLE handoffs ( + id TEXT PRIMARY KEY, + run_id TEXT NOT NULL REFERENCES runs(id), + sequence INTEGER NOT NULL CHECK(sequence > 0), + packet_json TEXT NOT NULL, + packet_sha256 TEXT NOT NULL + CHECK(length(packet_sha256) = 64 AND packet_sha256 NOT GLOB '*[^0-9a-f]*'), + status TEXT NOT NULL CHECK(status IN ('open', 'submitted', 'consumed', 'cancelled')), + created_run_version INTEGER NOT NULL CHECK(created_run_version > 0), + submitted_run_version INTEGER, + created_at TEXT NOT NULL, + closed_at TEXT, + UNIQUE(run_id, sequence) +); + +CREATE UNIQUE INDEX one_pending_handoff_per_run +ON handoffs(run_id) +WHERE status IN ('open', 'submitted'); + +ALTER TABLE claims ADD COLUMN handoff_id TEXT REFERENCES handoffs(id); +ALTER TABLE claims ADD COLUMN lease_expires_at TEXT; +ALTER TABLE claims ADD COLUMN renewed_at TEXT; + +CREATE UNIQUE INDEX one_current_claim_per_handoff +ON claims(handoff_id) +WHERE handoff_id IS NOT NULL; + +CREATE TABLE handoff_submissions ( + id TEXT PRIMARY KEY, + handoff_id TEXT NOT NULL REFERENCES handoffs(id), + submission_id TEXT NOT NULL, + submission_hash TEXT NOT NULL + CHECK(length(submission_hash) = 64 AND submission_hash NOT GLOB '*[^0-9a-f]*'), + owner_id TEXT NOT NULL, + fencing_token INTEGER NOT NULL CHECK(fencing_token > 0), + disposition TEXT NOT NULL CHECK(disposition IN ('accept', 'revise', 'reject')), + decision_json TEXT NOT NULL, + evidence_refs_json TEXT NOT NULL, + outcome TEXT NOT NULL CHECK(outcome IN ('recorded', 'rejected')), + rejection_code TEXT, + recorded_run_version INTEGER NOT NULL CHECK(recorded_run_version > 0), + created_at TEXT NOT NULL, + CHECK( + (outcome = 'recorded' AND rejection_code IS NULL) + OR (outcome = 'rejected' AND rejection_code IS NOT NULL) + ), + UNIQUE(handoff_id, submission_id, submission_hash) +); + +CREATE UNIQUE INDEX one_recorded_submission_per_handoff +ON handoff_submissions(handoff_id) +WHERE outcome = 'recorded'; + +CREATE INDEX handoff_submission_id_lookup +ON handoff_submissions(handoff_id, submission_id); diff --git a/plugin/core/src/devsquad/migrations/006_attempt_roles.sql b/plugin/core/src/devsquad/migrations/006_attempt_roles.sql new file mode 100644 index 0000000..b286484 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/006_attempt_roles.sql @@ -0,0 +1 @@ +ALTER TABLE attempts ADD COLUMN role TEXT NOT NULL DEFAULT 'worker'; diff --git a/plugin/core/src/devsquad/migrations/007_attempt_account_pools.sql b/plugin/core/src/devsquad/migrations/007_attempt_account_pools.sql new file mode 100644 index 0000000..b740081 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/007_attempt_account_pools.sql @@ -0,0 +1,5 @@ +ALTER TABLE attempts ADD COLUMN account_pool_id TEXT; + +CREATE INDEX attempts_active_account_pool +ON attempts(account_pool_id, status) +WHERE account_pool_id IS NOT NULL; diff --git a/plugin/core/src/devsquad/migrations/008_attempt_routing.sql b/plugin/core/src/devsquad/migrations/008_attempt_routing.sql new file mode 100644 index 0000000..b1e332a --- /dev/null +++ b/plugin/core/src/devsquad/migrations/008_attempt_routing.sql @@ -0,0 +1,2 @@ +ALTER TABLE attempts ADD COLUMN profile_id TEXT; +ALTER TABLE attempts ADD COLUMN profile_index INTEGER; diff --git a/plugin/core/src/devsquad/migrations/009_capacity.sql b/plugin/core/src/devsquad/migrations/009_capacity.sql new file mode 100644 index 0000000..f5055db --- /dev/null +++ b/plugin/core/src/devsquad/migrations/009_capacity.sql @@ -0,0 +1,67 @@ +CREATE TABLE pool_observations ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + observation_id TEXT NOT NULL UNIQUE, + pool_id TEXT NOT NULL, + window_id TEXT NOT NULL, + applies_to_json TEXT NOT NULL, + observed_at TEXT NOT NULL, + expires_at TEXT NOT NULL, + source TEXT NOT NULL CHECK(source IN ('native_reported', 'manual_reported', 'estimated')), + used REAL, + limit_value REAL, + unit TEXT NOT NULL CHECK(unit IN ('percent', 'requests', 'tokens', 'provider-native-string')), + resets_at TEXT, + confidence TEXT NOT NULL CHECK(confidence IN ('confirmed', 'reported', 'estimated')), + recorded_at TEXT NOT NULL, + CHECK(used IS NULL OR used >= 0), + CHECK(limit_value IS NULL OR limit_value >= 0) +); + +CREATE INDEX pool_observations_lookup +ON pool_observations(pool_id, window_id, observed_at); + +CREATE TABLE pool_reservations ( + id TEXT PRIMARY KEY, + pool_id TEXT NOT NULL, + run_id TEXT NOT NULL REFERENCES runs(id), + attempt_id TEXT REFERENCES attempts(id), + purpose TEXT NOT NULL CHECK(purpose IN ('attempt', 'qualification', 'classifier')), + profile_id TEXT, + reserved_at TEXT NOT NULL, + reconciled_at TEXT, + reconcile_reason TEXT, + UNIQUE(attempt_id), + CHECK( + (reconciled_at IS NULL AND reconcile_reason IS NULL) + OR (reconciled_at IS NOT NULL AND reconcile_reason IS NOT NULL) + ) +); + +CREATE INDEX pool_reservations_active +ON pool_reservations(pool_id, reserved_at) +WHERE reconciled_at IS NULL; + +INSERT INTO pool_reservations( + id, pool_id, run_id, attempt_id, purpose, profile_id, reserved_at +) +SELECT + 'migrated-attempt-' || id, + account_pool_id, + run_id, + id, + 'attempt', + profile_id, + created_at +FROM attempts +WHERE account_pool_id IS NOT NULL + AND status IN ('reserved', 'running', 'cancelling', 'ownership_ambiguous'); + +CREATE TRIGGER reconcile_pool_reservation_after_attempt_status +AFTER UPDATE OF status ON attempts +WHEN NEW.status NOT IN ('reserved', 'running', 'cancelling', 'ownership_ambiguous') +BEGIN + UPDATE pool_reservations + SET reconciled_at = COALESCE(NEW.finished_at, NEW.heartbeat_at), + reconcile_reason = 'attempt_status_' || NEW.status + WHERE attempt_id = NEW.id AND reconciled_at IS NULL; +END; diff --git a/plugin/core/src/devsquad/migrations/010_outcomes.sql b/plugin/core/src/devsquad/migrations/010_outcomes.sql new file mode 100644 index 0000000..fe5bcf9 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/010_outcomes.sql @@ -0,0 +1,26 @@ +CREATE TABLE outcomes ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + outcome_id TEXT NOT NULL UNIQUE, + run_id TEXT NOT NULL REFERENCES runs(id), + kind TEXT NOT NULL CHECK(kind IN ('final', 'late_correction')), + verdict TEXT NOT NULL CHECK(verdict IN ('succeeded', 'failed', 'cancelled', 'escaped_defect', 'corrected')), + selection_mode TEXT NOT NULL CHECK(selection_mode IN ('automatic', 'pinned', 'experimental')), + observed_at TEXT NOT NULL, + corrects_outcome_id TEXT REFERENCES outcomes(outcome_id), + payload_json TEXT NOT NULL, + payload_sha256 TEXT NOT NULL + CHECK(length(payload_sha256) = 64 AND payload_sha256 NOT GLOB '*[^0-9a-f]*'), + recorded_at TEXT NOT NULL, + CHECK( + (kind = 'final' AND verdict IN ('succeeded', 'failed', 'cancelled') AND corrects_outcome_id IS NULL) + OR + (kind = 'late_correction' AND verdict IN ('escaped_defect', 'corrected') AND corrects_outcome_id IS NOT NULL) + ) +); + +CREATE UNIQUE INDEX one_final_outcome_per_run +ON outcomes(run_id) +WHERE kind = 'final'; + +CREATE INDEX outcomes_run_history +ON outcomes(run_id, observed_at, id); diff --git a/plugin/core/src/devsquad/migrations/011_experiments.sql b/plugin/core/src/devsquad/migrations/011_experiments.sql new file mode 100644 index 0000000..0c78530 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/011_experiments.sql @@ -0,0 +1,17 @@ +CREATE TABLE experiments ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_id TEXT NOT NULL UNIQUE, + project_id TEXT REFERENCES projects(id), + project_path TEXT NOT NULL, + spec_json TEXT NOT NULL, + spec_sha256 TEXT NOT NULL + CHECK(length(spec_sha256) = 64 AND spec_sha256 NOT GLOB '*[^0-9a-f]*'), + evaluation_json TEXT NOT NULL, + evaluation_sha256 TEXT NOT NULL + CHECK(length(evaluation_sha256) = 64 AND evaluation_sha256 NOT GLOB '*[^0-9a-f]*'), + verdict TEXT NOT NULL CHECK(verdict IN ('no_change', 'promotion_proposal')), + recorded_at TEXT NOT NULL +); + +CREATE INDEX experiments_project_history +ON experiments(project_id, recorded_at, id); diff --git a/plugin/core/src/devsquad/migrations/012_profile_lifecycle.sql b/plugin/core/src/devsquad/migrations/012_profile_lifecycle.sql new file mode 100644 index 0000000..83b7a44 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/012_profile_lifecycle.sql @@ -0,0 +1,87 @@ +CREATE TABLE profile_templates ( + template_id TEXT PRIMARY KEY, + alias TEXT NOT NULL, + update_mode TEXT NOT NULL CHECK(update_mode IN ('reviewed', 'guarded_auto')), + policy_id TEXT NOT NULL, + policy_version INTEGER NOT NULL CHECK(policy_version >= 1), + payload_json TEXT NOT NULL, + payload_sha256 TEXT NOT NULL + CHECK(length(payload_sha256) = 64 AND payload_sha256 NOT GLOB '*[^0-9a-f]*'), + recorded_at TEXT NOT NULL +); + +CREATE INDEX profile_templates_alias_policy +ON profile_templates(alias, policy_id, policy_version, recorded_at); + +CREATE TABLE concrete_profiles ( + profile_id TEXT PRIMARY KEY, + profile_json TEXT NOT NULL, + profile_sha256 TEXT NOT NULL + CHECK(length(profile_sha256) = 64 AND profile_sha256 NOT GLOB '*[^0-9a-f]*'), + recorded_at TEXT NOT NULL +); + +CREATE TABLE qualification_runs ( + qualification_id TEXT PRIMARY KEY, + alias TEXT NOT NULL, + template_id TEXT NOT NULL REFERENCES profile_templates(template_id), + profile_id TEXT NOT NULL REFERENCES concrete_profiles(profile_id), + experiment_id TEXT REFERENCES experiments(experiment_id), + evaluation_sha256 TEXT, + verdict TEXT NOT NULL CHECK(verdict IN ('qualified', 'rejected', 'incomplete')), + payload_json TEXT NOT NULL, + payload_sha256 TEXT NOT NULL + CHECK(length(payload_sha256) = 64 AND payload_sha256 NOT GLOB '*[^0-9a-f]*'), + gate_failures_json TEXT NOT NULL, + recorded_at TEXT NOT NULL, + CHECK( + (experiment_id IS NULL AND evaluation_sha256 IS NULL) + OR + (experiment_id IS NOT NULL AND length(evaluation_sha256) = 64 + AND evaluation_sha256 NOT GLOB '*[^0-9a-f]*') + ) +); + +CREATE INDEX qualification_runs_alias_history +ON qualification_runs(alias, recorded_at, qualification_id); + +CREATE TABLE profile_bindings ( + alias TEXT PRIMARY KEY, + template_id TEXT NOT NULL REFERENCES profile_templates(template_id), + profile_id TEXT NOT NULL REFERENCES concrete_profiles(profile_id), + qualification_id TEXT REFERENCES qualification_runs(qualification_id), + version INTEGER NOT NULL CHECK(version >= 1), + updated_at TEXT NOT NULL +); + +CREATE TABLE profile_binding_versions ( + alias TEXT NOT NULL, + version INTEGER NOT NULL CHECK(version >= 1), + template_id TEXT NOT NULL REFERENCES profile_templates(template_id), + profile_id TEXT NOT NULL REFERENCES concrete_profiles(profile_id), + qualification_id TEXT REFERENCES qualification_runs(qualification_id), + decision_id TEXT, + recorded_at TEXT NOT NULL, + PRIMARY KEY(alias, version) +); + +CREATE TABLE binding_decisions ( + decision_id TEXT PRIMARY KEY, + alias TEXT NOT NULL, + action TEXT NOT NULL CHECK(action IN ('promote', 'rollback')), + actor TEXT NOT NULL CHECK(actor IN ('human', 'guarded_auto')), + from_version INTEGER NOT NULL CHECK(from_version >= 1), + to_version INTEGER NOT NULL CHECK(to_version > from_version), + from_profile_id TEXT NOT NULL REFERENCES concrete_profiles(profile_id), + to_profile_id TEXT NOT NULL REFERENCES concrete_profiles(profile_id), + qualification_id TEXT REFERENCES qualification_runs(qualification_id), + request_sha256 TEXT NOT NULL + CHECK(length(request_sha256) = 64 AND request_sha256 NOT GLOB '*[^0-9a-f]*'), + receipt_json TEXT NOT NULL, + receipt_sha256 TEXT NOT NULL + CHECK(length(receipt_sha256) = 64 AND receipt_sha256 NOT GLOB '*[^0-9a-f]*'), + recorded_at TEXT NOT NULL +); + +CREATE INDEX binding_decisions_alias_history +ON binding_decisions(alias, to_version, decision_id); diff --git a/plugin/core/src/devsquad/migrations/013_decision_observations.sql b/plugin/core/src/devsquad/migrations/013_decision_observations.sql new file mode 100644 index 0000000..b494873 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/013_decision_observations.sql @@ -0,0 +1,42 @@ +CREATE TABLE decision_cache ( + cache_key TEXT PRIMARY KEY + CHECK(length(cache_key) = 64 AND cache_key NOT GLOB '*[^0-9a-f]*'), + request_json TEXT NOT NULL, + status TEXT NOT NULL CHECK(status IN ( + 'reserved', 'running', 'succeeded', 'abstained', 'invalid', + 'unavailable', 'indeterminate', 'cancelled' + )), + response_json TEXT, + response_sha256 TEXT + CHECK(response_sha256 IS NULL OR ( + length(response_sha256) = 64 + AND response_sha256 NOT GLOB '*[^0-9a-f]*' + )), + billable_calls INTEGER NOT NULL DEFAULT 0 CHECK(billable_calls IN (0, 1)), + usage_json TEXT, + owner_id TEXT NOT NULL, + error TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + CHECK( + (status IN ('reserved', 'running') AND response_json IS NULL) + OR status NOT IN ('reserved', 'running') + ) +); + +CREATE TABLE run_decision_observations ( + run_id TEXT NOT NULL REFERENCES runs(id), + purpose_id TEXT NOT NULL, + cache_key TEXT NOT NULL REFERENCES decision_cache(cache_key), + mode TEXT NOT NULL CHECK(mode IN ('shadow', 'advisory')), + applied INTEGER NOT NULL DEFAULT 0 CHECK(applied IN (0, 1)), + effect_json TEXT, + recorded_at TEXT NOT NULL, + PRIMARY KEY(run_id, purpose_id) +); + +CREATE INDEX decision_cache_status +ON decision_cache(status, updated_at, cache_key); + +CREATE INDEX run_decision_observations_cache +ON run_decision_observations(cache_key, run_id); diff --git a/plugin/core/src/devsquad/migrations/014_experiment_assignments.sql b/plugin/core/src/devsquad/migrations/014_experiment_assignments.sql new file mode 100644 index 0000000..2dbf0f8 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/014_experiment_assignments.sql @@ -0,0 +1,29 @@ +-- Predeclared experiments and arms are frozen by the run preparation fence, +-- before attempt reservation. Existing evaluations remain immutable history. +CREATE TABLE experiment_specs ( + experiment_id TEXT PRIMARY KEY, + project_id TEXT NOT NULL REFERENCES projects(id), + spec_json TEXT NOT NULL, + spec_sha256 TEXT NOT NULL + CHECK(length(spec_sha256) = 64 AND spec_sha256 NOT GLOB '*[^0-9a-f]*'), + recorded_at TEXT NOT NULL +); + +CREATE TABLE experiment_assignments ( + run_id TEXT PRIMARY KEY REFERENCES runs(id), + experiment_id TEXT NOT NULL REFERENCES experiment_specs(experiment_id), + case_id TEXT NOT NULL, + arm TEXT NOT NULL CHECK(arm IN ('control', 'candidate')), + outcome_id TEXT NOT NULL, + assignment_json TEXT NOT NULL, + assignment_sha256 TEXT NOT NULL + CHECK(length(assignment_sha256) = 64 AND assignment_sha256 NOT GLOB '*[^0-9a-f]*'), + snapshot_json TEXT NOT NULL, + package_digest TEXT NOT NULL + CHECK(length(package_digest) = 64 AND package_digest NOT GLOB '*[^0-9a-f]*'), + frozen_run_version INTEGER NOT NULL, + preparation_fencing_token INTEGER NOT NULL, + recorded_at TEXT NOT NULL, + UNIQUE(experiment_id, case_id, arm), + UNIQUE(experiment_id, outcome_id) +); diff --git a/plugin/core/src/devsquad/migrations/015_experiment_evaluation_revisions.sql b/plugin/core/src/devsquad/migrations/015_experiment_evaluation_revisions.sql new file mode 100644 index 0000000..d0785ea --- /dev/null +++ b/plugin/core/src/devsquad/migrations/015_experiment_evaluation_revisions.sql @@ -0,0 +1,18 @@ +-- Never overwrite the original evaluation. Explicit reviews pin a predecessor +-- and reuse the original frozen spec/assignments, not new independent samples. +CREATE TABLE experiment_evaluation_revisions ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + revision_id TEXT NOT NULL UNIQUE, + experiment_id TEXT NOT NULL REFERENCES experiments(experiment_id), + previous_evaluation_sha256 TEXT NOT NULL + CHECK(length(previous_evaluation_sha256) = 64 AND previous_evaluation_sha256 NOT GLOB '*[^0-9a-f]*'), + evaluation_json TEXT NOT NULL, + evaluation_sha256 TEXT NOT NULL + CHECK(length(evaluation_sha256) = 64 AND evaluation_sha256 NOT GLOB '*[^0-9a-f]*'), + verdict TEXT NOT NULL CHECK(verdict IN ('no_change', 'promotion_proposal')), + recorded_at TEXT NOT NULL, + UNIQUE(experiment_id, evaluation_sha256) +); + +CREATE INDEX experiment_evaluation_review_history +ON experiment_evaluation_revisions(experiment_id, id); diff --git a/plugin/core/src/devsquad/migrations/016_objective_outcome_jobs.sql b/plugin/core/src/devsquad/migrations/016_objective_outcome_jobs.sql new file mode 100644 index 0000000..7cfcc07 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/016_objective_outcome_jobs.sql @@ -0,0 +1,8 @@ +-- New public runs request objective projection at admission. Legacy final +-- outcomes and missing legacy history are not rewritten or retroactively armed. +CREATE TABLE objective_outcome_jobs ( + run_id TEXT PRIMARY KEY REFERENCES runs(id), + requested_at TEXT NOT NULL, + completed_outcome_id TEXT REFERENCES outcomes(outcome_id) +); +CREATE INDEX pending_objective_outcomes ON objective_outcome_jobs(completed_outcome_id, run_id); diff --git a/plugin/core/src/devsquad/migrations/017_council_contract_epoch.sql b/plugin/core/src/devsquad/migrations/017_council_contract_epoch.sql new file mode 100644 index 0000000..5f412c4 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/017_council_contract_epoch.sql @@ -0,0 +1,5 @@ +-- Contract epoch only: Council's read/claim/decision authority is not safe for +-- schema-16 packages sharing this ledger. Existing exclusive migration, +-- active/recoverable deferral and per-connection write triggers fence them. +-- No table, scheduling service or historical outcome representation changes. +SELECT 1; diff --git a/plugin/core/src/devsquad/migrations/__init__.py b/plugin/core/src/devsquad/migrations/__init__.py new file mode 100644 index 0000000..690dbef --- /dev/null +++ b/plugin/core/src/devsquad/migrations/__init__.py @@ -0,0 +1 @@ +"""Packaged SQLite migrations for the local DevSquad ledger.""" diff --git a/plugin/core/src/devsquad/native_catalog.py b/plugin/core/src/devsquad/native_catalog.py new file mode 100644 index 0000000..109157b --- /dev/null +++ b/plugin/core/src/devsquad/native_catalog.py @@ -0,0 +1,175 @@ +"""Private, scoped last-good discovery and native subscription observations. + +No inference, login, billing, credit-reset, or configuration mutation lives here. +The short-lived OS lock is the refresh lease: process death releases ownership. +""" +from __future__ import annotations + +from datetime import datetime, timedelta, timezone +import fcntl +import hashlib +import json +import math +import os +from pathlib import Path +import uuid +from typing import Any, Callable + +from .catalog import analyze_catalog_drift, normalize_models +from .contracts import ContractError +from .store import canonical_json + +CATALOG_TTL = timedelta(hours=24) +REFRESH_BACKOFF = timedelta(minutes=2) +QUOTA_TTL = timedelta(seconds=60) + + +def _account_identity(account_result: dict[str, Any]) -> dict[str, Any]: + account = account_result.get("account") + if not isinstance(account, dict) or account.get("type") != "chatgpt": + raise ContractError("normal entry requires native ChatGPT subscription authentication") + identity = {key: account.get(key) for key in ("type", "id", "accountId", "email")} + if not any(isinstance(identity[key], str) and identity[key] for key in ("id", "accountId", "email")): + raise ContractError("native account identity is unknown; discovery cannot be reused") + return identity + + +def native_account_pool(account_result: dict[str, Any]) -> str: + """One reservation fence for one subscription, across discovery contexts.""" + return "codex-subscription-" + hashlib.sha256(canonical_json(_account_identity(account_result)).encode()).hexdigest() + + +def native_scope(account_result: dict[str, Any], config: dict[str, Any], binary: str, version: str) -> str: + identity = _account_identity(account_result) + # Hash in memory only. Neither account identifiers nor effective configuration + # (which can contain sensitive provider fields) are persisted or displayed. + return hashlib.sha256(canonical_json({ + "account": identity, "plan_type": account_result["account"].get("planType"), + "config": config, "binary": binary, "version": version, + }).encode()).hexdigest() + + +def normalize_codex_limits(result: dict[str, Any], pool_id: str, *, now: datetime | None = None) -> list[dict[str, Any]]: + current = now or datetime.now(timezone.utc) + buckets = result.get("rateLimitsByLimitId") + bucket = buckets.get("codex") if isinstance(buckets, dict) else result.get("rateLimits") + bucket = bucket if isinstance(bucket, dict) else {} + observations = [] + for slot in ("primary", "secondary"): + window = bucket.get(slot) + window = window if isinstance(window, dict) else {} + used, duration, reset = (window.get(key) for key in ("usedPercent", "windowDurationMins", "resetsAt")) + valid = (type(used) in (int, float) and math.isfinite(used) and 0 <= used <= 100 + and type(duration) is int and duration > 0 and type(reset) is int) + resets_at = None + if valid: + try: + resets_at = datetime.fromtimestamp(reset, timezone.utc) + valid = resets_at > current + except (ValueError, OverflowError, OSError): + valid = False + expires = min(current + QUOTA_TTL, resets_at) if valid else current + QUOTA_TTL + evidence = { + "schema_version": 1, "pool_id": pool_id, "window_id": f"codex:{slot}", + "applies_to": {"harnesses": ["codex"], "model_families": [], "model_ids": []}, + "observed_at": current.isoformat(), "expires_at": expires.isoformat(), + "source": "native_reported", "used": used if valid else None, + "limit": 100 if valid else None, "unit": "percent", + "resets_at": resets_at.isoformat() if valid else None, + "confidence": "reported", + } + evidence["observation_id"] = "native-" + hashlib.sha256(canonical_json(evidence).encode()).hexdigest() + observations.append(evidence) + return observations + + +class NativeCatalogCache: + def __init__(self, directory: Path, scope: str, version: str): + self.directory, self.scope, self.version = directory, scope, version + self.path = directory / (hashlib.sha256(scope.encode()).hexdigest() + ".json") + self.failure_path = self.path.with_suffix(".failure.json") + + def _read(self) -> dict[str, Any] | None: + if not self.path.exists(): + return None + try: + value = json.loads(self.path.read_bytes()) + if (value["scope"] != self.scope or value["harness_version"] != self.version + or value["complete"] is not True or not isinstance(value["models"], list)): + raise ValueError() + return value + except (ValueError, KeyError, TypeError) as exc: + raise ContractError("native catalog cache integrity is invalid") from exc + + @staticmethod + def _recent(timestamp: str, current: datetime, ttl: timedelta) -> bool: + try: + at = datetime.fromisoformat(timestamp) + return at.tzinfo is not None and timedelta(0) <= current - at < ttl + except (TypeError, ValueError): + return False + + def _write(self, value: dict[str, Any], path: Path | None = None) -> None: + destination = path or self.path + temporary = destination.with_suffix(f".tmp-{uuid.uuid4().hex}") + try: + with temporary.open("x", encoding="utf-8") as output: + os.chmod(temporary, 0o600) + output.write(canonical_json(value) + "\n") + output.flush() + os.fsync(output.fileno()) + os.replace(temporary, destination) + finally: + temporary.unlink(missing_ok=True) + + def refresh(self, fetch: Callable[[], list[dict[str, Any]]], *, now: datetime | None = None) -> dict[str, Any]: + current = now or datetime.now(timezone.utc) + self.directory.mkdir(parents=True, exist_ok=True) + with self.path.with_suffix(".lock").open("a+") as lease: + os.chmod(lease.name, 0o600) + try: + fcntl.flock(lease, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError: + previous = self._read() + if previous is None: + raise ContractError("native catalog refresh is in progress; retry shortly") + return previous + try: + previous = self._read() + if previous is None and self.failure_path.exists(): + try: + failure = json.loads(self.failure_path.read_bytes()) + backing_off = self._recent(failure["at"], current, REFRESH_BACKOFF) + except (ValueError, KeyError, TypeError): + raise ContractError("native discovery backoff evidence is invalid") from None + if backing_off: + raise ContractError("native discovery is backing off; retry shortly") + if previous is not None and ( + self._recent(previous["fetched_at"], current, CATALOG_TTL) + or self._recent(previous["last_refresh"]["at"], current, REFRESH_BACKOFF) + ): + return previous + try: + raw_models = fetch() + if not isinstance(raw_models, list): + raise ContractError("native catalog response is incomplete") + models = normalize_models("codex", self.version, raw_models) + except (ContractError, EOFError, OSError, TimeoutError): + if previous is None: + self._write({"at": current.isoformat(), "status": "error"}, self.failure_path) + raise ContractError("native discovery failed with no scoped last-good catalog") from None + previous["last_refresh"] = {"at": current.isoformat(), "status": "error", "error": "discovery_failed"} + self._write(previous) + return previous + value = { + "schema_version": 1, "scope": self.scope, "harness": "codex", + "harness_version": self.version, "fetched_at": current.isoformat(), + "complete": True, "models": models, + "last_refresh": {"at": current.isoformat(), "status": "ok", "error": None}, + } + value["catalog_change"] = analyze_catalog_drift(previous, value) + self._write(value) + self.failure_path.unlink(missing_ok=True) + return value + finally: + fcntl.flock(lease, fcntl.LOCK_UN) diff --git a/plugin/core/src/devsquad/objective_outcomes.py b/plugin/core/src/devsquad/objective_outcomes.py new file mode 100644 index 0000000..c315730 --- /dev/null +++ b/plugin/core/src/devsquad/objective_outcomes.py @@ -0,0 +1,102 @@ +"""Objective projection only; subjective later corrections stay append-only.""" +from __future__ import annotations + +import json +from datetime import datetime, timezone +from typing import Any + +from .contracts import ContractError + + +def project_outcome(run: dict[str, Any], receipt: dict[str, Any], attempts: list[dict[str, Any]], + artifact_refs: dict[str, str], assignment: dict[str, Any] | None) -> dict[str, Any]: + snapshot = json.loads(run["mutable_snapshot"] or "null") + if ("internal_fake_delay" in (snapshot or {}) and set(receipt) - {"error"} == { + "returncode", "cancelled", "timed_out", "stdout", "stderr", "finished_at"}): + # The existing generic runner fixture retains its native exit receipt, + # unlike the managed workflow's semantic receipt. Do not rewrite it. + if (type(receipt["returncode"]) is not int or type(receipt["cancelled"]) is not bool + or type(receipt["timed_out"]) is not bool + or type(receipt["finished_at"]) not in {int, float} + or "error" in receipt and (receipt["error"] != "TIMEOUT" or not receipt["timed_out"])): + raise ContractError("objective generic exit receipt is invalid") + expected = "cancelled" if receipt["cancelled"] else "failed" if receipt["timed_out"] or receipt["returncode"] != 0 else "succeeded" + if run["state"] != expected and run["state"] != "cancelled": + raise ContractError("objective generic exit contradicts terminal state") + receipt = {**receipt, "run_id": run["id"], "state": run["state"], + "finished_at": datetime.fromtimestamp(receipt["finished_at"], timezone.utc).isoformat(), + "attempt_id": attempts[-1]["id"] if attempts else None, + "error": {"returncode": receipt["returncode"]} if expected == "failed" else None} + if receipt.get("run_id") != run["id"] or receipt.get("state") != run["state"]: + raise ContractError("objective receipt does not match the terminal run") + roles = (snapshot or {}).get("routing", {}).get("roles", {}) + mode = ("experimental" if assignment is not None or (snapshot or {}).get("experiment_assignment") is not None else "pinned" + if any(role.get("source") == "override" for role in roles.values()) else "automatic") + terminal_ref = artifact_refs.get("receipt.json", artifact_refs["result-receipt.json"]) + accepted = run["state"] == "succeeded" and receipt.get("lead", {}).get("disposition") == "accept" + criteria = [] + for criterion in receipt.get("criteria", []): + # The receipt's evaluation supplies evidence, not a lead verdict. A + # fenced final acceptance attests the criteria; cite that attestation + # rather than manufacturing a check-level result from evidence_available. + criteria.append({ + "criterion_id": criterion["id"], + "status": "passed" if accepted else "unknown", + "evidence_refs": [terminal_ref] if accepted else [], + }) + reported_attempts = {item["id"]: item for item in + receipt.get("attempts", []) + receipt.get("lead", {}).get("attempts", []) + if item.get("id") is not None} + failed_roles, contributions = set(), [] + for attempt in attempts: + # Reservation and prelaunch cancellation are not worker exposure. + if attempt["status"] != "finished" or attempt.get("pid") is None: + continue + metadata = json.loads(attempt["output_metadata"] or "null") + reported = reported_attempts.get(attempt["id"], {}) + failed = (isinstance(metadata, dict) and metadata.get("failure") is not None + or reported.get("status") == "failed" or reported.get("error") is not None + or receipt.get("attempt_id") == attempt["id"] and receipt.get("error") is not None) + role = attempt["role"] + repaired = role in failed_roles + if failed: + result = "failed" + failed_roles.add(role) + elif repaired: + result = "repair" + elif role == "reviewer" and reported.get("review", {}).get("verdict") == "findings": + result = "finding" + elif run["state"] == "succeeded" and criteria and all(c["status"] == "passed" for c in criteria): + result = "successful" + else: + result = "neutral" + refs = [value for name, value in artifact_refs.items() if attempt["id"] in name] + [terminal_ref] + contributions.append({"attempt_id": attempt["id"], "role": role, "result": result, + "independent_success": result == "successful", "evidence_refs": refs}) + imported_leads = [a["id"] for a in attempts if a["role"] == "lead" + and f"lead-attempt-{a['id']}.json" in artifact_refs] + lead_repairs = [] + for index, decision in enumerate(receipt.get("dispositions", [])): + if decision.get("disposition") != "revise": + continue + lead_id = imported_leads[index] if index < len(imported_leads) else None + refs = [terminal_ref] + if lead_id is not None: + refs.append(artifact_refs[f"lead-attempt-{lead_id}.json"]) + lead_repairs.append({"lead_attempt_id": lead_id, + "description": decision.get("reason") or "Explicit lead revision requested.", + "evidence_refs": refs}) + if lead_repairs: + # A successful final repair is not an independently successful original. + for contribution in contributions: + if contribution["result"] == "successful": + contribution.update(result="repair", independent_success=False) + lead = receipt.get("lead", {}).get("disposition") or "not_reached" + return { + "schema_version": 1, "outcome_id": assignment["outcome_id"] if assignment is not None else f"objective-final-{run['id']}", + "kind": "final", "verdict": run["state"], "selection_mode": mode, + "observed_at": receipt.get("completed_at") or receipt.get("finished_at") or run["updated_at"], + "corrects_outcome_id": None, "summary": f"Objective terminal state: {run['state']}; lead disposition: {lead}.", + "criteria": criteria, "contributions": contributions, "lead_repairs": lead_repairs, + "evidence_refs": list(artifact_refs.values()), + } diff --git a/plugin/core/src/devsquad/probe_process.py b/plugin/core/src/devsquad/probe_process.py new file mode 100644 index 0000000..f97ee3a --- /dev/null +++ b/plugin/core/src/devsquad/probe_process.py @@ -0,0 +1,270 @@ +"""Bounded ownership and cleanup for nongenerating native provider probes.""" + +from __future__ import annotations + +from contextlib import contextmanager +import getpass +import os +from pathlib import Path +import signal +import subprocess +import time +from typing import Any + +from .contracts import ContractError +from .supervisor import process_start_identity + +PROBE_CLEANUP_SECONDS = 2 +PROBE_TERM_GRACE_SECONDS = 0.25 +_POPEN_TYPE = subprocess.Popen + + +class _ProbeOwnershipUnavailable(ContractError): + pass + + +def subscription_environment(home: Path | None = None) -> dict[str, str]: + """Use saved subscription login without ambient keys/provider overrides.""" + return { + "HOME": str(home if home is not None else Path.home()), + "USER": os.environ.get("USER") or getpass.getuser(), + "PATH": os.environ.get("PATH", ""), + } + + +def capture_probe_identity(process: subprocess.Popen[Any]) -> str | None: + """Capture immediately after spawning with start_new_session=True.""" + return process_start_identity(process.pid) + + +def _probe_group_exists(pgid: int, *, deadline: float) -> bool: + """The supervisor's live-member rule with a diagnostic time bound.""" + remaining = deadline - time.monotonic() + if remaining <= 0: + raise ContractError("diagnostic cleanup deadline expired") + try: + inventory = subprocess.run( + ["/bin/ps", "-axo", "pgid=,stat="], text=True, + stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, + env=subscription_environment(), timeout=min(0.5, remaining), check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise ContractError("diagnostic process inventory unavailable") from exc + if inventory.returncode != 0 or len(inventory.stdout) > 1024 * 1024: + raise ContractError("diagnostic process inventory unavailable") + parsed = 0 + for line in inventory.stdout.splitlines(): + fields = line.split() + if len(fields) != 2 or not fields[0].isdigit(): + raise ContractError("diagnostic process inventory malformed") + parsed += 1 + if int(fields[0]) == pgid and not fields[1].startswith("Z"): + return True + if parsed == 0: + raise ContractError("diagnostic process inventory empty") + return False + + +@contextmanager +def _reap_guard(process: subprocess.Popen[Any], *, deadline: float, locked: bool = False): + if locked or not isinstance(process, _POPEN_TYPE): + yield + return + remaining = deadline - time.monotonic() + if remaining <= 0 or not process._waitpid_lock.acquire(timeout=remaining): + raise ContractError("diagnostic cleanup deadline expired") + try: + yield + finally: + process._waitpid_lock.release() + + +def _retained_child_anchor( + process: subprocess.Popen[Any], *, deadline: float, locked: bool = False, +) -> bool: + """Verify an unreaped child without releasing its PID for reuse. + + Callers exclusively reap this Popen through wait/poll, never a separate + os.waitpid/SIGCHLD reaper. Under its reap lock, waitid(WNOWAIT) or an exact + kernel zombie-child row retains the PID through any final group signal. + A live ps row or Popen.returncode alone is not this authority. + """ + if not isinstance(process, _POPEN_TYPE): + return False + with _reap_guard(process, deadline=deadline, locked=locked): + if process.returncode is not None: + return False + waitid = getattr(os, "waitid", None) + if callable(waitid): + try: + result = waitid(os.P_PID, process.pid, os.WEXITED | os.WNOHANG | os.WNOWAIT) + except (OSError, AttributeError): + return False + return result is None or result.si_pid == process.pid + # Python 3.12 on macOS lacks waitid. Only its unreaped zombie child + # in our original new-session group can substitute; never a live row. + remaining = deadline - time.monotonic() + if remaining <= 0: + raise ContractError("diagnostic cleanup deadline expired") + try: + result = subprocess.run( + ["/bin/ps", "-p", str(process.pid), "-o", "pid=,ppid=,pgid=,stat="], + text=True, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, + env=subscription_environment(), timeout=min(0.5, remaining), check=False, + ) + except (OSError, subprocess.TimeoutExpired): + return False + if result.returncode != 0 or len(result.stdout) > 1024: + return False + rows = result.stdout.splitlines() + if len(rows) != 1: + return False + fields = rows[0].split() + return (len(fields) == 4 and all(field.isascii() and field.isdigit() for field in fields[:3]) + and tuple(map(int, fields[:3])) == (process.pid, os.getpid(), process.pid) + and fields[3].startswith("Z")) + + +def _close_direct_child(process: subprocess.Popen[Any], *, deadline: float) -> None: + """Stop only a kernel-confirmed live child, never its unowned group. + + WNOHANG verifies child authority under Popen's reap lock before each + direct signal. An exited child is reaped, not signaled. No waitid or + platform-specific ABI is needed for this fail-closed fallback. + """ + if not isinstance(process, _POPEN_TYPE): + raise ContractError("diagnostic direct-child ownership is unavailable") + + def child_live(value: int | None = None) -> bool: + if time.monotonic() >= deadline: + raise ContractError("diagnostic cleanup deadline expired") + with _reap_guard(process, deadline=deadline): + if process.returncode is not None: + return False + try: + waited, status = os.waitpid(process.pid, os.WNOHANG) + except OSError as exc: + raise ContractError("diagnostic direct-child ownership is unavailable") from exc + if waited == process.pid: + process._handle_exitstatus(status) + return False + if waited != 0: + raise ContractError("diagnostic direct-child ownership is unavailable") + if value is not None: + try: + os.kill(process.pid, value) + except ProcessLookupError: + pass + return True + + if child_live(signal.SIGTERM): + grace = min(deadline, time.monotonic() + PROBE_TERM_GRACE_SECONDS) + while child_live() and time.monotonic() < grace: + time.sleep(min(0.02, max(0, grace - time.monotonic()))) + if child_live(signal.SIGKILL): + while child_live(): + time.sleep(min(0.02, max(0, deadline - time.monotonic()))) + + +def wait_probe_exit( + process: subprocess.Popen[Any], *, start_identity: str | None, deadline: float, +) -> None: + """Observe natural completion without reaping the group ownership anchor. + + EOF does not imply child exit. Share the caller's probe deadline and use + WNOWAIT, or the exact retained zombie-child observation on older macOS, + before cleanup is allowed to terminate/reap the process group. + """ + if start_identity is None or not isinstance(process, _POPEN_TYPE): + raise ContractError("diagnostic process ownership is unavailable") + while True: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("diagnostic probe timed out") + with _reap_guard(process, deadline=deadline): + if process.returncode is not None: + raise ContractError("diagnostic process ownership anchor was already reaped") + observed = process_start_identity(process.pid) + if observed is not None and observed != start_identity: + raise ContractError("diagnostic process identity changed before natural exit") + waitid = getattr(os, "waitid", None) + if callable(waitid): + try: + result = waitid(os.P_PID, process.pid, os.WEXITED | os.WNOHANG | os.WNOWAIT) + except (OSError, AttributeError) as exc: + raise ContractError("diagnostic process ownership is unavailable") from exc + exited = result is not None and result.si_pid == process.pid + else: + exited = _retained_child_anchor(process, deadline=deadline, locked=True) + if exited: + return + time.sleep(min(0.02, max(0, deadline - time.monotonic()))) + + +def close_probe(process: subprocess.Popen[Any], *, start_identity: str | None) -> None: + """Stop the owned group, confirm no live members, reap and close streams. + + The process must have been spawned by the caller with a new session and + its identity captured before protocol use. Callers retain exclusive Popen + reaping until this function; no separate waitpid/SIGCHLD reaper may release + the child PID. Inspection/ownership failures + raise rather than pretending cleanup or readiness was confirmed. Missing + captured identity permits no group signals: only independently verified + direct-child stop/reap, followed by group absence or an unconfirmed error. + """ + deadline = time.monotonic() + PROBE_CLEANUP_SECONDS + def owned_group_exists(*, locked: bool = False) -> bool: + if not _probe_group_exists(process.pid, deadline=deadline): + return False + observed = process_start_identity(process.pid) + if observed is not None and observed != start_identity: + raise ContractError("diagnostic process identity changed before cleanup") + if observed is None and not _retained_child_anchor(process, deadline=deadline, locked=locked): + raise _ProbeOwnershipUnavailable("diagnostic process ownership unavailable") + return True + + def signal_group(value: int) -> None: + # Recheck immediately before each signal. A recycled leader PID is + # never authority to signal a new group. + # The lock retains any unreaped child anchor through the signal, even + # if another caller tries Popen.wait/poll during cleanup. + with _reap_guard(process, deadline=deadline): + if owned_group_exists(locked=True): + try: + os.killpg(process.pid, value) + except ProcessLookupError: + pass + + def close_without_group_authority() -> None: + _close_direct_child(process, deadline=deadline) + if _probe_group_exists(process.pid, deadline=deadline): + raise ContractError("diagnostic process group cleanup unconfirmed") + + try: + if start_identity is None: + close_without_group_authority() + return + if owned_group_exists(): + signal_group(signal.SIGTERM) + grace = min(deadline, time.monotonic() + PROBE_TERM_GRACE_SECONDS) + # Keep an exited direct child unreaped while descendants survive; + # its PID remains an ownership anchor through escalation. + while owned_group_exists() and time.monotonic() < grace: + time.sleep(min(0.02, max(0, grace - time.monotonic()))) + if owned_group_exists(): + signal_group(signal.SIGKILL) + while owned_group_exists(): + remaining = deadline - time.monotonic() + if remaining <= 0: + raise ContractError("diagnostic process group survived cleanup") + time.sleep(min(0.02, remaining)) + remaining = deadline - time.monotonic() + if remaining <= 0: + raise ContractError("diagnostic cleanup deadline expired") + process.wait(timeout=remaining) + except _ProbeOwnershipUnavailable: + close_without_group_authority() + finally: + for stream in (process.stdin, process.stdout, process.stderr): + if stream is not None: + stream.close() diff --git a/plugin/core/src/devsquad/release_activation.py b/plugin/core/src/devsquad/release_activation.py new file mode 100644 index 0000000..2d70b41 --- /dev/null +++ b/plugin/core/src/devsquad/release_activation.py @@ -0,0 +1,29 @@ +"""Select a prepared release without advancing an old ledger's schema.""" + +import os +from pathlib import Path +import sqlite3 + +def activate_release(temporary: Path, selector: Path, runtime: Path, *, supported_schema_version: int) -> None: + connection = None + try: + database = runtime / "state.sqlite3" + if database.exists(): + connection = sqlite3.connect(database, isolation_level=None, timeout=10) + connection.execute("BEGIN EXCLUSIVE") + version = connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] + if version > supported_schema_version: + raise RuntimeError("runtime schema is newer than this release; refusing downgrade") + if version < supported_schema_version: + pending = connection.execute("SELECT id FROM runs WHERE state NOT IN ('succeeded','failed','cancelled')").fetchall() + if pending: + raise RuntimeError( + "schema upgrade deferred: active/recoverable runs " + ", ".join(row[0] for row in pending) + + "; finish or cancel with the previous release, then retry" + ) + os.replace(temporary, selector) + if connection is not None: + connection.execute("COMMIT") + finally: + if connection is not None: + connection.close() diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py new file mode 100644 index 0000000..b0270d7 --- /dev/null +++ b/plugin/core/src/devsquad/reports.py @@ -0,0 +1,794 @@ +"""Deterministic M3 terminal receipt, Markdown, event-export and manifest builders.""" + +from __future__ import annotations + +import hashlib +from typing import Any + +from .contracts import ContractError +from .store import canonical_json, request_hash +from .workflows import ( + validate_branch_review_handoff, + validate_handoff_decision_evidence, +) + + +TERMINAL_REPORT_NAMES = frozenset({ + "receipt.json", + "receipt.md", + "events.jsonl", + "artifact-manifest.json", + "result-receipt.json", +}) +HANDOFF_REPORT_BASE_NAMES = frozenset({"handoff.json", "handoff.md"}) + + +def handoff_report_names(sequence: int) -> tuple[str, str]: + """Keep the first public names stable and retain later revision packets.""" + if type(sequence) is not int or sequence < 1: + raise ContractError("handoff report sequence is invalid") + if sequence == 1: + return "handoff.json", "handoff.md" + return f"handoff-{sequence}.json", f"handoff-{sequence}.md" + + +def _artifact_projection(artifact: dict[str, Any]) -> dict[str, Any]: + required = {"id", "name", "sha256", "byte_size"} + if not isinstance(artifact, dict) or not required <= set(artifact): + raise ContractError("report artifact projection is invalid") + if (not isinstance(artifact["id"], str) or not artifact["id"] + or not isinstance(artifact["name"], str) or not artifact["name"] + or not isinstance(artifact["sha256"], str) + or len(artifact["sha256"]) != 64 + or any(character not in "0123456789abcdef" + for character in artifact["sha256"]) + or type(artifact["byte_size"]) is not int + or artifact["byte_size"] < 0): + raise ContractError("report artifact projection values are invalid") + return {key: artifact[key] for key in ("id", "name", "sha256", "byte_size")} + + +def _event_export(events: list[dict[str, Any]]) -> tuple[bytes, int, int]: + if not isinstance(events, list): + raise ContractError("events export input must be an array") + lines = [] + last_cursor = 0 + last_version = 0 + for event in events: + if not isinstance(event, dict): + raise ContractError("events export entries must be objects") + cursor, version = event.get("id"), event.get("run_version") + if (type(cursor) is not int or cursor <= last_cursor + or type(version) is not int or version <= last_version): + raise ContractError("events export order is invalid") + last_cursor, last_version = cursor, version + lines.append(canonical_json(event)) + content = (("\n".join(lines) + "\n") if lines else "").encode() + return content, last_cursor, last_version + + +def _unreferenced_artifact_projection(artifact: dict[str, Any]) -> dict[str, Any]: + """Project an artifact before the atomic import has assigned its database id.""" + required = {"name", "sha256", "byte_size"} + if not isinstance(artifact, dict) or not required <= set(artifact): + raise ContractError("unreferenced report artifact projection is invalid") + name, digest, size = ( + artifact["name"], artifact["sha256"], artifact["byte_size"], + ) + if (not isinstance(name, str) or not name + or not isinstance(digest, str) or len(digest) != 64 + or any(character not in "0123456789abcdef" for character in digest) + or type(size) is not int or size < 0): + raise ContractError("unreferenced report artifact values are invalid") + return {"id": None, "name": name, "sha256": digest, "byte_size": size} + + +def _contents_with_manifest( + run_id: str, + receipt: dict[str, Any], + markdown: str, + events_content: bytes, + projected_artifacts: list[dict[str, Any]], +) -> dict[str, bytes]: + receipt_content = (canonical_json(receipt) + "\n").encode() + contents = { + "receipt.json": receipt_content, + "receipt.md": markdown.encode(), + "events.jsonl": events_content, + "result-receipt.json": receipt_content, + } + manifest_entries = list(projected_artifacts) + for name, content in contents.items(): + manifest_entries.append({ + "id": None, + "name": name, + "sha256": hashlib.sha256(content).hexdigest(), + "byte_size": len(content), + }) + manifest = { + "schema_version": 1, + "run_id": run_id, + "candidate_sha256": receipt["candidate"]["sha256"], + "artifacts": manifest_entries, + "self_excluded": True, + } + contents["artifact-manifest.json"] = ( + canonical_json(manifest) + "\n" + ).encode() + if set(contents) != TERMINAL_REPORT_NAMES: + raise ContractError("terminal report set is incomplete") + return contents + + +def build_handoff_reports( + *, + run_id: str, + handoff_id: str, + sequence: int, + packet: dict[str, Any], + packet_sha256: str, + created_at: str, +) -> dict[str, bytes]: + """Build the portable JSON and Markdown view of an open host handoff.""" + if not all(isinstance(value, str) and value for value in ( + run_id, handoff_id, packet_sha256, created_at, + )): + raise ContractError("handoff report identity is invalid") + packet_json = canonical_json(packet) + if hashlib.sha256(packet_json.encode()).hexdigest() != packet_sha256: + raise ContractError("handoff report packet hash is invalid") + json_name, markdown_name = handoff_report_names(sequence) + workflow = packet.get("workflow", "branch-review") + if workflow not in {"branch-review", "issue-delivery", "council-decision"}: + raise ContractError("handoff report workflow is invalid") + report = { + "schema_version": 1, + "run_id": run_id, + "workflow": workflow, + "state": "awaiting_host", + "created_at": created_at, + "handoff_id": handoff_id, + "sequence": sequence, + "packet_sha256": packet_sha256, + "candidate": { + "sha256": packet.get("candidate_sha256"), + "base_oid": packet.get("base_oid"), + "target_oid": packet.get("target_oid"), + }, + "packet": packet, + "next_action": "claim_handoff", + } + if workflow == "council-decision": + return {json_name: (canonical_json(report) + "\n").encode(), + markdown_name: (f"# DevSquad Council handoff\n\nRun: {run_id}\n\n" + canonical_json(packet) + "\n").encode()} + review = packet.get("review") if isinstance(packet.get("review"), dict) else {} + lines = [ + f"# DevSquad {workflow} handoff", + "", + f"- Run: `{run_id}`", + f"- Handoff: `{handoff_id}`", + f"- Sequence: `{sequence}`", + f"- Candidate: `{packet.get('candidate_sha256')}`", + f"- Review verdict: `{review.get('verdict', 'unavailable')}`", + "- Next action: claim this saved handoff and submit one disposition.", + "", + "## Review", + "", + str(review.get("summary") or "No review summary was supplied."), + "", + "## Findings", + "", + ] + findings = review.get("findings") + if isinstance(findings, list) and findings: + for finding in findings: + lines.append( + f"- **{str(finding.get('severity', 'unknown')).upper()} — " + f"{finding.get('title', 'Untitled finding')}** " + f"(`{finding.get('path', '?')}:{finding.get('start_line', '?')}`)" + ) + else: + lines.append("- No supported findings were reported.") + lines.extend(_check_markdown(packet.get("checks", []))) + lines.extend(["", "## Instructions", "", str(packet.get("instructions") or "")]) + return { + json_name: (canonical_json(report) + "\n").encode(), + markdown_name: ("\n".join(lines) + "\n").encode(), + } + + +def build_early_terminal_reports( + *, + run_id: str, + state: str, + task: dict[str, Any], + snapshot: dict[str, Any] | None, + run_artifacts: list[dict[str, Any]], + events: list[dict[str, Any]], + completed_at: str, + phase: str, + error: dict[str, Any] | None, + attempt: dict[str, Any] | None = None, + prior_attempts: list[dict[str, Any]] | None = None, + prior_dispositions: list[dict[str, Any]] | None = None, +) -> dict[str, bytes]: + """Build the M3 report set when no valid handoff/lead decision exists.""" + if not isinstance(run_id, str) or not run_id: + raise ContractError("report run id is invalid") + if state not in {"failed", "cancelled"}: + raise ContractError("early terminal report state is invalid") + if (not isinstance(task, dict) + or task.get("workflow") not in {"branch-review", "issue-delivery"}): + raise ContractError("early terminal report task is invalid") + if snapshot is not None and not isinstance(snapshot, dict): + raise ContractError("early terminal report snapshot is invalid") + if not isinstance(completed_at, str) or not completed_at or not phase: + raise ContractError("early terminal report completion is invalid") + if error is not None and not isinstance(error, dict): + raise ContractError("early terminal report error is invalid") + + frozen = snapshot or {} + workflow = task["workflow"] + workspace = frozen.get("workspace") + if (not isinstance(workspace, dict) and workflow == "issue-delivery" + and isinstance(frozen.get("delivery_iterations"), list) + and frozen["delivery_iterations"]): + workspace = frozen["delivery_iterations"][-1].get("workspace") + workspace = workspace if isinstance(workspace, dict) else {} + delivery = frozen.get("delivery_workspace") + delivery = delivery if isinstance(delivery, dict) else {} + projected = [ + _unreferenced_artifact_projection(artifact) for artifact in run_artifacts + ] + events_content, through_cursor, through_version = _event_export(events) + attempt_projection = None + prior = [] if prior_attempts is None else prior_attempts + if not isinstance(prior, list) or not all( + isinstance(item, dict) for item in prior): + raise ContractError("early terminal prior attempts are invalid") + dispositions = [] if prior_dispositions is None else prior_dispositions + if not isinstance(dispositions, list) or not all( + isinstance(item, dict) for item in dispositions): + raise ContractError("early terminal prior dispositions are invalid") + if attempt is not None: + if not isinstance(attempt, dict) or not isinstance(attempt.get("id"), str): + raise ContractError("early terminal report attempt is invalid") + attempt_projection = { + "id": attempt["id"], + "role": attempt.get("role", "reviewer"), + "status": state, + "returncode": attempt.get("returncode"), + "cancelled": bool(attempt.get("cancelled", state == "cancelled")), + "timed_out": bool(attempt.get("timed_out", False)), + "selected_profile": ( + attempt.get("selected_profile") + or frozen.get("routing", {}).get("roles", {}).get( + attempt.get("role", "reviewer"), {} + ).get("selected") + if isinstance(frozen.get("routing"), dict) else None + ), + "observed_identity": None, + "worker_invocations": 1, + "native_model_requests": None, + "usage": { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + "output_artifacts": projected, + "error": error, + } + if "native_diagnostics" in attempt: + attempt_projection["native_diagnostics"] = attempt["native_diagnostics"] + attempt_projection["usage"] = attempt["native_diagnostics"]["usage"] + criteria = [{ + "id": criterion.get("id"), + "description": criterion.get("description"), + "evidence_kind": criterion.get("evidence_kind"), + "status": "not_evaluated", + "evidence_refs": [], + } for criterion in task.get("acceptance", []) if isinstance(criterion, dict)] + # A failed lead does not erase the current candidate's completed review or + # check integrity verdict. Never project a prior delivery candidate here. + latest_review = next((item for item in reversed(prior) + if item.get("role") == "reviewer" + and isinstance(item.get("evaluation"), dict) + and item["evaluation"].get("candidate_sha256") == workspace.get("candidate_sha256") + ), {}) + receipt = { + "schema_version": 1, + "run_id": run_id, + "workflow": workflow, + "state": state, + "phase": phase, + "completed_at": completed_at, + "candidate": { + "sha256": workspace.get("candidate_sha256"), + "base_oid": workspace.get( + "base_oid", delivery.get("baseline_oid", frozen.get("base_oid")), + ), + "target_oid": workspace.get( + "target_oid", delivery.get("baseline_oid", frozen.get("target_oid")), + ), + }, + "routing": frozen.get("routing"), + "review": latest_review.get("review"), + "checks": latest_review.get("checks", []), + "evaluation": latest_review.get("evaluation"), + "criteria": criteria, + "attempts": prior + ([attempt_projection] if attempt_projection else []), + "dispositions": dispositions, + "lead": { + "mode": task.get("lead", {}).get("mode"), + "status": ( + "cancelled" + if state == "cancelled" and phase in {"lead", "awaiting_host"} + else "failed" if phase == "lead" else "not_reached" + ), + "disposition": None, + "reason": None, + "usage": { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + }, + "accounting": { + "worker_invocations": sum( + item.get("worker_invocations", 0) for item in prior + ) + (1 if attempt_projection else 0), + "native_model_requests": None if (prior or attempt_projection) else 0, + "attempt_usage": ( + [item["usage"] for item in prior] + + ([attempt_projection["usage"]] if attempt_projection else []) + ), + "host_usage_measured": False, + }, + "artifacts": projected, + "evidence_artifacts": [], + "events_export": { + "through_cursor": through_cursor, + "through_run_version": through_version, + "includes_terminal_event": False, + "excludes_terminal_report_artifact_events": True, + }, + "limitations": [( + "The run was cancelled before a lead disposition became terminal." + if state == "cancelled" + else "The headless lead failed before a valid disposition was recorded." + if phase == "lead" + else "No valid review handoff was produced, so lead disposition was not reached." + )], + "error": error, + } + lines = [ + f"# DevSquad {workflow}", + "", + f"- Run: `{run_id}`", + f"- State: `{state}`", + f"- Phase: `{phase}`", + "- Lead disposition: `not_reached`", + "", + "## Failure", + "", + str((error or {}).get("message") or (error or {}).get("error") or "Cancelled."), + "", + "## Recovery", + "", + "Start a new run with a new idempotency key after correcting the recorded error.", + "", + ] + lines.extend(_check_markdown(receipt["checks"])) + return _contents_with_manifest( + run_id, receipt, "\n".join(lines), events_content, projected, + ) + + +def _decision(value: Any) -> dict[str, Any]: + fields = { + "schema_version", "submission_id", "submission_hash", "disposition", + "reason", "evidence_refs", + } + if not isinstance(value, dict) or set(value) != fields: + raise ContractError("report lead decision fields are invalid") + if value["schema_version"] != 1 or type(value["schema_version"]) is not int: + raise ContractError("report lead decision schema_version is invalid") + if (not isinstance(value["submission_id"], str) or not value["submission_id"] + or value["disposition"] not in {"accept", "revise", "reject"} + or not isinstance(value["reason"], str) + or (value["disposition"] == "revise" and not value["reason"])): + raise ContractError("report lead decision values are invalid") + body = {key: value[key] for key in fields if key != "submission_hash"} + if value["submission_hash"] != request_hash(body): + raise ContractError("report lead decision hash is invalid") + return value + + +def validate_saved_review_handoff( + packet: dict[str, Any], + snapshot: dict[str, Any], +) -> dict[str, Any]: + """Validate current or archived delivery evidence against its own workspace.""" + if snapshot.get("task", {}).get("workflow") != "issue-delivery": + return validate_branch_review_handoff(packet, snapshot) + candidate_sha256 = packet.get("candidate_sha256") + for iteration in snapshot.get("delivery_iterations", []): + if (isinstance(iteration, dict) + and iteration.get("candidate", {}).get("candidate_sha256") + == candidate_sha256 + and isinstance(iteration.get("workspace"), dict)): + historical = dict(snapshot) + historical["candidate"] = iteration["candidate"] + historical["workspace"] = iteration["workspace"] + historical["check_workspace"] = iteration.get("check_workspace") + return validate_branch_review_handoff(packet, historical) + raise ContractError("delivery handoff does not match a saved candidate iteration") + + +def _history( + entries: list[dict[str, Any]], + snapshot: dict[str, Any], +) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + if not isinstance(entries, list) or not entries: + raise ContractError("branch review report history must be non-empty") + attempts = [] + dispositions = [] + prior_sequence = 0 + candidate = None + workflow = snapshot.get("task", {}).get("workflow") + if workflow not in {"branch-review", "issue-delivery"}: + raise ContractError("review report workflow is invalid") + for index, entry in enumerate(entries): + required = { + "handoff_id", "sequence", "packet", "packet_sha256", "decision", + "recorded_run_version", + } + if not isinstance(entry, dict) or set(entry) != required: + raise ContractError("branch review report history entry is invalid") + if (not isinstance(entry["handoff_id"], str) or not entry["handoff_id"] + or type(entry["sequence"]) is not int + or entry["sequence"] <= prior_sequence + or type(entry["recorded_run_version"]) is not int + or entry["recorded_run_version"] < 1): + raise ContractError("branch review report history order is invalid") + packet_json = canonical_json(entry["packet"]) + if hashlib.sha256(packet_json.encode()).hexdigest() != entry["packet_sha256"]: + raise ContractError("branch review report handoff hash is invalid") + packet = validate_saved_review_handoff(entry["packet"], snapshot) + decision = _decision(entry["decision"]) + validate_handoff_decision_evidence(decision, packet) + identity = ( + packet["candidate_sha256"], packet["base_oid"], packet["target_oid"], + ) + if candidate is None: + candidate = identity + elif identity != candidate and workflow == "branch-review": + raise ContractError("branch review report history changes the candidate") + elif identity == candidate and workflow == "issue-delivery": + raise ContractError("delivery report history repeats a candidate") + if workflow == "issue-delivery": + iterations = snapshot.get("delivery_iterations", []) + if (index >= len(iterations) + or packet["candidate_sha256"] + != iterations[index].get("candidate", {}).get( + "candidate_sha256" + )): + raise ContractError("delivery report candidate order is invalid") + if index < len(entries) - 1 and decision["disposition"] != "revise": + raise ContractError("only a revision may precede another review attempt") + attempts.append({ + "id": packet["attempt_id"], + "sequence": entry["sequence"], + **packet["attempt"], + "review": packet["review"], + "checks": packet["checks"], + "evaluation": packet["evaluation"], + "evidence_refs": packet["artifacts"], + }) + dispositions.append({ + "handoff_id": entry["handoff_id"], + "sequence": entry["sequence"], + "submission_id": decision["submission_id"], + "submission_hash": decision["submission_hash"], + "disposition": decision["disposition"], + "reason": decision["reason"], + "evidence_refs": decision["evidence_refs"], + "recorded_run_version": entry["recorded_run_version"], + }) + prior_sequence = entry["sequence"] + return attempts, dispositions + + +def project_branch_review_history( + entries: list[dict[str, Any]], + snapshot: dict[str, Any], +) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + """Validate and project any completed review revisions for early reports.""" + if entries == []: + return [], [] + return _history(entries, snapshot) + + +def _check_markdown(checks: list[dict[str, Any]]) -> list[str]: + lines = ["", "## Checks", ""] + for check in checks: + requirement = "required" if check["required_to_pass"] else "report-only" + lines.append(f"- `{check['id']}`: **{check['status']}** ({requirement})") + for reason in check.get("integrity", {}).get("reasons", []): + lines.append(f" - Candidate integrity: `{reason}`") + if not checks: + lines.append("- No checks were declared.") + return lines + + +def _markdown(receipt: dict[str, Any]) -> str: + review = receipt["review"] + lines = [ + f"# DevSquad {receipt['workflow']}", + "", + f"- Run: `{receipt['run_id']}`", + f"- State: `{receipt['state']}`", + f"- Disposition: `{receipt['lead']['disposition']}`", + f"- Candidate: `{receipt['candidate']['sha256']}`", + f"- Base: `{receipt['candidate']['base_oid']}`", + f"- Target: `{receipt['candidate']['target_oid']}`", + f"- Review verdict: `{review['verdict']}`", + f"- Review attempts: `{len(receipt['attempts'])}`", + "", + "## Review", + "", + review["summary"], + "", + ] + if review["findings"]: + lines.extend(["## Findings", ""]) + for finding in review["findings"]: + lines.extend([ + f"- **{finding['severity'].upper()} — {finding['title']}** " + f"(`{finding['path']}:{finding['start_line']}`): " + f"{finding['description']}", + f" Evidence: {finding['evidence']}", + ]) + lines.append("") + lines.extend(_check_markdown(receipt["checks"])) + lines.extend([ + "", + "## Lead disposition", + "", + receipt["lead"]["reason"] or "No reason supplied.", + "", + "## Limitations", + "", + ]) + if receipt["limitations"]: + lines.extend(f"- {item}" for item in receipt["limitations"]) + else: + lines.append("- None recorded.") + return "\n".join(lines) + "\n" + + +def build_terminal_reports( + *, + run_id: str, + state: str, + snapshot: dict[str, Any], + history: list[dict[str, Any]], + run_artifacts: list[dict[str, Any]], + events: list[dict[str, Any]], + completed_at: str, + error: dict[str, Any] | None = None, + headless_leads: list[dict[str, Any]] | None = None, + failed_attempts: list[dict[str, Any]] | None = None, +) -> dict[str, bytes]: + if not isinstance(run_id, str) or not run_id: + raise ContractError("report run id is invalid") + if state not in {"succeeded", "failed"}: + raise ContractError("review report state is invalid") + if not isinstance(snapshot, dict) or not isinstance(snapshot.get("routing"), dict): + raise ContractError("branch review report snapshot is invalid") + workflow = snapshot.get("task", {}).get("workflow") + if workflow not in {"branch-review", "issue-delivery"}: + raise ContractError("review report workflow is invalid") + attempts, dispositions = _history(history, snapshot) + final_packet = validate_saved_review_handoff(history[-1]["packet"], snapshot) + final_decision = _decision(history[-1]["decision"]) + if (state == "succeeded") != (final_decision["disposition"] == "accept"): + raise ContractError("terminal state and lead disposition disagree") + if not isinstance(completed_at, str) or not completed_at: + raise ContractError("report completion timestamp is invalid") + if error is not None and not isinstance(error, dict): + raise ContractError("report error must be an object or null") + lead_mode = snapshot["task"]["lead"]["mode"] + lead_evidence = [] if headless_leads is None else headless_leads + if not isinstance(lead_evidence, list): + raise ContractError("headless lead report evidence must be an array") + if lead_mode == "headless": + if len(lead_evidence) != len(history): + raise ContractError("headless lead report history is incomplete") + for evidence, history_entry in zip(lead_evidence, history): + if (not isinstance(evidence, dict) + or evidence.get("handoff_id") != history_entry["handoff_id"] + or evidence.get("choice", {}).get("disposition") + != history_entry["decision"]["disposition"]): + raise ContractError("headless lead report evidence changes its decision") + elif lead_evidence: + raise ContractError("host-led report cannot contain headless lead evidence") + + projected = [_artifact_projection(artifact) for artifact in run_artifacts] + artifact_by_id = {artifact["id"]: artifact for artifact in projected} + if len(artifact_by_id) != len(projected): + raise ContractError("report evidence artifacts are duplicated") + expected_ids = set() + for entry in history: + for reference in entry["packet"]["artifacts"]: + artifact = artifact_by_id.get(reference["artifact_id"]) + if (artifact is None or artifact["name"] != reference["name"] + or artifact["sha256"] != reference["sha256"]): + raise ContractError( + "handoff evidence does not match stored report artifacts" + ) + expected_ids.add(reference["artifact_id"]) + evidence_projected = [ + artifact for artifact in projected if artifact["id"] in expected_ids + ] + + events_content, through_cursor, through_version = _event_export(events) + limitations = [] + if "internal_review_fixture" in snapshot: + limitations.append( + "Reviewer output came from the explicit offline fixture; " + "it is not live-provider evidence." + ) + if "internal_implementation_fixture" in snapshot: + limitations.append( + "Implementation output came from the explicit offline fixture; " + "it is not live-provider evidence." + ) + lead_attempts = [evidence["attempt"] for evidence in lead_evidence] + failed = [] if failed_attempts is None else failed_attempts + if not isinstance(failed, list) or not all( + isinstance(attempt, dict) + and attempt.get("role") in {"implementer", "reviewer", "lead"} + for attempt in failed): + raise ContractError("failed fallback attempts are invalid") + failed_reviewers = [ + attempt for attempt in failed if attempt["role"] == "reviewer" + ] + failed_implementers = [ + attempt for attempt in failed if attempt["role"] == "implementer" + ] + failed_leads = [attempt for attempt in failed if attempt["role"] == "lead"] + reviewer_attempts = failed_reviewers + attempts + lead_attempts = failed_leads + lead_attempts + implementation_attempts = [] + if workflow == "issue-delivery": + iterations = snapshot.get("delivery_iterations") + if not isinstance(iterations, list) or not iterations: + raise ContractError("delivery report has no candidate iterations") + for expected_iteration, iteration in enumerate(iterations, 1): + if (not isinstance(iteration, dict) + or iteration.get("iteration") != expected_iteration + or not isinstance(iteration.get("candidate"), dict) + or not isinstance(iteration.get("implementation"), dict) + or not isinstance(iteration["implementation"].get("attempt"), dict)): + raise ContractError("delivery report candidate iteration is invalid") + artifact_name = iteration["candidate"].get("implementation_artifact") + if (not isinstance(artifact_name, str) + or not artifact_name.startswith("implementation-attempt-") + or not artifact_name.endswith(".json")): + raise ContractError("delivery report implementation artifact is invalid") + implementation_attempts.append({ + "id": artifact_name[ + len("implementation-attempt-"):-len(".json") + ], + "status": "succeeded", + "sequence": expected_iteration, + **iteration["implementation"]["attempt"], + "summary": iteration["implementation"].get("summary"), + "candidate": iteration["candidate"], + "evidence_refs": [artifact_name], + }) + implementation_attempts = failed_implementers + implementation_attempts + all_attempts = implementation_attempts + reviewer_attempts + lead_attempts + all_native_counts = [ + attempt["native_model_requests"] for attempt in all_attempts + ] + receipt = { + "schema_version": 1, + "run_id": run_id, + "workflow": workflow, + "state": state, + "completed_at": completed_at, + "candidate": { + "sha256": final_packet["candidate_sha256"], + "base_oid": final_packet["base_oid"], + "target_oid": final_packet["target_oid"], + }, + "routing": snapshot["routing"], + "review": final_packet["review"], + "checks": final_packet["checks"], + "evaluation": final_packet["evaluation"], + "criteria": final_packet["evaluation"]["criteria"], + "attempts": ( + implementation_attempts + reviewer_attempts + if workflow == "issue-delivery" else reviewer_attempts + ), + **({"delivery_iterations": snapshot["delivery_iterations"]} + if workflow == "issue-delivery" else {}), + "dispositions": dispositions, + "revisions": { + "requested": sum( + item["disposition"] == "revise" for item in dispositions + ), + "executed": len(attempts) - 1, + "maximum": snapshot["task"]["budget"]["max_revisions"], + }, + "lead": { + "mode": lead_mode, + "disposition": final_decision["disposition"], + "reason": final_decision["reason"], + "submission_id": final_decision["submission_id"], + "submission_hash": final_decision["submission_hash"], + "evidence_refs": final_decision["evidence_refs"], + "attempts": lead_attempts, + "usage": ( + lead_attempts[-1]["usage"] if lead_attempts else { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + } + ), + }, + "accounting": { + "worker_invocations": sum( + attempt["worker_invocations"] for attempt in all_attempts + ), + "native_model_requests": ( + None if any(value is None for value in all_native_counts) + else sum(all_native_counts) + ), + "attempt_usage": [attempt["usage"] for attempt in all_attempts], + "host_usage_measured": False if lead_mode == "host" else None, + }, + "artifacts": projected, + "evidence_artifacts": evidence_projected, + "events_export": { + "through_cursor": through_cursor, + "through_run_version": through_version, + "includes_terminal_event": False, + "excludes_terminal_report_artifact_events": True, + }, + "limitations": limitations, + "error": error, + } + receipt_content = (canonical_json(receipt) + "\n").encode() + contents = { + "receipt.json": receipt_content, + "receipt.md": _markdown(receipt).encode(), + "events.jsonl": events_content, + "result-receipt.json": receipt_content, + } + manifest_entries = list(projected) + for name, content in contents.items(): + manifest_entries.append({ + "id": None, + "name": name, + "sha256": hashlib.sha256(content).hexdigest(), + "byte_size": len(content), + }) + manifest = { + "schema_version": 1, + "run_id": run_id, + "candidate_sha256": final_packet["candidate_sha256"], + "artifacts": manifest_entries, + "self_excluded": True, + } + contents["artifact-manifest.json"] = ( + canonical_json(manifest) + "\n" + ).encode() + if set(contents) != TERMINAL_REPORT_NAMES: + raise ContractError("terminal report set is incomplete") + return contents diff --git a/plugin/core/src/devsquad/review_worker.py b/plugin/core/src/devsquad/review_worker.py new file mode 100644 index 0000000..feaee50 --- /dev/null +++ b/plugin/core/src/devsquad/review_worker.py @@ -0,0 +1,277 @@ +"""Run one frozen branch-review fixture and its trusted declared checks.""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import time +from typing import Any + +from .contracts import ContractError +from .store import canonical_json +from .supervisor import BoundedDrain +from .workflows import ( + MAX_PREVIEW_CHARS, + make_branch_review_evidence, + validate_review_document, +) +from .workspaces import candidate_input_state, dirty_paths, reset_check_workspace + + +MAX_SNAPSHOT_BYTES = 2 * 1024 * 1024 + + +def _empty_stream() -> dict[str, Any]: + return { + "preview": "", + "captured_bytes": 0, + "total_bytes": 0, + "truncated": False, + "full_sha256": hashlib.sha256(b"").hexdigest(), + } + + +def _stream_result(drain: BoundedDrain) -> dict[str, Any]: + metadata = drain.finish() + return { + "preview": bytes(drain.content).decode("utf-8", "replace"), + "captured_bytes": metadata["captured_bytes"], + "total_bytes": metadata["total_bytes"], + "truncated": metadata["truncated"], + "full_sha256": metadata["full_sha256"], + } + + +def _safe_check_cwd(root: Path, relative: str) -> Path: + try: + candidate = (root / relative).resolve(strict=True) + except OSError as exc: + raise ContractError(f"declared check cwd does not exist: {relative}") from exc + if (candidate != root and root not in candidate.parents) or not candidate.is_dir(): + raise ContractError(f"declared check cwd escapes the check workspace: {relative}") + return candidate + + +def _run_check( + check: dict[str, Any], + check_workspace: Path, + candidate_sha256: str, + target_oid: str, +) -> dict[str, Any]: + started = time.monotonic() + stdout_result, stderr_result = _empty_stream(), _empty_stream() + cwd = _safe_check_cwd(check_workspace, check["cwd"]) + with tempfile.TemporaryDirectory(prefix="devsquad-check-home-") as check_home: + environment = os.environ.copy() + environment["HOME"] = check_home + environment["PYTHONDONTWRITEBYTECODE"] = "1" + environment["PYTHONPYCACHEPREFIX"] = str(Path(check_home) / "python-cache") + try: + process = subprocess.Popen( + check["argv"], + cwd=cwd, + env=environment, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + start_new_session=False, + close_fds=True, + ) + except OSError: + status, returncode, error_code = "launch_failed", None, "CLI_ERROR" + else: + assert process.stdout is not None and process.stderr is not None + stdout = BoundedDrain(process.stdout, MAX_PREVIEW_CHARS) + stderr = BoundedDrain(process.stderr, MAX_PREVIEW_CHARS) + stdout.start() + stderr.start() + try: + returncode = process.wait(timeout=check["timeout_seconds"]) + status = "passed" if returncode == 0 else "failed" + error_code = None + except subprocess.TimeoutExpired: + status, returncode, error_code = "timed_out", None, "TIMEOUT" + process.terminate() + try: + process.wait(timeout=1) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=2) + stdout_result, stderr_result = _stream_result(stdout), _stream_result(stderr) + return { + "schema_version": 1, + "candidate_sha256": candidate_sha256, + "target_oid": target_oid, + "id": check["id"], + "argv": check["argv"], + "cwd": check["cwd"], + "required_to_pass": check["required_to_pass"], + "status": status, + "returncode": returncode, + "error_code": error_code, + "duration_ms": max(0, int((time.monotonic() - started) * 1000)), + "stdout": stdout_result, + "stderr": stderr_result, + } + + +def _verify_finding_locations(review: dict[str, Any], workspace: Path) -> None: + for finding in review["findings"]: + try: + target = (workspace / finding["path"]).resolve(strict=True) + except OSError as exc: + raise ContractError( + f"review finding path does not exist: {finding['path']}" + ) from exc + if target != workspace and workspace not in target.parents: + raise ContractError("review finding path escapes the frozen workspace") + if not target.is_file(): + raise ContractError(f"review finding path is not a file: {finding['path']}") + lines = 0 + with target.open("rb") as stream: + for lines, _ in enumerate(stream, 1): + if lines >= finding["end_line"]: + break + if lines < finding["end_line"]: + raise ContractError( + f"review finding line is outside the frozen file: {finding['path']}" + ) + + +def run_review_and_checks( + snapshot: dict[str, Any], + review: dict[str, Any], + **attempt_metadata: Any, +) -> dict[str, Any]: + """Validate one review, run the frozen checks, and build combined evidence.""" + task = snapshot.get("task") + workspace = snapshot.get("workspace") + check_workspace = snapshot.get("check_workspace") + if not all(isinstance(value, dict) for value in (task, workspace, check_workspace)): + raise ContractError("branch review snapshot is incomplete") + normalized_review = validate_review_document(review, task, workspace) + review_root = Path(workspace["path"]).resolve(strict=True) + checks_root = Path(check_workspace["path"]).resolve(strict=True) + reset_check_workspace( + review_root, + checks_root, + workspace["target_oid"], + check_workspace["scope"], + ) + if dirty_paths(review_root): + raise ContractError("frozen review workspace is dirty before reviewer execution") + _verify_finding_locations(normalized_review, review_root) + if dirty_paths(review_root): + raise ContractError("reviewer modified the frozen read-only workspace") + target_oid = workspace["target_oid"] + roots = {"check": checks_root, "review": review_root} + outputs = [path for check in task["checks"] for path in check.get("output_paths", [])] + baseline = { + label: candidate_input_state(root, target_oid, outputs if label == "check" else ()) + for label, root in roots.items() + } + + baseline_sha256 = hashlib.sha256(canonical_json(baseline).encode()).hexdigest() + + def inspect_integrity() -> dict[str, Any]: + reasons, changes, states = [], [], {} + for label, root in roots.items(): + try: + current = candidate_input_state(root, target_oid, outputs if label == "check" else ()) + except (ContractError, OSError, ValueError): + reasons.append(f"{label}:inspection_failed") + states[label] = None + continue + states[label] = current + reasons.extend( + f"{label}:{field}_changed" for field, value in baseline[label].items() + if field != "paths" and current[field] != value + ) + before, after = baseline[label]["paths"], current["paths"] + for path in sorted(before.keys() | after.keys()): + if before.get(path) != after.get(path): + changes.append({ + "workspace": label, "path": path, + "before": before.get(path), "after": after.get(path), + }) + return { + "reasons": reasons, + "before_state_sha256": baseline_sha256, + "after_state_sha256": hashlib.sha256(canonical_json(states).encode()).hexdigest(), + "changes": changes[:100], "changes_truncated": len(changes) > 100, + } + + checks = [] + invalidated = False + for check in task["checks"]: + inspection = inspect_integrity() if not invalidated else { + "reasons": [], "before_state_sha256": None, "after_state_sha256": None, + "changes": [], "changes_truncated": False, + } + reasons = inspection["reasons"] + if invalidated or reasons: + # Keep one outcome for every declared check without executing any + # dependent command on a contaminated candidate. + result = { + "candidate_sha256": workspace["candidate_sha256"], + "target_oid": target_oid, + **{key: check[key] for key in ("id", "argv", "cwd", "required_to_pass")}, + "status": "not_run" if invalidated else "invalidated", + "returncode": None, + "error_code": "CLI_ERROR", + "duration_ms": 0, + "stdout": _empty_stream(), + "stderr": _empty_stream(), + } + integrity_status = "not_run" if invalidated else "violated" + reasons = reasons or ["prior_check_invalidated_candidate"] + else: + result = _run_check( + check, checks_root, workspace["candidate_sha256"], target_oid, + ) + inspection = inspect_integrity() + reasons = inspection["reasons"] + integrity_status = "violated" if reasons else "verified" + if reasons: + result.update(status="invalidated", error_code="CLI_ERROR") + result["schema_version"] = 2 + result["output_paths"] = check.get("output_paths", []) + result["integrity"] = {**inspection, "status": integrity_status, "reasons": reasons} + invalidated = invalidated or integrity_status != "verified" + checks.append(result) + return make_branch_review_evidence( + snapshot, normalized_review, checks, **attempt_metadata, + ) + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("workflow snapshot must be an object") + fixture = snapshot.get("internal_review_fixture") + if not isinstance(fixture, dict): + raise ContractError("offline review snapshot is incomplete") + selected = snapshot["routing"]["roles"]["reviewer"]["selected"] + if selected.get("profile_id", "").endswith("-fixture-fail"): + raise ContractError("offline reviewer fixture requested a failed attempt") + return run_review_and_checks(snapshot, fixture) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_SNAPSHOT_BYTES + 1) + if len(payload) > MAX_SNAPSHOT_BYTES: + raise ContractError("workflow snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("workflow snapshot is not valid UTF-8 JSON") from exc + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/router.py b/plugin/core/src/devsquad/router.py new file mode 100644 index 0000000..fbb2fd3 --- /dev/null +++ b/plugin/core/src/devsquad/router.py @@ -0,0 +1,542 @@ +"""Deterministic, side-effect-free M3 execution-profile selection.""" + +from __future__ import annotations + +from datetime import datetime +import hashlib +import json +from typing import Any, Callable + +from .contracts import ( + CapabilityUnavailable, + ContractError, + PolicyDenied, + ProfileUnsupported, +) +from .store import canonical_json +from .validation import validate_policy, validate_profile_registry, validate_task + + +QUALITY_RANK = {"unvalidated": 0, "trial": 1, "proven": 2} +ROLE_PERMISSIONS = { + "implementer": "workspace_write", + "reviewer": "read_only", + "lead": "read_only", + "researcher": "read_only", + "proposer_a": "read_only", "proposer_b": "read_only", "critic": "read_only", +} +WORKFLOW_ROLES = { + "branch-review": ("reviewer",), + "issue-delivery": ("implementer", "reviewer"), + "council-decision": ("proposer_a", "proposer_b", "critic"), +} +CAPACITY_STATES = {"available", "exhausted", "unknown"} +CAPACITY_EVIDENCE_FIELDS = { + "schema_version", "pool_id", "status", "in_flight", "observed_at", + "evaluated_at", "target", "windows", "reasons", +} + + +def _strict_json(payload: bytes | str, label: str) -> tuple[dict[str, Any], str]: + if isinstance(payload, str): + encoded = payload.encode() + elif isinstance(payload, bytes): + encoded = payload + else: + raise ContractError(f"{label} must be JSON bytes or text") + + def object_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result = {} + for key, value in pairs: + if key in result: + raise ContractError(f"{label} contains duplicate key: {key}") + result[key] = value + return result + + def reject_constant(value: str) -> None: + raise ContractError(f"{label} contains non-finite number: {value}") + + try: + value = json.loads( + encoded.decode("utf-8"), + object_pairs_hook=object_pairs, + parse_constant=reject_constant, + ) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError(f"{label} is not valid UTF-8 JSON") from exc + if not isinstance(value, dict): + raise ContractError(f"{label} must contain a JSON object") + return value, hashlib.sha256(encoded).hexdigest() + + +def _capacity_timestamp(value: Any, field: str, *, nullable: bool = False) -> None: + if value is None and nullable: + return + if not isinstance(value, str) or not value: + raise ContractError(f"capacity {field} must be a timestamp") + try: + parsed = datetime.fromisoformat(value) + except ValueError as exc: + raise ContractError(f"capacity {field} must be an ISO timestamp") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ContractError(f"capacity {field} must include a timezone") + + +def _capacity_evidence( + value: Any, + *, + pool_id: str, + target: dict[str, str] | None, +) -> dict[str, Any]: + if not isinstance(value, dict) or set(value) != CAPACITY_EVIDENCE_FIELDS: + raise ContractError("capacity evidence fields are invalid") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("capacity evidence schema_version is invalid") + if value["pool_id"] != pool_id or value["target"] != target: + raise ContractError("capacity evidence identity is invalid") + if not isinstance(value["status"], str) or value["status"] not in CAPACITY_STATES: + raise ContractError("capacity status is invalid") + if type(value["in_flight"]) is not int or value["in_flight"] < 0: + raise ContractError("capacity in_flight must be a non-negative integer") + _capacity_timestamp(value["observed_at"], "observed_at", nullable=True) + _capacity_timestamp(value["evaluated_at"], "evaluated_at") + if not isinstance(value["windows"], list) or not all( + isinstance(item, dict) for item in value["windows"] + ): + raise ContractError("capacity windows must be an object array") + if not isinstance(value["reasons"], list) or not all( + isinstance(item, str) and item for item in value["reasons"] + ): + raise ContractError("capacity reasons must be a string array") + return json.loads(canonical_json(value)) + + +def _availability_snapshot( + policy: dict[str, Any], + profiles: dict[str, dict[str, Any]], + availability: dict[str, Any] | None, +) -> dict[str, dict[str, Any]]: + supplied = {} if availability is None else availability + if not isinstance(supplied, dict): + raise ContractError("capacity availability must be an object") + unknown_pools = set(supplied) - set(policy["account_pools"]) + if unknown_pools: + raise ContractError(f"capacity references unknown account pools: {sorted(unknown_pools)}") + result = {} + for pool_id, pool_policy in policy["account_pools"].items(): + observation = supplied.get(pool_id, {"status": "unknown", "in_flight": 0}) + if not isinstance(observation, dict): + raise ContractError("capacity observation must be an object") + if "profiles" in observation: + expected_profiles = { + profile_id: profile + for profile_id, profile in profiles.items() + if profile["account_pool_id"] == pool_id + } + profile_values = observation["profiles"] + if not isinstance(profile_values, dict): + raise ContractError("capacity profile evidence must be an object") + unknown_profiles = set(profile_values) - set(expected_profiles) + if unknown_profiles: + raise ContractError( + f"capacity references unknown pool profiles: {sorted(unknown_profiles)}", + ) + root = _capacity_evidence( + {key: value for key, value in observation.items() if key != "profiles"}, + pool_id=pool_id, + target=None, + ) + root["profiles"] = {} + for profile_id, evidence in profile_values.items(): + profile = expected_profiles[profile_id] + target = { + field: profile[field] + for field in ("harness", "model_family", "model_id") + } + root["profiles"][profile_id] = _capacity_evidence( + evidence, pool_id=pool_id, target=target, + ) + root["max_concurrency"] = pool_policy["max_concurrency"] + root["unknown_capacity_policy"] = pool_policy.get( + "unknown_capacity_policy", "allow_bounded", + ) + result[pool_id] = root + continue + unknown = set(observation) - {"status", "in_flight", "observed_at"} + missing = {"status", "in_flight"} - set(observation) + if unknown or missing: + raise ContractError( + f"capacity observation fields invalid: unknown={sorted(unknown)} " + f"missing={sorted(missing)}" + ) + status, in_flight = observation["status"], observation["in_flight"] + if not isinstance(status, str) or status not in CAPACITY_STATES: + raise ContractError("capacity status is invalid") + if type(in_flight) is not int or in_flight < 0: + raise ContractError("capacity in_flight must be a non-negative integer") + observed_at = observation.get("observed_at") + if observed_at is not None: + if not isinstance(observed_at, str) or not observed_at: + raise ContractError("capacity observed_at must be a timestamp or null") + try: + parsed = datetime.fromisoformat(observed_at) + except ValueError as exc: + raise ContractError("capacity observed_at must be an ISO timestamp") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ContractError("capacity observed_at must include a timezone") + result[pool_id] = { + "status": status, + "in_flight": in_flight, + "observed_at": observed_at, + "max_concurrency": pool_policy["max_concurrency"], + "unknown_capacity_policy": pool_policy.get( + "unknown_capacity_policy", "allow_bounded", + ), + } + return result + + +def _resolve_reference( + reference: dict[str, str], + profiles: dict[str, dict[str, Any]], + bindings: dict[str, dict[str, Any]], +) -> tuple[dict[str, Any] | None, dict[str, Any] | None, str | None]: + if reference["kind"] == "profile": + profile = profiles.get(reference["id"]) + return profile, None, None if profile else "profile_not_found" + binding = bindings.get(reference["id"]) + if binding is None: + return None, None, "alias_unbound" + profile = profiles.get(binding["profile_id"]) + if profile is None: # Defensive: registry validation already establishes this. + return None, None, "binding_target_missing" + return profile, { + "alias": reference["id"], + "version": binding["version"], + "profile_id": binding["profile_id"], + }, None + + +def _static_reason( + profile: dict[str, Any], + role: str, + minimum_quality: str, + pool_policy: dict[str, Any] | None, + implementer: dict[str, Any] | None, + require_different_model: bool, +) -> str | None: + if profile["quality_status"] == "suspended": + return "profile_suspended" + minimum_rank = QUALITY_RANK.get(minimum_quality) + profile_rank = QUALITY_RANK.get(profile["quality_status"]) + if minimum_rank is None or profile_rank is None or profile_rank < minimum_rank: + return "quality_below_task_minimum" + if profile["permission_policy"] != ROLE_PERMISSIONS[role]: + return "permission_mismatch" + if pool_policy is None: + return "account_pool_not_declared" + if profile["billing_mode"] not in pool_policy["allowed_billing_modes"]: + return "billing_mode_not_allowed" + if (role == "reviewer" and implementer is not None and require_different_model + and profile["model_family"] == implementer["model_family"] + and profile["model_id"] == implementer["model_id"]): + return "review_model_not_independent" + return None + + +def _capacity_reason( + profile: dict[str, Any], capacity: dict[str, dict[str, Any]], +) -> str | None: + pool = capacity[profile["account_pool_id"]] + profile_capacity = pool.get("profiles", {}).get(profile["id"], pool) + if profile_capacity["status"] == "exhausted": + return "account_pool_exhausted" + if profile_capacity["in_flight"] >= pool["max_concurrency"]: + return "account_pool_concurrency_full" + if profile_capacity["status"] == "unknown": + if pool["unknown_capacity_policy"] == "block": + return "unknown_capacity_blocked" + if profile_capacity["in_flight"] >= 1: + return "unknown_capacity_trial_in_flight" + return None + + +def _profile_snapshot( + profile: dict[str, Any], reference: dict[str, str], binding: dict[str, Any] | None, +) -> dict[str, Any]: + frozen = json.loads(canonical_json(profile)) + return { + "reference": dict(reference), + "binding": dict(binding) if binding is not None else None, + "profile_id": profile["id"], + "profile_sha256": hashlib.sha256(canonical_json(profile).encode()).hexdigest(), + "profile": frozen, + } + + +def _policy_candidates( + role: str, + policy: dict[str, Any], + profiles: dict[str, dict[str, Any]], + bindings: dict[str, dict[str, Any]], + minimum_quality: str, + capacity: dict[str, dict[str, Any]], + implementer: dict[str, Any] | None, +) -> tuple[list[dict[str, Any]], list[dict[str, Any]], bool]: + references = list(policy["roles"].get(role, [])) + if role == "reviewer" and implementer is not None and policy.get( + "prefer_different_harness_for_review", False, + ): + references.sort( + key=lambda reference: ( + (_resolve_reference(reference, profiles, bindings)[0] or {}).get("harness") + == implementer["harness"] + ) + ) + eligible, excluded, has_static_candidate = [], [], False + seen = set() + for reference in references: + profile, binding, resolution_error = _resolve_reference( + reference, profiles, bindings, + ) + reason = resolution_error + if profile is not None and reason is None: + reason = _static_reason( + profile, + role, + minimum_quality, + policy["account_pools"].get(profile["account_pool_id"]), + implementer, + policy["require_different_model_for_review"], + ) + if reason is None: + has_static_candidate = True + reason = _capacity_reason(profile, capacity) + profile_id = profile["id"] if profile is not None else None + if reason is None and profile_id in seen: + reason = "duplicate_effective_profile" + if reason is not None: + excluded.append({ + "reference": dict(reference), + "profile_id": profile_id, + "reason": reason, + }) + continue + seen.add(profile_id) + eligible.append(_profile_snapshot(profile, reference, binding)) + return eligible, excluded, has_static_candidate + + +def resolve_routing( + task: dict[str, Any], + profile_registry: dict[str, Any], + policy: dict[str, Any], + *, + availability: dict[str, Any] | None = None, + profiles_sha256: str | None = None, + policy_sha256: str | None = None, +) -> dict[str, Any]: + """Resolve and freeze every model role used by a fixed workflow.""" + validate_task(task) + validate_profile_registry(profile_registry) + validate_policy(policy) + profiles = {profile["id"]: profile for profile in profile_registry["profiles"]} + bindings = profile_registry["bindings"] + capacity = _availability_snapshot(policy, profiles, availability) + minimum_quality = policy["task_classes"].get(task["task_class"]) + if minimum_quality is None: + raise PolicyDenied(f"policy does not authorize task class: {task['task_class']}") + + roles = list(WORKFLOW_ROLES[task["workflow"]]) + if task["lead"]["mode"] == "headless": + roles.append("lead") + overrides = task["routing"].get("overrides", {}) + unsupported_overrides = set(overrides) - set(roles) + if unsupported_overrides: + raise ContractError( + f"routing overrides are not roles in {task['workflow']}: " + f"{sorted(unsupported_overrides)}" + ) + missing_roles = [role for role in roles if not policy["roles"].get(role)] + if missing_roles: + raise PolicyDenied(f"policy has no candidates for required roles: {missing_roles}") + + selected_roles: dict[str, Any] = {} + implementer = None + council_models = set() + max_fallbacks = task["budget"]["max_fallbacks_per_step"] + for role in roles: + eligible, excluded, has_static_candidate = _policy_candidates( + role, + policy, + profiles, + bindings, + minimum_quality, + capacity, + implementer, + ) + if role in {"proposer_a", "proposer_b", "critic"}: + reused = [item for item in eligible if item["profile"]["model_id"].casefold() in council_models] + excluded.extend({"profile_id": item["profile_id"], "reason": "Council model identity already allocated"} for item in reused) + eligible = [item for item in eligible if item not in reused] + # Distinct model IDs are mandatory; qualified cross-family choices + # are a preference, not a requirement for a third provider. + families = {entry["selected"]["profile"]["model_family"].casefold() for key, entry in selected_roles.items() + if key in {"proposer_a", "proposer_b", "critic"}} + eligible.sort(key=lambda item: item["profile"]["model_family"].casefold() in families) + override = overrides.get(role) + selected = None + source = "automatic" + fallback_mode = "policy" + if override is not None: + source = "override" + fallback_mode = override.get("fallback", "none") + pinned = profiles.get(override["profile_id"]) + if pinned is None: + raise ProfileUnsupported( + f"pinned {role} profile does not exist: {override['profile_id']}" + ) + if role in {"proposer_a", "proposer_b", "critic"} and pinned["model_id"].casefold() in council_models: + raise ProfileUnsupported("Council override repeats a proposer/critic model identity") + static_reason = _static_reason( + pinned, + role, + minimum_quality, + policy["account_pools"].get(pinned["account_pool_id"]), + implementer, + policy["require_different_model_for_review"], + ) + if static_reason is not None: + raise ProfileUnsupported( + f"pinned {role} profile is unsupported: {static_reason}" + ) + capacity_reason = _capacity_reason(pinned, capacity) + if capacity_reason is None: + selected = _profile_snapshot( + pinned, {"kind": "profile", "id": pinned["id"]}, None, + ) + elif fallback_mode == "none": + raise CapabilityUnavailable( + f"pinned {role} profile is unavailable: {capacity_reason}" + ) + else: + excluded.insert(0, { + "reference": {"kind": "profile", "id": pinned["id"]}, + "profile_id": pinned["id"], + "reason": capacity_reason, + }) + if selected is None: + if not eligible: + if has_static_candidate: + raise CapabilityUnavailable( + f"no currently available profile for role: {role}" + ) + raise PolicyDenied(f"no policy-eligible profile for role: {role}") + selected = eligible.pop(0) + fallbacks = [] + if fallback_mode == "policy": + fallbacks = [ + candidate for candidate in eligible + if candidate["profile_id"] != selected["profile_id"] + ][:max_fallbacks] + selected_roles[role] = { + "source": source, + "fallback_mode": fallback_mode, + "selected": selected, + "fallbacks": fallbacks, + "excluded": excluded, + } + if role == "implementer": + implementer = selected["profile"] + if role in {"proposer_a", "proposer_b", "critic"}: + council_models.update(item["profile"]["model_id"].casefold() for item in + [selected_roles[role]["selected"], *selected_roles[role]["fallbacks"]]) + + return { + "schema_version": 1, + "policy": { + "id": policy["id"], + "version": policy["version"], + "sha256": policy_sha256 or hashlib.sha256(canonical_json(policy).encode()).hexdigest(), + }, + "profile_registry": { + "schema_version": profile_registry["schema_version"], + "sha256": profiles_sha256 + or hashlib.sha256(canonical_json(profile_registry).encode()).hexdigest(), + }, + "task_class": task["task_class"], + "workflow": task["workflow"], + "capacity": capacity, + "roles": selected_roles, + } + + +def load_routing( + task: dict[str, Any], + profiles_payload: bytes | str, + policy_payload: bytes | str, + *, + availability: dict[str, Any] | None = None, +) -> dict[str, Any]: + """Strictly decode profile/policy files and freeze hashes with selection.""" + profile_registry, profiles_sha256 = _strict_json(profiles_payload, "profiles file") + policy, policy_sha256 = _strict_json(policy_payload, "policy file") + return resolve_routing( + task, + profile_registry, + policy, + availability=availability, + profiles_sha256=profiles_sha256, + policy_sha256=policy_sha256, + ) + + +def capacity_with_live_reservations( + policy_payload: bytes | str, + in_flight: dict[str, int], +) -> dict[str, dict[str, Any]]: + """Bind transactionally observed local reservations to one policy snapshot.""" + policy, _ = _strict_json(policy_payload, "policy file") + validate_policy(policy) + if (not isinstance(in_flight, dict) + or not all( + isinstance(pool_id, str) and pool_id + and type(count) is int and count >= 0 + for pool_id, count in in_flight.items() + )): + raise ContractError("live capacity reservations are invalid") + return { + pool_id: { + "status": "unknown", + "in_flight": in_flight.get(pool_id, 0), + } + for pool_id in policy["account_pools"] + } + + +def capacity_with_saved_observations( + profiles_payload: bytes | str, + policy_payload: bytes | str, + snapshot: Callable[..., dict[str, Any]], +) -> dict[str, dict[str, Any]]: + """Build per-profile capacity evidence from the shared persisted ledger.""" + registry, _ = _strict_json(profiles_payload, "profiles file") + policy, _ = _strict_json(policy_payload, "policy file") + validate_profile_registry(registry) + validate_policy(policy) + if not callable(snapshot): + raise ContractError("capacity snapshot provider is invalid") + result: dict[str, dict[str, Any]] = {} + for pool_id in policy["account_pools"]: + root = snapshot(pool_id, target=None) + profile_evidence = {} + for profile in registry["profiles"]: + if profile["account_pool_id"] != pool_id: + continue + target = { + field: profile[field] + for field in ("harness", "model_family", "model_id") + } + profile_evidence[profile["id"]] = snapshot(pool_id, target=target) + result[pool_id] = {**root, "profiles": profile_evidence} + return result diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py new file mode 100644 index 0000000..05c43f7 --- /dev/null +++ b/plugin/core/src/devsquad/service.py @@ -0,0 +1,2228 @@ +"""Durable M2 application operations shared by CLI and later MCP surfaces.""" +from __future__ import annotations + +import getpass +import hashlib +import json +import os +from datetime import datetime, timezone +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile +import threading +from typing import Any + +from .codex_lead_worker import freeze_codex_lead +from .codex_review_worker import freeze_codex_reviewer +from .claude_delivery_worker import freeze_claude_implementer +from .contracts import ( + BudgetExhausted, + CapabilityUnavailable, + ContractError, + ProfileUnsupported, +) +from .decision import ( + apply_decision_response, + build_decision_request, + decision_fallback, + validate_decision_policy, +) +from .reports import ( + build_early_terminal_reports, + build_terminal_reports, + project_branch_review_history, + validate_saved_review_handoff, +) +from .router import capacity_with_saved_observations, load_routing +from .store import ( + ConflictError, + HandoffClaim, + HandoffSnapshot, + Store, + TERMINAL_STATES, + canonical_json, + request_hash, +) +from .validation import validate_task +from .workflows import ( + apply_lead_disposition, + require_check_integrity, + decode_headless_lead_evidence, + review_mode, + validate_branch_review_handoff, + validate_handoff_decision_evidence, + validate_review_document, +) +from .workspaces import ( + assert_clean_inputs, + committed_regular_file, + prepare_check_workspace, + prepare_delivery_workspace, + prepare_review_workspace, + repo_relative_config, + resolve_commit, +) + + +class Service: + def __init__(self, runtime: Path): + self.runtime = runtime.resolve() + self.runtime.mkdir(parents=True, exist_ok=True) + self.database = self.runtime / "state.sqlite3" + self.artifacts = self.runtime / "artifacts" + + def _store(self) -> Store: + return Store(self.database, self.artifacts) + + def normal_entry_bindings(self, workflow: str, *, pinned_roles: tuple[str, ...] = ()) -> dict[str, Any]: + """Read policy-matched approved incumbents before native discovery.""" + from .lifecycle import profile_fingerprint, profile_template_violation + from .task_entry import NORMAL_ALIASES, NORMAL_POLICY + + if workflow not in {"branch-review", "issue-delivery"}: + raise ContractError("normal entry workflow is unsupported") + task_class = "managed-review" if workflow == "branch-review" else "managed-fix" + roles = ("reviewer",) if workflow == "branch-review" else ("implementer", "reviewer") + store = self._store() + try: + store.connection.execute("BEGIN") + result = {} + for role in roles: + if role in pinned_roles: + continue + record = store.profile_binding(NORMAL_ALIASES[role]) + if record is None or record["template"]["policy"] != NORMAL_POLICY: + continue + template = record["template"] + profile = record["profile"] + if (profile_fingerprint(profile) != record["profile_sha256"] + or hashlib.sha256(canonical_json(template).encode()).hexdigest() != record["template_sha256"] + or profile_template_violation(profile, template) is not None + or task_class not in template["allowed_task_classes"] + or profile["quality_status"] != "proven"): + raise ContractError("normal alias incumbent is not approved for this task") + if record["qualification_id"] is not None: + store._require_current_qualification(record["qualification_id"], now=datetime.now(timezone.utc)) + result[role] = record + store.connection.execute("COMMIT") + return result + finally: + store.close() + + def _preparation_failure_artifacts( + self, + store: Store, + run_id: str, + task: dict[str, Any], + snapshot: dict[str, Any] | None, + error: dict[str, Any], + ) -> list[dict[str, Any]] | None: + if task.get("workflow") == "council-decision": + from .council_runtime import prepared_reports + return prepared_reports(store, run_id, snapshot, "failed", error=error) + if task.get("workflow") not in {"branch-review", "issue-delivery"}: + return None + reports = build_early_terminal_reports( + run_id=run_id, + state="failed", + task=task, + snapshot=snapshot, + run_artifacts=[], + events=store.events_for_run(run_id), + completed_at=datetime.now(timezone.utc).isoformat(), + phase="preparing", + error=error, + ) + prepared = [] + for name in sorted(reports): + path, digest, size = store.finalize_artifact(run_id, name, reports[name]) + prepared.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + return prepared + + @staticmethod + def _saved_review_progress( + store: Store, + run_id: str, + snapshot: dict[str, Any], + handoff: HandoffSnapshot | None, + ) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + """Project every persisted attempt and completed disposition in DB order.""" + history = store.branch_review_history(run_id) + _, dispositions = project_branch_review_history(history, snapshot) + frozen_handoffs = { + item["handoff_id"]: { + "handoff_id": item["handoff_id"], + "packet": item["packet"], + "packet_sha256": item["packet_sha256"], + } + for item in history + } + if handoff is not None: + frozen_handoffs[handoff.handoff_id] = { + "handoff_id": handoff.handoff_id, + "packet": handoff.packet, + "packet_sha256": handoff.packet_sha256, + } + packets = [item["packet"] for item in history] + if (handoff is not None and not any( + packet.get("attempt_id") == handoff.packet.get("attempt_id") + for packet in packets)): + packets.append(handoff.packet) + packets_by_attempt = { + packet["attempt_id"]: validate_saved_review_handoff(packet, snapshot) + for packet in packets + } + artifacts = store.artifacts_for_run(run_id) + artifacts_by_name = {artifact["name"]: artifact for artifact in artifacts} + failed_by_id = { + attempt["id"]: attempt + for attempt in Service._failed_fallback_attempts( + store, run_id, snapshot, + ) + } + attempts = [] + for attempt in store.attempts_for_run(run_id): + if attempt["id"] in failed_by_id: + attempts.append(failed_by_id[attempt["id"]]) + continue + if attempt.get("role") == "implementer": + iteration = next(( + item for item in snapshot.get("delivery_iterations", []) + if item.get("candidate", {}).get("implementation_artifact") + == f"implementation-attempt-{attempt['id']}.json" + ), None) + if iteration is not None: + attempts.append({ + "id": attempt["id"], + "status": "succeeded", + **iteration["implementation"]["attempt"], + "summary": iteration["implementation"].get("summary"), + "candidate": iteration["candidate"], + }) + elif attempt.get("role") == "reviewer": + packet = packets_by_attempt.get(attempt["id"]) + if packet is not None: + attempts.append({ + "id": attempt["id"], + "status": "succeeded", + **packet["attempt"], + "review": packet["review"], + "checks": packet["checks"], + "evaluation": packet["evaluation"], + }) + elif attempt.get("role") == "lead" and attempt.get("status") == "finished": + artifact = artifacts_by_name.get( + f"lead-attempt-{attempt['id']}.json" + ) + if artifact is None: + raise ConflictError("saved headless lead evidence is missing") + content = Path(artifact["path"]).read_bytes() + try: + raw = json.loads(content) + frozen_handoff = frozen_handoffs[raw["handoff_id"]] + except (KeyError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError( + "saved headless lead evidence is invalid" + ) from exc + evidence = decode_headless_lead_evidence( + content, snapshot, frozen_handoff, + ) + attempts.append({ + "id": attempt["id"], + "status": "succeeded", + **evidence["attempt"], + }) + return attempts, dispositions + + def _paused_review_terminal_artifacts( + self, + store: Store, + run_id: str, + snapshot: dict[str, Any], + handoff: HandoffSnapshot | None, + *, + state: str, + phase: str, + error: dict[str, Any] | None, + ) -> list[dict[str, Any]]: + """Materialize complete M3 reports before a paused run terminalizes.""" + if snapshot.get("task", {}).get("workflow") == "council-decision": + from .council_runtime import prepared_reports + return prepared_reports(store, run_id, snapshot, state, error=error) + attempts, dispositions = self._saved_review_progress( + store, run_id, snapshot, handoff, + ) + artifacts = store.artifacts_for_run(run_id) + reports = build_early_terminal_reports( + run_id=run_id, + state=state, + task=snapshot["task"], + snapshot=snapshot, + run_artifacts=artifacts, + events=store.events_for_run(run_id), + completed_at=datetime.now(timezone.utc).isoformat(), + phase=phase, + error=error, + prior_attempts=attempts, + prior_dispositions=dispositions, + ) + prepared = [] + for name in sorted(reports): + path, digest, size = store.finalize_artifact( + run_id, name, reports[name], + ) + prepared.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + return prepared + + def fail_budget_exhausted( + self, + run_id: str, + expected_version: int, + ) -> dict[str, Any]: + """Terminalize a queued branch review that cannot launch another worker.""" + store = self._store() + try: + run = store.run(run_id) + snapshot = self._review_snapshot(run) + if snapshot.get("task", {}).get("workflow") == "council-decision": + from .council_runtime import prepared_reports + error = {"error": "BUDGET_EXHAUSTED", "message": "Council run budget is exhausted"} + artifacts = prepared_reports(store, run_id, snapshot, "failed", error=error) + version = store.fail_queued_budget(run_id, expected_version, error, artifacts) + return {"run_id": run_id, "state": "failed", "version": version} + if snapshot.get("task", {}).get("workflow") not in { + "branch-review", "issue-delivery", + }: + raise ConflictError("queued budget failure is not a review workflow") + error = { + "error": "BUDGET_EXHAUSTED", + "message": "run wall-time budget is exhausted", + } + history = store.branch_review_history(run_id) + if history: + run_artifacts = store.artifacts_for_run(run_id) + reports = build_terminal_reports( + run_id=run_id, + state="failed", + snapshot=snapshot, + history=history, + run_artifacts=run_artifacts, + events=store.events_for_run(run_id), + completed_at=datetime.now(timezone.utc).isoformat(), + error=error, + headless_leads=self._headless_leads_for_history( + snapshot, history, run_artifacts, + ), + failed_attempts=self._failed_fallback_attempts( + store, run_id, snapshot, + ), + ) + terminal_artifacts = [] + for name in sorted(reports): + path, digest, size = store.finalize_artifact( + run_id, name, reports[name], + ) + terminal_artifacts.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + else: + terminal_artifacts = self._paused_review_terminal_artifacts( + store, + run_id, + snapshot, + store.handoff_snapshot(run_id), + state="failed", + phase="budget", + error=error, + ) + version = store.fail_queued_budget( + run_id, expected_version, error, terminal_artifacts, + ) + return {"run_id": run_id, "state": "failed", "version": version} + finally: + store.close() + + def _freeze_package(self) -> tuple[Path, str]: + source = Path(__file__).resolve().parent + digest = hashlib.sha256() + members = sorted(p for p in source.rglob("*") if p.is_file() and "__pycache__" not in p.parts and p.suffix != ".pyc") + for path in members: + relative = path.relative_to(source) + digest.update(str(relative).encode() + b"\0" + path.read_bytes()) + value = digest.hexdigest() + destination = self.runtime / "packages" / value / "devsquad" + if not destination.exists(): + destination.parent.parent.mkdir(parents=True, exist_ok=True) + temporary = Path(tempfile.mkdtemp(prefix=f".{value}.", dir=destination.parent.parent)) + try: + shutil.copytree(source, temporary / "devsquad", ignore=shutil.ignore_patterns("__pycache__", "*.pyc")) + for copied in (temporary/"devsquad").rglob("*"): + if copied.is_file(): + with copied.open("rb") as stream: os.fsync(stream.fileno()) + try: + os.rename(temporary, destination.parent) + except OSError: + if not destination.exists(): raise + finally: + shutil.rmtree(temporary, ignore_errors=True) + if self._package_digest(destination.parent)!=value: + raise ConflictError("frozen package cache is corrupt") + for directory_path in (destination,destination.parent,destination.parent.parent): + descriptor=os.open(directory_path,os.O_RDONLY) + try: os.fsync(descriptor) + finally: os.close(descriptor) + return destination.parent, value + + @staticmethod + def _package_digest(package: Path) -> str: + source=package/"devsquad"; digest=hashlib.sha256() + for path in sorted(p for p in source.rglob("*") if p.is_file() and "__pycache__" not in p.parts and p.suffix != ".pyc"): + relative=path.relative_to(source); digest.update(str(relative).encode()+b"\0"+path.read_bytes()) + return digest.hexdigest() + + def _verified_package(self, run: dict[str, Any]) -> tuple[Path,str]: + if not run.get("package_path") or not run.get("package_digest"): + raise ConflictError("run has no pinned package") + package=Path(run["package_path"]).resolve() + expected_root=(self.runtime/"packages").resolve() + if expected_root not in package.parents or self._package_digest(package)!=run["package_digest"]: + raise ConflictError("pinned package is missing or corrupt") + return package,run["package_digest"] + + @staticmethod + def _claim_payload(claim: HandoffClaim) -> dict[str, Any]: + return { + "schema_version": 1, + "run_id": claim.run_id, + "handoff_id": claim.handoff_id, + "owner": claim.owner_id, + "fencing_token": claim.fencing_token, + "expires_at": claim.expires_at, + "run_version": claim.run_version, + } + + @staticmethod + def _decode_claim(value: dict[str, Any]) -> HandoffClaim: + fields = { + "schema_version", "run_id", "handoff_id", "owner", + "fencing_token", "expires_at", "run_version", + } + if not isinstance(value, dict) or set(value) != fields: + raise ContractError("handoff claim fields are invalid") + if value["schema_version"] != 1 or type(value["schema_version"]) is not int: + raise ContractError("handoff claim schema_version is invalid") + for field in ("run_id", "handoff_id", "owner", "expires_at"): + if not isinstance(value[field], str) or not value[field]: + raise ContractError(f"handoff claim {field} is invalid") + for field in ("fencing_token", "run_version"): + if type(value[field]) is not int or value[field] < 1: + raise ContractError(f"handoff claim {field} is invalid") + try: + expires_at = datetime.fromisoformat(value["expires_at"]) + except ValueError as exc: + raise ContractError("handoff claim expires_at is invalid") from exc + if expires_at.tzinfo is None or expires_at.utcoffset() is None: + raise ContractError("handoff claim expires_at is invalid") + return HandoffClaim( + run_id=value["run_id"], + handoff_id=value["handoff_id"], + owner_id=value["owner"], + fencing_token=value["fencing_token"], + expires_at=value["expires_at"], + run_version=value["run_version"], + action="presented", + ) + + @staticmethod + def _decision_fixture_response( + request: dict[str, Any], fixture: dict[str, Any], + ) -> dict[str, Any]: + if (not isinstance(fixture, dict) + or set(fixture) != {"rankings", "confidence", "elapsed_ms"} + or not isinstance(fixture["rankings"], dict) + or isinstance(fixture["confidence"], bool) + or not isinstance(fixture["confidence"], (int, float)) + or type(fixture["elapsed_ms"]) is not int + or fixture["elapsed_ms"] < 0): + raise ContractError("internal decision fixture is invalid") + if set(fixture["rankings"]) - set(request["candidates"]): + raise ContractError("internal decision fixture role is invalid") + recommendations = {} + for role, candidates in request["candidates"].items(): + ranking = fixture["rankings"].get(role, candidates) + if not isinstance(ranking, list): + raise ContractError("internal decision fixture ranking is invalid") + denominator = sum(range(1, len(ranking) + 1)) + recommendations[role] = { + "ranking": list(ranking), + "probabilities": { + profile_id: weight / denominator + for profile_id, weight in zip( + ranking, range(len(ranking), 0, -1), + ) + }, + "confidence": fixture["confidence"], + "abstain_reason": None, + } + return { + "schema_version": 1, + "request_sha256": hashlib.sha256( + canonical_json(request).encode(), + ).hexdigest(), + "adapter": request["adapter"], + "language": request["language"], + "truncation": {"occurred": False, "detail": None}, + "recommendations": recommendations, + "usage": { + "source": "fake", "billable_requests": 1, + "input_tokens": 10, "output_tokens": 2, "cost_usd": 0.0, + }, + "elapsed_ms": fixture["elapsed_ms"], + } + + def _apply_optional_decision_helper( + self, + store: Store, + run_id: str, + fencing_token: int, + task: dict[str, Any], + routing: dict[str, Any], + policy: dict[str, Any], + base_oid: str, + target_oid: str, + fixture: dict[str, Any] | None, + ) -> tuple[dict[str, Any], dict[str, Any] | None]: + config = validate_decision_policy(policy.get("decision_helper")) + if config["mode"] == "off": + return routing, None + evidence_payload = canonical_json({ + "base_oid": base_oid, + "target_oid": target_oid, + "goal": task["goal"], + "task_class": task["task_class"], + "acceptance": task["acceptance"], + "checks": task["checks"], + "scope": task["scope"], + }) + try: + request = build_decision_request( + task, routing, policy, evidence_payload, + ) + except ContractError: + return ( + decision_fallback( + routing, config["mode"], "request_invalid", None, + ), + None, + ) + if request is None: # Defensive: enabled modes always build a request. + return routing, None + owner_id = f"preparation:{run_id}:{fencing_token}" + observation = store.claim_decision_observation( + run_id, config["mode"], request, owner_id, + ) + response = observation["response"] + if observation["action"] == "claimed": + if fixture is None: + store.finish_decision_without_call( + observation["cache_key"], owner_id, "unavailable", + "configured decision adapter is unavailable", + ) + observation = store.decision_observation( + run_id, request["purpose"]["id"], + ) + else: + store.launch_decision_call(observation["cache_key"], owner_id) + completed = store.complete_decision_call( + observation["cache_key"], owner_id, + self._decision_fixture_response(request, fixture), config, + ) + response = completed["response"] + observation = store.decision_observation( + run_id, request["purpose"]["id"], + ) + if (observation is not None + and observation["status"] in {"succeeded", "abstained"} + and response is not None): + routed = apply_decision_response(routing, request, response, config) + else: + status = ( + observation["status"] if observation is not None + else "observation_unavailable" + ) + routed = decision_fallback( + routing, config["mode"], status, + hashlib.sha256(canonical_json(request).encode()).hexdigest(), + ) + effect = routed["decision_helper"] + saved = store.record_decision_effect( + run_id, request["purpose"]["id"], observation["cache_key"], effect, + ) + return routed, saved + + @staticmethod + def _handoff_payload(snapshot: HandoffSnapshot, *, include_packet: bool) -> dict[str, Any]: + payload = { + "handoff_id": snapshot.handoff_id, + "sequence": snapshot.sequence, + "status": snapshot.status, + "packet_sha256": snapshot.packet_sha256, + "created_run_version": snapshot.created_run_version, + "submitted_run_version": snapshot.submitted_run_version, + } + if include_packet: + payload["packet"] = snapshot.packet + if snapshot.claim is not None: + payload["claimed_by"] = snapshot.claim.owner_id + payload["claim_expires_at"] = snapshot.claim.expires_at + else: + payload["claimed_by"] = None + payload["claim_expires_at"] = None + return payload + + def _resolve_snapshot( + self, + task: dict[str, Any], + internal_delay: float | None, + resolved_repo: Path | None = None, + *, + project_id: str | None = None, + run_id: str | None = None, + internal_review_fixture: dict[str, Any] | None = None, + internal_lead_fixture: dict[str, Any] | None = None, + internal_implementation_fixture: dict[str, Any] | None = None, + internal_decision_fixture: dict[str, Any] | None = None, + internal_council_fixture: dict[str, Any] | None = None, + preparation_fencing_token: int | None = None, + capacity_store: Store | None = None, + ) -> dict[str, Any]: + repo = resolved_repo or Path(task["project"]["repo_path"]).resolve(strict=True) + base_oid = resolve_commit(repo, task["project"]["base_ref"]) + target_oid = resolve_commit(repo, task["project"]["target_ref"]) + scope_paths = tuple(dict.fromkeys( + task["scope"]["read_paths"] + task["scope"]["write_paths"] + )) + embedded_routing = "profiles" in task["routing"] + config_paths = ( + {} + if embedded_routing + else { + label: repo_relative_config(repo, task["routing"][label], label) + for label in ("profiles_file", "policy_file") + } + ) + if internal_delay is None: + assert_clean_inputs(repo, scope_paths, config_paths.values()) + configs = {} + config_payloads = {} + if embedded_routing: + for label, routing_label in ( + ("profiles_file", "profiles"), ("policy_file", "policy"), + ): + data = canonical_json(task["routing"][routing_label]).encode() + config_payloads[label] = data + configs[label] = { + "path": f"embedded://routing/{routing_label}", + "sha256": hashlib.sha256(data).hexdigest(), + } + else: + for label, relative_path in config_paths.items(): + path = repo / relative_path + data = ( + committed_regular_file(repo, target_oid, relative_path) + if internal_delay is None else path.read_bytes() + ) + config_payloads[label] = data + configs[label] = { + "path": str(path), "sha256": hashlib.sha256(data).hexdigest(), + } + snapshot = { + "task": task, + "base_oid": base_oid, + "target_oid": target_oid, + "configs": configs, + } + if internal_delay is not None: + if internal_delay < 0 or internal_delay > 60: raise ContractError("internal fake delay is invalid") + snapshot["internal_fake_delay"] = internal_delay + else: + if capacity_store is None: + raise ContractError("public preflight requires shared capacity state") + effective_registry = capacity_store.effective_profile_registry( + config_payloads["profiles_file"], + config_payloads["policy_file"], + ) + snapshot["routing"] = load_routing( + task, + effective_registry["profiles_payload"], + config_payloads["policy_file"], + availability=capacity_with_saved_observations( + effective_registry["profiles_payload"], + config_payloads["policy_file"], + capacity_store.capacity_snapshot, + ), + ) + snapshot["routing"]["profile_registry"].update({ + "source_sha256": effective_registry["source_sha256"], + "lifecycle_bindings": effective_registry[ + "lifecycle_bindings" + ], + }) + if project_id is None or run_id is None: + raise ContractError("public preflight requires run-owned workspace identity") + if preparation_fencing_token is None: + raise ContractError("public preflight requires a preparation fence") + policy_document = json.loads(config_payloads["policy_file"]) + snapshot["routing"], decision_observation = ( + self._apply_optional_decision_helper( + capacity_store, + run_id, + preparation_fencing_token, + task, + snapshot["routing"], + policy_document, + base_oid, + target_oid, + internal_decision_fixture, + ) + ) + if decision_observation is not None: + snapshot["decision_observation"] = decision_observation + if task["workflow"] in {"branch-review", "council-decision"}: + snapshot["workspace"] = prepare_review_workspace( + repo, + self.runtime, + project_id, + run_id, + base_oid, + target_oid, + scope_paths, + required_clean_paths=config_paths.values(), + ) + snapshot["check_workspace"] = prepare_check_workspace( + repo, + self.runtime, + project_id, + run_id, + target_oid, + scope_paths, + required_clean_paths=config_paths.values(), + ) + if task["workflow"] == "council-decision": + from .council_runtime import prepare + prepare(snapshot, store=capacity_store, run_id=run_id, runtime=self.runtime, + fixture=internal_council_fixture) + else: + snapshot["delivery_workspace"] = prepare_delivery_workspace( + repo, + self.runtime, + project_id, + run_id, + target_oid, + task["scope"]["read_paths"], + task["scope"]["write_paths"], + required_clean_paths=config_paths.values(), + ) + fixture_fields = ( + {"writes", "delay_seconds"}, + {"writes", "delay_seconds", "fail_profile_ids"}, + ) + def valid_implementation_fixture(value: Any) -> bool: + return ( + isinstance(value, dict) + and set(value) in fixture_fields + and ( + "fail_profile_ids" not in value + or isinstance(value["fail_profile_ids"], list) + and all( + isinstance(item, str) and item + for item in value["fail_profile_ids"] + ) + ) + ) + + valid_fixture = valid_implementation_fixture( + internal_implementation_fixture + ) + if (isinstance(internal_implementation_fixture, dict) + and set(internal_implementation_fixture) == {"iterations"}): + fixtures = internal_implementation_fixture["iterations"] + valid_fixture = ( + isinstance(fixtures, list) + and 1 <= len(fixtures) + <= task["budget"]["max_revisions"] + 1 + and all( + valid_implementation_fixture(item) + for item in fixtures + ) + ) + if (internal_implementation_fixture is not None + and not valid_fixture): + raise ContractError("internal implementation fixture is invalid") + if internal_implementation_fixture is not None: + snapshot["internal_implementation_fixture"] = json.loads( + canonical_json(internal_implementation_fixture) + ) + if internal_review_fixture is not None: + if (not isinstance(internal_review_fixture, dict) + or set(internal_review_fixture) + != {"verdict", "summary", "findings"}): + raise ContractError("internal review fixture fields are invalid") + if task["workflow"] == "branch-review": + fixture_document = { + "schema_version": 1, + "candidate_sha256": snapshot["workspace"]["candidate_sha256"], + "base_oid": base_oid, + "target_oid": target_oid, + "review_mode": review_mode(task), + **internal_review_fixture, + } + snapshot["internal_review_fixture"] = validate_review_document( + fixture_document, task, snapshot["workspace"], + ) + else: + snapshot["pending_review_fixture"] = json.loads( + canonical_json(internal_review_fixture) + ) + if internal_lead_fixture is not None: + if (task["lead"]["mode"] != "headless" + or not isinstance(internal_lead_fixture, dict) + or set(internal_lead_fixture) != {"disposition", "reason"} + or internal_lead_fixture["disposition"] + not in {"accept", "revise", "reject"} + or not isinstance(internal_lead_fixture["reason"], str)): + raise ContractError("internal lead fixture is invalid") + snapshot["internal_lead_fixture"] = json.loads( + canonical_json(internal_lead_fixture) + ) + return snapshot + + def _continue_preparation( + self, + store: Store, + run_id: str, + fencing_token: int, + submitted: dict[str, Any], + ) -> tuple[tuple[int, Path, str] | None, dict[str, Any] | None]: + snapshot = None + supersedes_run_id = submitted.get("supersedes_run_id") + validated_supersedes_run_id = None + try: + task = submitted["task"] + internal_delay = submitted.get("_internal_fake_delay") + internal_review_fixture = submitted.get("_internal_review_fixture") + internal_lead_fixture = submitted.get("_internal_lead_fixture") + internal_implementation_fixture = submitted.get( + "_internal_implementation_fixture" + ) + internal_decision_fixture = submitted.get( + "_internal_decision_fixture" + ) + internal_council_fixture = submitted.get("_internal_council_fixture") + store.validate_predecessor(run_id, fencing_token, supersedes_run_id) + validated_supersedes_run_id = supersedes_run_id + validate_task(task, require_existing_repo=True) + minimum_headless_invocations = ( + 4 if task["workflow"] == "council-decision" else 3 if task["workflow"] == "issue-delivery" else 2 + ) + if (task["lead"]["mode"] == "headless" + and task["budget"]["max_worker_invocations"] + < minimum_headless_invocations): + raise ContractError( + "headless workflow has insufficient worker invocations" + ) + worktree = store.preparation_worktree( + run_id, fencing_token, Path(task["project"]["repo_path"]), + ) + project_id = store.run(run_id)["project_id"] + snapshot = self._resolve_snapshot( + task, + internal_delay, + worktree, + project_id=project_id, + run_id=run_id, + internal_review_fixture=internal_review_fixture, + internal_lead_fixture=internal_lead_fixture, + internal_implementation_fixture=internal_implementation_fixture, + internal_decision_fixture=internal_decision_fixture, + internal_council_fixture=internal_council_fixture, + preparation_fencing_token=fencing_token, + capacity_store=store, + ) + if (task["workflow"] == "issue-delivery" + and internal_delay is None + and internal_implementation_fixture is None): + implementer_route = snapshot["routing"]["roles"]["implementer"] + implementer_candidates = [ + implementer_route["selected"], + *implementer_route["fallbacks"], + ] + snapshot["implementation_adapters"] = { + candidate["profile_id"]: freeze_claude_implementer(candidate) + for candidate in implementer_candidates + } + snapshot["implementation_adapter"] = ( + snapshot["implementation_adapters"][ + implementer_route["selected"]["profile_id"] + ] + ) + if ((task["workflow"] == "branch-review" + or (task["workflow"] == "issue-delivery" + and internal_implementation_fixture is None)) + and internal_delay is None + and internal_review_fixture is None): + reviewer_route = snapshot["routing"]["roles"]["reviewer"] + reviewer_candidates = [ + reviewer_route["selected"], *reviewer_route["fallbacks"], + ] + snapshot["review_adapters"] = { + candidate["profile_id"]: freeze_codex_reviewer(candidate) + for candidate in reviewer_candidates + } + snapshot["review_adapter"] = snapshot["review_adapters"][ + reviewer_route["selected"]["profile_id"] + ] + if (internal_delay is None and task["lead"]["mode"] == "headless" and task["workflow"] != "council-decision" + and internal_lead_fixture is None): + lead_route = snapshot["routing"]["roles"]["lead"] + lead_candidates = [lead_route["selected"], *lead_route["fallbacks"]] + snapshot["lead_adapters"] = { + candidate["profile_id"]: freeze_codex_lead(candidate) + for candidate in lead_candidates + } + snapshot["lead_adapter"] = snapshot["lead_adapters"][ + lead_route["selected"]["profile_id"] + ] + if store.remaining_wall_seconds(run_id) == 0: + raise BudgetExhausted("run wall-time budget is exhausted in preflight") + package, digest = self._freeze_package() + if submitted.get("trial") is not None: + from .experiment_provenance import assignment_for + from .store import git_common_dir + trial = submitted["trial"] + snapshot["experiment_spec"] = trial["experiment"] + snapshot["experiment_assignment"] = assignment_for( + trial["experiment"], trial["case_id"], trial["arm"], + project_common_dir=str(git_common_dir(Path(task["project"]["repo_path"]))), + ) + version = store.complete_preparation( + run_id, + fencing_token, + snapshot, + package_path=str(package), + package_digest=digest, + supersedes_run_id=supersedes_run_id, + worktree_path=( + snapshot.get("workspace") or snapshot.get("delivery_workspace") or {} + ).get("path"), + ) + return (version, package, digest), None + except (BudgetExhausted, CapabilityUnavailable, ProfileUnsupported) as exc: + error = {"error": exc.code, "message": str(exc)} + terminal_artifacts = self._preparation_failure_artifacts( + store, run_id, task, snapshot, error, + ) + store.fail_preparation( + run_id, + fencing_token, + error, + mutable_snapshot=snapshot, + supersedes_run_id=validated_supersedes_run_id, + terminal_artifacts=terminal_artifacts, + ) + return None, error + except Exception as exc: + error = {"error": "PREPARATION_FAILED", "message": str(exc)} + try: + terminal_artifacts = self._preparation_failure_artifacts( + store, run_id, task, snapshot, error, + ) + # Preserve validated lineage through unrelated failures; a + # rejected predecessor remains only in submitted_request. + store.fail_preparation( + run_id, + fencing_token, + error, + mutable_snapshot=snapshot, + supersedes_run_id=validated_supersedes_run_id, + terminal_artifacts=terminal_artifacts, + ) + except ConflictError: + # Cancellation or another recovery owner may have fenced us. + raise exc + return None, error + + def start( + self, + task: dict[str, Any], + idempotency_key: str, + supersedes_run_id: str | None = None, + *, + trial: dict[str, Any] | None = None, + _internal_fake_delay: float | None = None, + _internal_review_fixture: dict[str, Any] | None = None, + _internal_lead_fixture: dict[str, Any] | None = None, + _internal_implementation_fixture: dict[str, Any] | None = None, + _internal_decision_fixture: dict[str, Any] | None = None, + _internal_council_fixture: dict[str, Any] | None = None, + ) -> dict[str, Any]: + validate_task(task, require_existing_repo=True) + if task["workflow"] == "council-decision" and any(value is not None for value in + (_internal_fake_delay, _internal_review_fixture, _internal_lead_fixture, _internal_implementation_fixture)): + raise ContractError("Council requires its explicit all-fixture seam or native roles") + if _internal_council_fixture is not None and task["workflow"] != "council-decision": + raise ContractError("Council fixture requires council-decision") + if trial is not None: + from .learning import validate_experiment + from .experiment_provenance import assignment_for + from .store import git_common_dir + if not isinstance(trial, dict) or set(trial) != {"experiment", "case_id", "arm"}: + raise ContractError("trial requires an explicit experiment, case and arm") + experiment = validate_experiment(trial["experiment"]) + if experiment["schema_version"] != 2: + raise ContractError("public trials require a v2 predeclared experiment") + if (experiment["budget"]["max_cases"] > 100 + or experiment["budget"]["max_worker_invocations"] > 1000 + or experiment["budget"]["wall_seconds"] > 3600): + raise ContractError("public trial exceeds the bounded controller limits") + if ((experiment["variable"]["role"] == "reviewer" and task["workflow"] != "branch-review") + or (experiment["variable"]["role"] == "implementer" and task["workflow"] != "issue-delivery")): + raise ContractError("trial role requires a frozen review candidate or implementation baseline") + common_dir = str(git_common_dir(Path(task["project"]["repo_path"]))) + assignment_for(experiment, trial["case_id"], trial["arm"], project_common_dir=common_dir) + trial = json.loads(canonical_json({**trial, "experiment": experiment})) + if _internal_fake_delay is not None and _internal_review_fixture is not None: + raise ContractError("internal lifecycle fixtures are mutually exclusive") + if _internal_lead_fixture is not None and _internal_fake_delay is not None: + raise ContractError("internal lifecycle fixtures are mutually exclusive") + if (_internal_implementation_fixture is not None + and (_internal_fake_delay is not None + or task["workflow"] != "issue-delivery")): + raise ContractError("internal implementation fixture requires issue-delivery") + if (_internal_decision_fixture is not None + and _internal_fake_delay is not None): + raise ContractError( + "internal decision fixture requires public preflight", + ) + submitted = {"task": task, "supersedes_run_id": supersedes_run_id} + if trial is not None: + submitted["trial"] = trial + if _internal_fake_delay is not None: + submitted["_internal_fake_delay"] = _internal_fake_delay + if _internal_review_fixture is not None: + submitted["_internal_review_fixture"] = _internal_review_fixture + if _internal_lead_fixture is not None: + submitted["_internal_lead_fixture"] = _internal_lead_fixture + if _internal_implementation_fixture is not None: + submitted["_internal_implementation_fixture"] = ( + _internal_implementation_fixture + ) + if _internal_decision_fixture is not None: + submitted["_internal_decision_fixture"] = _internal_decision_fixture + if _internal_council_fixture is not None: + submitted["_internal_council_fixture"] = _internal_council_fixture + store = self._store() + try: + claim = store.claim_start(Path(task["project"]["repo_path"]), idempotency_key, submitted, f"preflight:{os.getpid()}", objective_outcome=True) + if not claim.created: + return {"run_id": claim.run_id, "state": store.run(claim.run_id)["state"], "created": False} + launch, error = self._continue_preparation( + store, claim.run_id, claim.fencing_token or 0, submitted, + ) + finally: + store.close() + if launch is None: + return {"run_id": claim.run_id, "state": "failed", "created": True, "error": error} + version, package, digest = launch + self._spawn_daemon(claim.run_id, version, package, digest) + return {"run_id": claim.run_id, "state": "queued", "created": True} + + def trial_start(self, experiment: dict[str, Any], case_id: str, arm: str, + task: dict[str, Any], idempotency_key: str, **fixtures) -> dict[str, Any]: + """Explicit one-arm controller; no automatic dispatch or promotion.""" + return self.start(task, idempotency_key, + trial={"experiment": experiment, "case_id": case_id, "arm": arm}, + **fixtures) + + def _spawn_daemon(self, run_id: str, expected_version: int, package: Path, digest: str) -> int: + command = [sys.executable, "-P", "-m", "devsquad.detached", "--database", str(self.database), "--artifacts", str(self.artifacts), "--run-id", run_id, "--expected-version", str(expected_version), "--package-digest", digest] + # Claude's native saved-login lookup needs the login name and HOME. + # Keep this explicit: ambient API keys/provider overrides never cross + # the detached boundary, and package imports remain frozen. + environment = { + "PATH": os.environ.get("PATH", ""), "PYTHONPATH": str(package), + "HOME": str(Path.home()), + "USER": os.environ.get("USER") or getpass.getuser(), + } + log_dir=self.runtime/"private-logs"; log_dir.mkdir(parents=True,exist_ok=True) + with (log_dir/f"{run_id}.supervisor.log").open("ab",buffering=0) as diagnostic: + process = subprocess.Popen(command, cwd=self.runtime, env=environment, stdin=subprocess.DEVNULL, stdout=diagnostic, stderr=diagnostic, start_new_session=True, close_fds=True) + threading.Thread( + target=process.wait, + name=f"devsquad-reap-{run_id}", + daemon=True, + ).start() + return process.pid + + def capacity_observe(self, observation: dict[str, Any]) -> dict[str, Any]: + """Record one capacity observation and return its current pool view.""" + store = self._store() + try: + recorded = store.record_pool_observation(observation) + return { + "record": recorded, + "capacity": store.capacity_snapshot(observation.get("pool_id")), + } + finally: + store.close() + + def outcome_add(self, run_id: str, outcome: dict[str, Any]) -> dict[str, Any]: + store = self._store() + try: + return store.record_outcome(run_id, outcome) + finally: + store.close() + + def learning_report(self, project: str | Path) -> dict[str, Any]: + if not isinstance(project, (str, Path)): + raise ContractError("report project path is invalid") + store = self._store() + try: + return store.learning_report(Path(project)) + finally: + store.close() + + def policy_evaluate( + self, experiment: dict[str, Any], *, revision_id: str | None = None, + previous_evaluation_sha256: str | None = None, + ) -> dict[str, Any]: + """Evaluate and save one frozen learning experiment without promotion.""" + store = self._store() + try: + return store.evaluate_learning_experiment( + experiment, revision_id=revision_id, + previous_evaluation_sha256=previous_evaluation_sha256, + ) + finally: + store.close() + + @staticmethod + def _finalize_learning_file( + directory: Path, name: str, content: bytes, + ) -> dict[str, Any]: + digest = hashlib.sha256(content).hexdigest() + component = Path(name) + if component.name != name or not component.stem or not component.suffix: + raise ContractError("learning file name is invalid") + destination = directory / f"{component.stem}.{digest}{component.suffix}" + descriptor, temporary = tempfile.mkstemp(prefix=f".{name}.", dir=directory) + try: + with os.fdopen(descriptor, "wb") as stream: + stream.write(content) + stream.flush() + os.fsync(stream.fileno()) + try: + os.link(temporary, destination) + except FileExistsError: + if destination.read_bytes() != content: + raise ConflictError("content-addressed learning file is corrupt") + directory_fd = os.open(directory, os.O_RDONLY) + try: + os.fsync(directory_fd) + finally: + os.close(directory_fd) + finally: + if os.path.exists(temporary): + os.unlink(temporary) + return { + "path": str(destination), + "sha256": digest, + "byte_size": len(content), + } + + def learning_propose(self, project: str | Path) -> dict[str, Any]: + """Write a local review draft from saved evidence without policy mutation.""" + from .learning import ( + build_learning_proposal, + render_learning_proposal_markdown, + ) + + if not isinstance(project, (str, Path)): + raise ContractError("proposal project path is invalid") + store = self._store() + try: + inputs = store.learning_proposal_inputs(Path(project)) + finally: + store.close() + generated_at = inputs["report"]["generated_at"] + proposal = build_learning_proposal( + inputs["report"], inputs["experiment"], generated_at=generated_at, + ) + json_content = (canonical_json(proposal) + "\n").encode() + markdown_content = render_learning_proposal_markdown(proposal).encode() + directory = self.runtime / "learning" / "proposals" + directory.mkdir(parents=True, exist_ok=True) + return { + "proposal": proposal, + "artifacts": { + "json": self._finalize_learning_file( + directory, f"{proposal['proposal_id']}.json", json_content, + ), + "markdown": self._finalize_learning_file( + directory, f"{proposal['proposal_id']}.md", markdown_content, + ), + }, + } + + def profile_template_add(self, template: dict[str, Any]) -> dict[str, Any]: + store = self._store() + try: + return store.register_profile_template(template) + finally: + store.close() + + def profile_binding_bootstrap(self, request: dict[str, Any]) -> dict[str, Any]: + if not isinstance(request, dict) or set(request) != { + "template", "profile", "version", + }: + raise ContractError("profile binding bootstrap fields are invalid") + store = self._store() + try: + return store.bootstrap_profile_binding( + request["template"], request["profile"], version=request["version"], + ) + finally: + store.close() + + def profile_qualification_add( + self, qualification: dict[str, Any], + ) -> dict[str, Any]: + store = self._store() + try: + return store.record_profile_qualification(qualification) + finally: + store.close() + + def profile_binding_change(self, change: dict[str, Any]) -> dict[str, Any]: + store = self._store() + try: + result = store.change_profile_binding(change) + finally: + store.close() + return self._profile_binding_decision_artifacts(result) + + def profile_binding_fallback(self, change: dict[str, Any]) -> dict[str, Any]: + """Apply a catalog-proven fallback to a prior qualified binding.""" + store = self._store() + try: + result = store.fallback_unavailable_profile_binding(change) + finally: + store.close() + return self._profile_binding_decision_artifacts(result) + + def _profile_binding_decision_artifacts( + self, result: dict[str, Any], + ) -> dict[str, Any]: + from .lifecycle import render_binding_decision_markdown + + receipt = result["receipt"] + json_content = (canonical_json(receipt) + "\n").encode() + markdown_content = render_binding_decision_markdown(receipt).encode() + directory = self.runtime / "learning" / "decisions" + directory.mkdir(parents=True, exist_ok=True) + return { + **result, + "artifacts": { + "json": self._finalize_learning_file( + directory, f"{receipt['decision_id']}.json", json_content, + ), + "markdown": self._finalize_learning_file( + directory, f"{receipt['decision_id']}.md", markdown_content, + ), + }, + } + + def profile_binding_status(self, alias: str) -> dict[str, Any]: + store = self._store() + try: + binding = store.profile_binding(alias) + if binding is None: + raise ContractError("profile binding does not exist") + return { + "binding": binding, + "decisions": store.profile_binding_decisions(alias), + } + finally: + store.close() + + @staticmethod + def _status_capacity(store: Store, run: dict[str, Any]) -> dict[str, Any] | None: + try: + snapshot = json.loads(run["mutable_snapshot"]) + routing = snapshot["routing"] + roles = routing["roles"] + frozen = routing["capacity"] + except (KeyError, TypeError, json.JSONDecodeError): + return None + current = {} + for role in roles.values(): + for candidate in [role["selected"], *role.get("fallbacks", [])]: + profile = candidate["profile"] + profile_id = candidate["profile_id"] + if profile_id in current: + continue + target = { + field: profile[field] + for field in ("harness", "model_family", "model_id") + } + current[profile_id] = store.capacity_snapshot( + profile["account_pool_id"], target=target, + ) + return {"frozen": frozen, "current": current} + + def status(self, run_id: str) -> dict[str, Any]: + store = self._store() + try: + store.project_final_outcome(run_id) + run, attempt, handoff = store.status_snapshot(run_id) + active=attempt if attempt and attempt.get("status") in {"reserved","running","cancelling","ownership_ambiguous"} else None + if run["state"] == "blocked": + next_action = "recovery_file_required" + elif run["state"] == "awaiting_host" and run["phase"] is None: + try: + snapshot = self._review_snapshot(run) + headless = snapshot["task"]["lead"]["mode"] == "headless" + except (ConflictError, KeyError, TypeError): + headless = False + next_action = "continue_headless_lead" if headless else "claim_handoff" + elif run["state"] == "awaiting_host": + next_action = "handoff_submission_saved" + elif run["state"] == "queued" and run["phase"] is None: + try: + snapshot = self._review_snapshot(run) + next_action = ( + "resume_council_stage" + if snapshot.get("task", {}).get("workflow") == "council-decision" + else "resume_candidate_review" + if snapshot.get("task", {}).get("workflow") == "issue-delivery" + and isinstance(snapshot.get("candidate"), dict) + and (handoff is None or handoff.status != "open") + else None + ) + except ConflictError: + next_action = None + else: + next_action = None + result = { + "run_id": run_id, + "state": run["state"], + "phase": run["phase"], + "version": run["version"], + "active_attempt": { + key: active.get(key) + for key in ("id", "status", "pid", "pgid", "heartbeat_at") + } if active else None, + "handoff": self._handoff_payload(handoff, include_packet=False) if handoff else None, + "next_action": next_action, + "capacity": self._status_capacity(store, run), + } + decision_observations = store.decision_observations_for_run(run_id) + if decision_observations: + result["decision_helper"] = decision_observations + return result + finally: store.close() + + def resolve_run_id(self, run_id: str | None, project: Path) -> str: + """Only infer a run when this canonical Git project has one saved choice.""" + if run_id is not None: + return run_id + store = self._store() + try: + choices = store.runs_for_project(project) + finally: + store.close() + if not choices: + raise ContractError("No saved runs for this Git project. Start with squad review --base main or squad fix \"the bounded issue\".") + if len(choices) == 1: + return choices[0]["run_id"] + listed = "; ".join(f"{item['run_id']} ({item['state']})" for item in choices[:20]) + more = "; additional runs omitted" if len(choices) > 20 else "" + raise ConflictError(f"Multiple saved runs for this Git project; specify RUN: {listed}{more}") + + def handoff_view(self, run_id: str) -> dict[str, Any]: + """Read the current validated review and its portable human report.""" + from .reports import handoff_report_names + + store = self._store() + try: + run, _, handoff = store.status_snapshot(run_id) + if handoff is None or handoff.status != "open": + raise ConflictError("run has no current open handoff to inspect") + if self._review_snapshot(run).get("task", {}).get("workflow") == "council-decision": + return self.council_handoff_view(run_id) + packet = validate_saved_review_handoff(handoff.packet, self._review_snapshot(run)) + report = store.artifact_named(run_id, handoff_report_names(handoff.sequence)[1]) + if report is None: + raise ConflictError("saved handoff report is missing") + try: + content = Path(report["path"]).read_bytes() + except OSError as exc: + raise ConflictError("saved handoff report is missing") from exc + if (len(content) != report["byte_size"] + or hashlib.sha256(content).hexdigest() != report["sha256"]): + raise ConflictError("saved handoff report is corrupt") + return { + "version": run["version"], "packet_sha256": handoff.packet_sha256, + "candidate_sha256": packet["candidate_sha256"], + "review": packet["review"], "checks": packet["checks"], + "report": report, + "pending_finish": store.terminal_finish_decision(run_id, handoff.handoff_id), + } + finally: + store.close() + + def finish(self, run_id: str, disposition: str, reason: str, *, chosen: str | None = None, + supported_claims: list[str] | None = None, + discarded_alternatives: list[str] | None = None, + validation: str | None = None) -> dict[str, Any]: + """Guided terminal host disposition over the existing fenced handoff gates.""" + if disposition not in {"accept", "reject", "revise"}: + raise ContractError("finish disposition is invalid") + if not isinstance(reason, str) or not reason.strip() or len(reason) > 2000: + raise ContractError("finish requires a non-empty reason of at most 2000 characters") + store = self._store() + try: + run, _, handoff = store.status_snapshot(run_id) + if (run["state"] != "awaiting_host" or run["phase"] is not None + or handoff is None or handoff.status != "open"): + raise ConflictError("run has no current open handoff to finish; inspect squad status RUN") + snapshot = self._review_snapshot(run) + if snapshot.get("task", {}).get("workflow") == "council-decision": + if chosen is None or validation is None: + raise ContractError("Council finish requires explicit --choose and --validation; inspect squad status RUN") + return self.finish_council(run_id, disposition, reason, chosen=chosen, + supported_claims=supported_claims or [], discarded_alternatives=discarded_alternatives or [], + validation=validation) + if any(value is not None for value in (chosen, supported_claims, discarded_alternatives, validation)): + raise ContractError("Council choice flags cannot be used for a review or delivery handoff") + if snapshot.get("task", {}).get("lead", {}).get("mode") != "host": + raise ConflictError("headless lead owns this handoff; run squad resume RUN") + packet = validate_saved_review_handoff(handoff.packet, snapshot) + for reference in packet["artifacts"]: + artifact = store.artifact_named(run_id, reference["name"]) + if (artifact is None or artifact["id"] != reference["artifact_id"] + or artifact["sha256"] != reference["sha256"]): + raise ConflictError("current handoff evidence reference is invalid") + try: + content = Path(artifact["path"]).read_bytes() + except OSError as exc: + raise ConflictError("current handoff evidence is missing") from exc + if (len(content) != artifact["byte_size"] + or hashlib.sha256(content).hexdigest() != reference["sha256"]): + raise ConflictError("current handoff evidence is corrupt") + body = { + "schema_version": 1, + "submission_id": f"terminal-{handoff.handoff_id}", + "disposition": disposition, + "reason": reason.strip(), + "evidence_refs": [ + {"artifact_id": ref["artifact_id"], "sha256": ref["sha256"]} + for ref in packet["artifacts"] + ], + } + decision = {**body, "submission_hash": request_hash(body)} + # Invalid decisions do not acquire a lease. Claim and completion + # still independently recheck their version and evidence fences. + self._review_gate(store, run_id, handoff, snapshot, decision) + version = run["version"] + finally: + store.close() + claimed = self.handoff_claim( + run_id, version, "terminal-operator", _initial_only=True, + _terminal_decision=decision, + ) + return self.handoff_complete(run_id, claimed["claim"], decision) + + def events(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str, Any]: + store = self._store() + try: return store.events_page(run_id, after, limit) + finally: store.close() + + def result(self, run_id: str) -> dict[str, Any]: + store = self._store() + try: + store.project_final_outcome(run_id) + run, artifacts = store.result_snapshot(run_id) + if run["state"] not in TERMINAL_STATES: + return {"run_id":run_id,"ready":False,"state":run["state"],"artifacts":[]} + if not any(item["name"]=="result-receipt.json" for item in artifacts): + raise ConflictError("terminal result has no durable receipt") + for item in artifacts: + path=Path(item["path"]) + if not path.is_file() or hashlib.sha256(path.read_bytes()).hexdigest()!=item["sha256"]: + raise ConflictError("referenced result artifact is missing or corrupt") + return {"run_id":run_id,"ready":True,"state":run["state"],"artifacts":artifacts} + finally: store.close() + + def cancel(self, run_id: str) -> dict[str, Any]: + store = self._store() + try: + run = store.run(run_id) + if run["state"] == "queued" and run["phase"] is None: + try: + snapshot = self._review_snapshot(run) + except ConflictError: + snapshot = {} + if snapshot.get("task", {}).get("workflow") == "council-decision": + from .council_runtime import prepared_reports + artifacts = prepared_reports(store, run_id, snapshot, "cancelled") + version = store.cancel_council_queued(run_id, run["version"], artifacts) + return {"run_id": run_id, "state": "cancelled", "version": version} + if run["state"] == "queued" and run["phase"] == "preparing": + store.cancel_run_decision_observations(run_id) + version = store.cancel_preparing(run_id) + elif run["state"] == "queued" and run["phase"] == "launching": version = store.cancel_launching(run_id) + elif run["state"] == "queued" and run["phase"] is None: version = store.cancel_queued(run_id) + elif run["state"] == "cancelling" and run["phase"] == "recovery_cleanup": + from .supervisor import Supervisor + version = Supervisor(store).cancel_orphan(run_id) + elif run["state"] in {"running", "cancelling"}: version, _ = store.request_cancel(run_id) + elif run["state"] == "awaiting_host": + terminal_artifacts = None + try: + snapshot = self._review_snapshot(run) + handoff = store.handoff_snapshot(run_id) + if (snapshot.get("task", {}).get("workflow") + in {"branch-review", "issue-delivery", "council-decision"} + and handoff is not None): + terminal_artifacts = self._paused_review_terminal_artifacts( + store, + run_id, + snapshot, + handoff, + state="cancelled", + phase="awaiting_host", + error=None, + ) + except (KeyError, TypeError): + terminal_artifacts = None + version = store.cancel_host_wait( + run_id, terminal_artifacts=terminal_artifacts, + ) + elif run["state"] == "blocked": + from .supervisor import Supervisor + version = Supervisor(store).cancel_orphan(run_id) + elif run["state"] in TERMINAL_STATES: version = run["version"] + else: raise ConflictError("run requires recovery before cancellation") + return {"run_id": run_id, "state": store.run(run_id)["state"], "version": version} + finally: store.close() + + def handoff_claim( + self, + run_id: str, + expected_version: int, + owner: str, + prior_claim: dict[str, Any] | None = None, + *, + _initial_only: bool = False, + _terminal_decision: dict[str, Any] | None = None, + ) -> dict[str, Any]: + if type(expected_version) is not int or expected_version < 1: + raise ContractError("handoff expected version is invalid") + if not isinstance(owner, str) or not owner: + raise ContractError("handoff owner is required") + decoded = self._decode_claim(prior_claim) if prior_claim is not None else None + store = self._store() + try: + run = store.run(run_id) + handoff_before_claim = store.handoff_snapshot(run_id) + if (handoff_before_claim is not None + and handoff_before_claim.packet.get("workflow") + in {"branch-review", "issue-delivery", "council-decision"} + and self._review_snapshot(run)["task"]["lead"]["mode"] + == "headless"): + raise ConflictError( + "headless review does not accept a host claim" + ) + claim = store.claim_handoff( + run_id, expected_version, owner, decoded, initial_only=_initial_only, + terminal_decision=_terminal_decision, + ) + snapshot = store.handoff_snapshot(run_id) + if snapshot is None: # Defensive: claim_handoff just verified it. + raise ConflictError("claimed handoff is missing") + return { + "run_id": run_id, + "state": "awaiting_host", + "phase": None, + "version": claim.run_version, + "action": claim.action, + "claim": self._claim_payload(claim), + "handoff": self._handoff_payload(snapshot, include_packet=True), + } + finally: + store.close() + + @staticmethod + def _review_snapshot(run: dict[str, Any]) -> dict[str, Any]: + try: + snapshot = json.loads(run["mutable_snapshot"]) + except (KeyError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError("frozen branch review snapshot is invalid") from exc + if (not isinstance(snapshot, dict) + or canonical_json(snapshot) != run["mutable_snapshot"]): + raise ConflictError("frozen branch review snapshot is not canonical") + return snapshot + + @staticmethod + def _review_gate( + store: Store, + run_id: str, + handoff: HandoffSnapshot, + snapshot: dict[str, Any], + decision: dict[str, Any], + ) -> tuple[dict[str, Any], dict[str, Any]]: + packet = validate_saved_review_handoff(handoff.packet, snapshot) + validate_handoff_decision_evidence(decision, packet) + if decision.get("disposition") == "accept": + require_check_integrity(packet["checks"]) + from .workflows import require_independent_delivery_review + require_independent_delivery_review(snapshot, packet) + revisions_used = sum( + entry["decision"]["disposition"] == "revise" + for entry in store.branch_review_history(run_id) + if entry["sequence"] < handoff.sequence + ) + gate = apply_lead_disposition( + packet["evaluation"], + decision.get("disposition"), + revisions_used=revisions_used, + max_revisions=snapshot["task"]["budget"]["max_revisions"], + ) + return packet, gate + + @staticmethod + def _headless_leads_for_history( + snapshot: dict[str, Any], + history: list[dict[str, Any]], + run_artifacts: list[dict[str, Any]], + ) -> list[dict[str, Any]]: + if snapshot["task"]["lead"]["mode"] != "headless": + return [] + artifacts_by_name = { + artifact["name"]: artifact for artifact in run_artifacts + } + evidence = [] + for item in history: + submission_id = item["decision"]["submission_id"] + if not submission_id.startswith("headless-"): + raise ConflictError("headless lead submission identity is invalid") + attempt_id = submission_id[len("headless-"):] + artifact = artifacts_by_name.get(f"lead-attempt-{attempt_id}.json") + if artifact is None: + raise ConflictError("headless lead evidence artifact is missing") + frozen_handoff = { + "handoff_id": item["handoff_id"], + "packet": item["packet"], + "packet_sha256": item["packet_sha256"], + } + evidence.append(decode_headless_lead_evidence( + Path(artifact["path"]).read_bytes(), snapshot, frozen_handoff, + )) + return evidence + + @staticmethod + def _failed_fallback_attempts( + store: Store, + run_id: str, + snapshot: dict[str, Any], + ) -> list[dict[str, Any]]: + failures = [] + for attempt in store.attempts_for_run(run_id): + encoded = attempt.get("output_metadata") + if not encoded: + continue + try: + metadata = json.loads(encoded) + if not isinstance(metadata, dict) or "failure" not in metadata: + continue + error = metadata["failure"] + role = attempt["role"] + index = attempt["profile_index"] + routed = snapshot["routing"]["roles"][role] + candidates = [routed["selected"], *routed["fallbacks"]] + selected = candidates[index] + except (IndexError, KeyError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError("saved fallback attempt is invalid") from exc + if (not isinstance(error, dict) + or selected["profile_id"] != attempt["profile_id"]): + raise ConflictError("saved fallback attempt changed its profile") + failures.append({ + "id": attempt["id"], + "role": role, + "status": "failed", + "profile_index": index, + "selected_profile": selected, + "observed_identity": None, + "worker_invocations": 1, + "native_model_requests": None, + "usage": error.get("native_diagnostics", {}).get("usage", { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }), + **({"native_diagnostics": error["native_diagnostics"]} + if "native_diagnostics" in error else {}), + "error": error, + }) + return failures + + def _terminalize_branch_review( + self, + store: Store, + run_id: str, + handoff: HandoffSnapshot, + snapshot: dict[str, Any], + entry: dict[str, Any], + terminal_state: str, + error: dict[str, Any] | None, + ) -> dict[str, Any]: + history = store.branch_review_history(run_id) + evidence_ids = { + reference["artifact_id"] + for item in history + for reference in item["packet"]["artifacts"] + } + run_artifacts = store.artifacts_for_run(run_id) + if evidence_ids - {artifact["id"] for artifact in run_artifacts}: + raise ConflictError("branch review evidence artifact is missing") + expected_parent = (self.artifacts / run_id).resolve() + for artifact in run_artifacts: + path = Path(artifact["path"]) + if (path.resolve().parent != expected_parent + or not path.is_file() + or path.stat().st_size != artifact["byte_size"] + or hashlib.sha256(path.read_bytes()).hexdigest() + != artifact["sha256"]): + raise ConflictError("branch review artifact is missing or corrupt") + headless_leads = self._headless_leads_for_history( + snapshot, history, run_artifacts, + ) + reports = build_terminal_reports( + run_id=run_id, + state=terminal_state, + snapshot=snapshot, + history=history, + run_artifacts=run_artifacts, + events=store.events_for_run(run_id), + completed_at=datetime.now(timezone.utc).isoformat(), + error=error, + headless_leads=headless_leads, + failed_attempts=self._failed_fallback_attempts( + store, run_id, snapshot, + ), + ) + prepared = [] + for name in sorted(reports): + content = reports[name] + path, digest, size = store.finalize_artifact(run_id, name, content) + prepared.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + decision = entry["decision"] + outcome = store.complete_handoff_terminal( + run_id, + handoff.handoff_id, + decision["submission_id"], + decision["submission_hash"], + prepared, + terminal_state, + { + "handoff_id": handoff.handoff_id, + "submission_id": decision["submission_id"], + "submission_hash": decision["submission_hash"], + "disposition": decision["disposition"], + "receipt": "result-receipt.json", + "error": error, + }, + ) + return { + "action": "terminal", + "state": outcome["state"], + "version": outcome["version"], + "replayed_continuation": outcome["replayed"], + "launch": None, + } + + def _continue_branch_review_submission( + self, + store: Store, + run_id: str, + handoff: HandoffSnapshot, + snapshot: dict[str, Any], + entry: dict[str, Any], + ) -> dict[str, Any]: + run = store.run(run_id) + if run["state"] in TERMINAL_STATES: + return { + "action": "already_terminal", + "state": run["state"], + "version": run["version"], + "replayed_continuation": True, + "launch": None, + } + _, gate = self._review_gate( + store, run_id, handoff, snapshot, entry["decision"], + ) + if gate["action"] == "repeat_review": + decision = entry["decision"] + workflow = snapshot["task"]["workflow"] + requeue_method = ( + store.requeue_delivery_revision + if workflow == "issue-delivery" + else store.requeue_review_revision + ) + requeue = requeue_method( + run_id, handoff.handoff_id, + decision["submission_id"], decision["submission_hash"], + ) + if requeue["action"] == "requeued": + queued = store.run(run_id) + package, digest = self._verified_package(queued) + return { + "action": "requeued", + "state": queued["state"], + "version": queued["version"], + "replayed_continuation": requeue["replayed"], + "launch": (queued["version"], package, digest), + } + if requeue["action"] == "already_advanced": + advanced = store.run(run_id) + return { + "action": "already_advanced", + "state": advanced["state"], + "version": advanced["version"], + "replayed_continuation": True, + "launch": None, + } + gate = { + **gate, + "action": "budget_exhausted", + "terminal_state": "failed", + } + error = { + "error": "BUDGET_EXHAUSTED", + "message": f"{workflow} worker invocation or wall budget is exhausted", + } + elif gate["action"] == "budget_exhausted": + error = { + "error": "BUDGET_EXHAUSTED", + "message": "branch review revision budget is exhausted", + } + elif gate["terminal_state"] == "failed": + error = { + "error": "REVIEW_REJECTED", + "message": entry["decision"]["reason"] or "host rejected the review", + } + else: + error = None + return self._terminalize_branch_review( + store, + run_id, + handoff, + snapshot, + entry, + gate["terminal_state"], + error, + ) + + def _saved_headless_lead( + self, + store: Store, + run_id: str, + snapshot: dict[str, Any], + handoff: HandoffSnapshot, + ) -> tuple[dict[str, Any], dict[str, Any]] | None: + frozen_handoff = { + "handoff_id": handoff.handoff_id, + "packet": handoff.packet, + "packet_sha256": handoff.packet_sha256, + } + for attempt in reversed(store.attempts_for_run(run_id)): + if attempt.get("role") != "lead" or attempt.get("status") != "finished": + continue + artifact = store.artifact_named( + run_id, f"lead-attempt-{attempt['id']}.json", + ) + if artifact is None: + continue + path = Path(artifact["path"]) + try: + content = path.read_bytes() + except OSError as exc: + raise ConflictError("saved headless lead evidence is missing") from exc + if (len(content) != artifact["byte_size"] + or hashlib.sha256(content).hexdigest() != artifact["sha256"]): + raise ConflictError("saved headless lead evidence is corrupt") + try: + evidence = decode_headless_lead_evidence( + content, snapshot, frozen_handoff, + ) + except ContractError: + continue + return attempt, evidence + return None + + def _continue_headless_lead( + self, + store: Store, + run_id: str, + run: dict[str, Any], + handoff: HandoffSnapshot, + snapshot: dict[str, Any], + ) -> dict[str, Any]: + saved = self._saved_headless_lead( + store, run_id, snapshot, handoff, + ) + if saved is None: + queued = store.queue_headless_lead(run_id, run["version"]) + if queued["action"] == "budget_exhausted": + error = { + "error": "BUDGET_EXHAUSTED", + "message": "headless lead worker invocation budget is exhausted", + } + terminal_artifacts = self._paused_review_terminal_artifacts( + store, + run_id, + snapshot, + handoff, + state="failed", + phase="lead", + error=error, + ) + version = store.fail_queued_budget( + run_id, run["version"], error, terminal_artifacts, + ) + return { + "action": "budget_exhausted", + "state": "failed", + "version": version, + "replayed_continuation": False, + "launch": None, + } + if queued["action"] != "queued": + raise ConflictError("headless lead handoff was not queueable") + prepared = store.run(run_id) + package, digest = self._verified_package(prepared) + return { + "action": "headless_lead_queued", + "state": prepared["state"], + "version": prepared["version"], + "replayed_continuation": False, + "launch": (prepared["version"], package, digest), + } + + attempt, evidence = saved + claim = store.claim_handoff( + run_id, + run["version"], + f"headless-lead:{attempt['id']}", + ) + choice = evidence["choice"] + body = { + "schema_version": 1, + "submission_id": f"headless-{attempt['id']}", + "disposition": choice["disposition"], + "reason": choice["reason"], + "evidence_refs": [ + { + "artifact_id": reference["artifact_id"], + "sha256": reference["sha256"], + } + for reference in handoff.packet["artifacts"] + ], + } + decision = {**body, "submission_hash": request_hash(body)} + self._review_gate(store, run_id, handoff, snapshot, decision) + submission = store.record_handoff_submission(run_id, claim, decision) + entry = store.recorded_handoff_submission(run_id, handoff.handoff_id) + if entry is None: + raise ConflictError("recorded headless lead submission is missing") + continuation = self._continue_branch_review_submission( + store, run_id, handoff, snapshot, entry, + ) + launch = continuation.pop("launch") + current = store.run(run_id) + return { + "action": continuation["action"], + "state": current["state"], + "version": current["version"], + "submission_id": submission.submission_id, + "disposition": submission.disposition, + "replayed_continuation": continuation["replayed_continuation"], + "launch": launch, + } + + def council_handoff_view(self, run_id: str) -> dict[str, Any]: + """Validated portable Council evidence, without claiming or choosing.""" + from .council_runtime import validate_saved_handoff, verified_artifact + from .reports import handoff_report_names + store = self._store() + try: + run = store.run(run_id) + snapshot = self._review_snapshot(run) + handoff = store.handoff_snapshot(run_id) + if (handoff is None or run["state"] != "awaiting_host" or run["phase"] is not None + or handoff.status != "open" + or snapshot["task"].get("workflow") != "council-decision"): + raise ConflictError("Council host handoff is not open") + validate_saved_handoff(store, run_id, handoff, snapshot) + _, markdown = handoff_report_names(handoff.sequence) + report = verified_artifact(store, run_id, markdown).decode("utf-8") + report_artifact = store.artifact_named(run_id, markdown) + return {"run_id": run_id, "workflow": "council-decision", "version": run["version"], "packet_sha256": handoff.packet_sha256, + "candidate_sha256": handoff.packet["candidate_sha256"], "proposals": handoff.packet["proposals"], + "critique": handoff.packet["critique"], "checks": handoff.packet["checks"], "report": report, + "report_artifact": report_artifact, + "pending_finish": store.terminal_finish_decision(run_id, handoff.handoff_id)} + finally: + store.close() + + def finish_council(self, run_id: str, disposition: str, reason: str, *, chosen: str, + supported_claims: list[str], discarded_alternatives: list[str], validation: str) -> dict[str, Any]: + """Explicit guided Council choice over the existing exact host claim.""" + from .council import validate_choice + from .council_runtime import validate_saved_handoff, decision_gate + if not isinstance(reason, str) or not reason.strip() or len(reason) > 2000: + raise ContractError("finish requires a non-empty reason of at most 2000 characters") + reason = reason.strip() + store = self._store() + try: + run = store.run(run_id) + snapshot = self._review_snapshot(run) + handoff = store.handoff_snapshot(run_id) + if (handoff is None or run["state"] != "awaiting_host" or run["phase"] is not None + or handoff.status != "open" + or snapshot["task"].get("workflow") != "council-decision" or snapshot["task"]["lead"]["mode"] != "host"): + raise ConflictError("guided Council finish requires a current open host handoff") + validate_saved_handoff(store, run_id, handoff, snapshot) + choice = {"disposition": disposition, "reason": reason, "chosen": chosen, + "supported_claims": supported_claims, "discarded_alternatives": discarded_alternatives, + "validation": validation, "unresolved_objections": [item["id"] for item in handoff.packet["critique"]["objections"]]} + validate_choice(choice, snapshot["task"]["council"], handoff.packet["critique"]) + if disposition == "revise": + raise ContractError("Council additional rounds require a new explicitly capped run") + if disposition == "accept" and not handoff.packet["evaluation"]["accept_allowed"]: + raise ContractError("Council acceptance is blocked by mandatory checks/integrity") + body = {"schema_version": 1, "submission_id": "terminal-" + handoff.handoff_id, + "disposition": disposition, "reason": reason, "council_choice": choice, + "evidence_refs": [{"artifact_id": ref["artifact_id"], "sha256": ref["sha256"]} for ref in handoff.packet["artifacts"]]} + decision = {**body, "submission_hash": request_hash(body)} + decision_gate(store, run_id, handoff, snapshot, decision) + version = run["version"] + finally: + store.close() + claimed = self.handoff_claim(run_id, version, "terminal-operator", _initial_only=True, + _terminal_decision=decision) + return self.handoff_complete(run_id, claimed["claim"], decision) + + def handoff_complete( + self, + run_id: str, + claim: dict[str, Any], + decision: dict[str, Any], + ) -> dict[str, Any]: + decoded = self._decode_claim(claim) + if decoded.run_id != run_id: + raise ConflictError("handoff completion claim targets a different run") + store = self._store() + launch = None + try: + run = store.run(run_id) + handoff = store.handoff_snapshot_by_id(run_id, decoded.handoff_id) + if handoff.packet.get("workflow") == "council-decision": + from .council_runtime import complete + snapshot = self._review_snapshot(run) + if snapshot["task"]["lead"]["mode"] == "headless" and not decoded.owner_id.startswith("headless-lead:"): + raise ConflictError("headless Council does not accept a host completion") + return complete(self, store, run_id, handoff, snapshot, {**decision, "_claim": claim}) + managed_review = handoff.packet.get("workflow") in { + "branch-review", "issue-delivery", + } + snapshot = self._review_snapshot(run) if managed_review else None + if (managed_review and snapshot["task"]["lead"]["mode"] == "headless" + and not decoded.owner_id.startswith("headless-lead:")): + raise ConflictError( + "headless review does not accept a host completion" + ) + if managed_review: + # The store still validates the submission hash/identity on a + # terminal replay. Do not apply new execution gates retroactively. + if run["state"] not in TERMINAL_STATES: + self._review_gate(store, run_id, handoff, snapshot, decision) + submission = store.record_handoff_submission(run_id, decoded, decision) + continuation = None + if managed_review: + entry = store.recorded_handoff_submission(run_id, decoded.handoff_id) + if entry is None: + raise ConflictError("recorded review submission is missing") + continuation = self._continue_branch_review_submission( + store, run_id, handoff, snapshot, entry, + ) + launch = continuation.pop("launch") + run = store.run(run_id) + response = { + "run_id": run_id, + "state": run["state"], + "phase": run["phase"], + "version": run["version"], + "handoff_id": submission.handoff_id, + "submission_id": submission.submission_id, + "submission_hash": submission.submission_hash, + "disposition": submission.disposition, + "recorded_run_version": submission.recorded_run_version, + "replayed": submission.replayed, + } + if continuation is not None: + response["continuation"] = continuation + finally: + store.close() + if launch is not None: + version, package, digest = launch + self._spawn_daemon(run_id, version, package, digest) + response["launched"] = True + elif managed_review: + response["launched"] = False + return response + + def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[str, Any]: + store = self._store() + launch: tuple[int, Path, str] | None = None + preparation_error: dict[str, Any] | None = None + branch_response: dict[str, Any] | None = None + try: + run = store.run(run_id) + if run["state"] in TERMINAL_STATES: raise ConflictError("terminal run cannot resume; start a superseding run") + if run["state"] == "awaiting_host": + handoff = store.handoff_snapshot(run_id) + if handoff is not None and handoff.packet.get("workflow") == "council-decision": + from .council_runtime import continue_lead + continuation = continue_lead(self, store, run_id, run, handoff, self._review_snapshot(run)) + launch = continuation.pop("launch", None) + if launch is not None: + self._spawn_daemon(run_id, *launch) + return continuation + if run["state"] == "awaiting_host" and run["phase"] is None: + handoff = store.handoff_snapshot(run_id) + if (handoff is None or handoff.packet.get("workflow") + not in {"branch-review", "issue-delivery"}): + raise ConflictError("run has no resumable review handoff") + snapshot = self._review_snapshot(run) + if snapshot["task"]["lead"]["mode"] == "headless": + continuation = self._continue_headless_lead( + store, run_id, run, handoff, snapshot, + ) + launch = continuation.pop("launch") + current = store.run(run_id) + branch_response = { + "run_id": run_id, + "disposition": continuation["action"], + "state": current["state"], + "version": current["version"], + "launched": launch is not None, + } + else: + raise ConflictError("host-led handoff must be completed by its host") + elif run["state"] == "awaiting_host" and run["phase"] == "handoff_submitted": + handoff = store.handoff_snapshot(run_id) + if (handoff is None or handoff.packet.get("workflow") + not in {"branch-review", "issue-delivery"}): + raise ConflictError("run has no resumable review submission") + snapshot = self._review_snapshot(run) + entry = store.recorded_handoff_submission(run_id, handoff.handoff_id) + if entry is None: + raise ConflictError("recorded review submission is missing") + continuation = self._continue_branch_review_submission( + store, run_id, handoff, snapshot, entry, + ) + launch = continuation.pop("launch") + current = store.run(run_id) + branch_response = { + "run_id": run_id, + "disposition": continuation["action"], + "state": current["state"], + "version": current["version"], + "launched": launch is not None, + } + if branch_response is None and run["state"] in {"running","cancelling"}: + from .supervisor import Supervisor + attempt=store.attempt(run_id) + disposition = Supervisor(store).import_durable(run_id) if attempt and attempt.get("exit_record") else Supervisor(store).recover(run_id) + if disposition != "requeued": + return {"run_id": run_id, "disposition": disposition, "launched": False} + run = store.run(run_id) + version = run["version"] + if branch_response is not None: + pass + elif run["state"] == "queued" and run["phase"] == "preparing": + submitted = json.loads(run["submitted_request"]) + claim = store.reclaim_preparation( + run_id, run["version"], f"preflight-recovery:{os.getpid()}", + ) + launch, preparation_error = self._continue_preparation( + store, run_id, claim.fencing_token, submitted, + ) + elif run["state"] == "queued" and run["phase"] == "launching": + version = store.recover_launching(run_id, run["version"]) + elif run["state"] == "queued" and run["phase"] is None: + version = run["version"] + elif run["state"] == "blocked": + if not isinstance(recovery,dict) or set(recovery)!={"attempt_id","disposition"} or recovery["disposition"] not in {"confirm_dead","retain_ownership"}: + raise ContractError("blocked resume requires a typed recovery disposition") + attempt=store.attempt(run_id) + if not attempt or recovery["attempt_id"]!=attempt["id"]: + raise ConflictError("recovery disposition targets a different attempt") + required="retain_ownership" if attempt["status"]=="ownership_ambiguous" else "confirm_dead" + if recovery["disposition"]!=required: + raise ConflictError("recovery disposition contradicts persisted ownership") + return {"run_id":run_id,"disposition":required,"launched":False} + else: + raise ConflictError("run is not resumable") + finally: store.close() + if branch_response is not None: + if launch is not None: + version, package, digest = launch + self._spawn_daemon(run_id, version, package, digest) + return branch_response + if run["phase"] == "preparing": + if launch is None: + return { + "run_id": run_id, + "disposition": "preparation_failed", + "launched": False, + "error": preparation_error, + } + version,package,digest=launch + else: + package,digest=self._verified_package(run) + self._spawn_daemon(run_id, version, package, digest) + return {"run_id": run_id, "disposition": "continued", "launched": True} diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py new file mode 100644 index 0000000..271018c --- /dev/null +++ b/plugin/core/src/devsquad/store.py @@ -0,0 +1,5619 @@ +"""Transactional SQLite ledger and atomic artifact storage for M2.""" + +from __future__ import annotations + +from dataclasses import dataclass +from datetime import datetime, timedelta, timezone +import hashlib +from functools import wraps +from importlib.resources import files +import json +import os +from pathlib import Path +import sqlite3 +import subprocess +import tempfile +import uuid +from typing import Any + +from .contracts import BudgetExhausted, ContractError + +SUPPORTED_SCHEMA_VERSION = 17 +TERMINAL_STATES = {"succeeded", "failed", "cancelled"} +HOST_LEASE_SECONDS = 10 * 60 +BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ + "receipt.json", + "receipt.md", + "events.jsonl", + "artifact-manifest.json", + "result-receipt.json", +}) + + +class ConflictError(ContractError): + code = "CONFLICT" + + +class SchemaVersionError(ContractError): + code = "SCHEMA_UNSUPPORTED" + + +@dataclass(frozen=True) +class StartClaim: + run_id: str + project_id: str + request_hash: str + version: int + created: bool + fencing_token: int | None + + +@dataclass(frozen=True) +class AttemptReservation: + run_id: str + attempt_id: str + attempt_token: str + supervisor_token: int + version: int + + +@dataclass(frozen=True) +class PreparationClaim: + run_id: str + fencing_token: int + version: int + + +@dataclass(frozen=True) +class HandoffClaim: + run_id: str + handoff_id: str + owner_id: str + fencing_token: int + expires_at: str + run_version: int + action: str + + +@dataclass(frozen=True) +class HandoffSnapshot: + run_id: str + run_state: str + run_phase: str | None + run_version: int + handoff_id: str + sequence: int + status: str + packet: dict[str, Any] + packet_sha256: str + created_run_version: int + submitted_run_version: int | None + created_at: str + closed_at: str | None + claim: HandoffClaim | None + + +@dataclass(frozen=True) +class HandoffSubmission: + run_id: str + handoff_id: str + submission_id: str + submission_hash: str + disposition: str + recorded_run_version: int + replayed: bool + + +def _utc_now() -> str: + return datetime.now(timezone.utc).isoformat() + + +def _authoritative_now(value: datetime | None = None) -> datetime: + current = datetime.now(timezone.utc) if value is None else value + if current.tzinfo is None or current.utcoffset() is None: + raise ContractError("authoritative time must include a timezone") + return current.astimezone(timezone.utc) + + +def _parse_utc(value: str) -> datetime: + try: + parsed = datetime.fromisoformat(value) + except (TypeError, ValueError) as exc: + raise ConflictError("persisted handoff lease timestamp is invalid") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ConflictError("persisted handoff lease timestamp is invalid") + return parsed.astimezone(timezone.utc) + + +def canonical_json(value: Any) -> str: + try: + return json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False) + except (TypeError, ValueError) as exc: + raise ContractError("request must be finite JSON") from exc + + +def request_hash(value: Any) -> str: + return hashlib.sha256(canonical_json(value).encode()).hexdigest() + + +def git_common_dir(worktree: Path) -> Path: + try: + result = subprocess.run( + ["git", "-C", str(worktree), "rev-parse", "--path-format=absolute", "--git-common-dir"], + text=True, capture_output=True, check=True, + ) + except (OSError, subprocess.CalledProcessError) as exc: + raise ContractError("project path is not a Git worktree") from exc + return Path(result.stdout.strip()).resolve(strict=True) + + +def _project_terminal(method): + @wraps(method) + def wrapped(self, run_id, *args, **kwargs): + result = method(self, run_id, *args, **kwargs) + self.project_final_outcome(run_id) + return result + return wrapped + + +class Store: + def __init__(self, database: Path, artifacts: Path): + self.database = database + self.artifacts = artifacts + database.parent.mkdir(parents=True, exist_ok=True) + artifacts.mkdir(parents=True, exist_ok=True) + self.connection = sqlite3.connect(database, timeout=10, isolation_level=None) + self.connection.row_factory = sqlite3.Row + self.connection.create_function("devsquad_connection_schema", 0, lambda: SUPPORTED_SCHEMA_VERSION) + try: + self.connection.execute("PRAGMA busy_timeout=10000") + self.connection.execute("PRAGMA foreign_keys=ON") + self.migrate() + self.connection.execute("PRAGMA journal_mode=WAL") + self.connection.execute("PRAGMA synchronous=FULL") + except Exception: + self.connection.close() + raise + + def close(self) -> None: + self.connection.close() + + def migrate(self) -> None: + self.connection.execute("BEGIN EXCLUSIVE") + try: + table = self.connection.execute("SELECT 1 FROM sqlite_master WHERE type='table' AND name='schema_migrations'").fetchone() + current = self.connection.execute("SELECT COALESCE(MAX(version), 0) FROM schema_migrations").fetchone()[0] if table else 0 + initial_version = current + if current > SUPPORTED_SCHEMA_VERSION: + raise SchemaVersionError(f"database schema {current} is newer than supported {SUPPORTED_SCHEMA_VERSION}") + if 0 < current < SUPPORTED_SCHEMA_VERSION: + runs_table = self.connection.execute("SELECT 1 FROM sqlite_master WHERE type='table' AND name='runs'").fetchone() + if runs_table is not None: + pending = self.connection.execute( + "SELECT id,state,phase FROM runs WHERE state NOT IN ('succeeded','failed','cancelled') ORDER BY created_at", + ).fetchall() + if pending: + identifiers = ", ".join(row["id"] for row in pending) + raise SchemaVersionError( + f"schema upgrade {current} → {SUPPORTED_SCHEMA_VERSION} deferred: active/recoverable runs {identifiers}; " + "finish or cancel them using the previous installed release, then retry the update" + ) + while current < SUPPORTED_SCHEMA_VERSION: + next_version = current + 1 + candidates = [entry for entry in files("devsquad.migrations").iterdir() if entry.name.startswith(f"{next_version:03d}_") and entry.name.endswith(".sql")] + if len(candidates) != 1: + raise SchemaVersionError(f"migration {next_version} is missing or ambiguous") + sql = candidates[0].read_text() + statement = "" + for line in sql.splitlines(keepends=True): + statement += line + if sqlite3.complete_statement(statement): + self.connection.execute(statement.strip()) + statement = "" + if statement.strip(): + raise SchemaVersionError( + f"migration {next_version} contains incomplete SQL", + ) + self.connection.execute("INSERT INTO schema_migrations(version, applied_at) VALUES(?, ?)", (next_version, _utc_now())) + current = next_version + if initial_version < SUPPORTED_SCHEMA_VERSION: + # An old Store opened before the exclusive upgrade can outlive + # the constructor's version check. Fence its later writes at + # the database boundary, including admission of a new run. + # Old packages either report their lower version or lack the + # function entirely; both fail before mutating a saved row. + tables = self.connection.execute( + "SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'sqlite_%' AND name!='schema_migrations'", + ).fetchall() + for table_row in tables: + name = table_row["name"].replace('"', '""') + for action in ("INSERT", "UPDATE", "DELETE"): + trigger = f"devsquad_schema_guard_{name}_{action.lower()}" + self.connection.execute( + f'CREATE TRIGGER IF NOT EXISTS "{trigger}" BEFORE {action} ON "{name}" BEGIN ' + "SELECT CASE WHEN devsquad_connection_schema() < (SELECT MAX(version) FROM schema_migrations) " + "THEN RAISE(ABORT, 'database schema is newer than connection supports') END; END" + ) + self.connection.execute("COMMIT") + except Exception: + self.connection.execute("ROLLBACK") + raise + + def _project(self, common_dir: Path) -> str: + key = str(common_dir) + row = self.connection.execute("SELECT id FROM projects WHERE git_common_dir=?", (key,)).fetchone() + if row: + return row[0] + project_id = str(uuid.uuid4()) + self.connection.execute("INSERT INTO projects(id, git_common_dir, created_at) VALUES(?,?,?)", (project_id, key, _utc_now())) + return project_id + + def runs_for_project(self, project: Path) -> list[dict[str, Any]]: + """Bounded choices from this canonical Git project, without registering it.""" + common_dir = git_common_dir(project) + return [dict(row) for row in self.connection.execute( + "SELECT r.id AS run_id,r.state,r.phase FROM runs r " + "JOIN projects p ON p.id=r.project_id WHERE p.git_common_dir=? " + "ORDER BY r.id LIMIT 21", (str(common_dir),), + ).fetchall()] + + def claim_start(self, worktree: Path, idempotency_key: str, submitted_request: Any, owner_id: str, *, objective_outcome: bool = False) -> StartClaim: + if not idempotency_key or not owner_id: + raise ContractError("idempotency key and owner are required") + encoded, digest = canonical_json(submitted_request), request_hash(submitted_request) + common_dir = git_common_dir(worktree) + self.connection.execute("BEGIN IMMEDIATE") + try: + project_id = self._project(common_dir) + existing = self.connection.execute( + "SELECT id, request_hash, version FROM runs WHERE project_id=? AND idempotency_key=?", + (project_id, idempotency_key), + ).fetchone() + if existing: + if existing["request_hash"] != digest: + raise ConflictError("idempotency key was already used with a different request") + self.connection.execute("COMMIT") + return StartClaim(existing["id"], project_id, digest, existing["version"], False, None) + run_id, now = str(uuid.uuid4()), _utc_now() + self.connection.execute( + "INSERT INTO runs(id,project_id,idempotency_key,request_hash,submitted_request,state,phase,version,created_at,updated_at,worktree_path) VALUES(?,?,?,?,?,'queued','preparing',1,?,?,?)", + (run_id, project_id, idempotency_key, digest, encoded, now, now, str(worktree.resolve(strict=True))), + ) + self.connection.execute( + "INSERT INTO claims(run_id,kind,fencing_token,owner_id,active,claimed_at) VALUES(?,'preparing',1,?,1,?)", + (run_id, owner_id, now), + ) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,1,'run.preparing','{}',?)", + (run_id, now), + ) + if objective_outcome: + self.connection.execute("INSERT INTO objective_outcome_jobs(run_id,requested_at) VALUES(?,?)", (run_id, now)) + self.connection.execute("COMMIT") + return StartClaim(run_id, project_id, digest, 1, True, 1) + except Exception: + self.connection.execute("ROLLBACK") + raise + + def reclaim_preparation(self, run_id: str, expected_version: int, owner_id: str) -> PreparationClaim: + """Take over an interrupted preflight while fencing its former owner.""" + if type(expected_version) is not int or expected_version < 1 or not owner_id: + raise ContractError("preparation recovery requires a version and owner") + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,c.fencing_token " + "FROM runs r JOIN claims c ON c.run_id=r.id WHERE r.id=?", + (run_id,), + ).fetchone() + if (not row or row["state"] != "queued" or row["phase"] != "preparing" + or row["version"] != expected_version): + raise ConflictError("preparation is no longer recoverable at that version") + token, version, now = row["fencing_token"] + 1, expected_version + 1, _utc_now() + self.connection.execute( + "UPDATE claims SET fencing_token=?,owner_id=?,active=1,claimed_at=? WHERE run_id=?", + (token, owner_id, now, run_id), + ) + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({"owner_id": owner_id, "fencing_token": token}) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.preparation_reclaimed',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return PreparationClaim(run_id, token, version) + except Exception: + self.connection.execute("ROLLBACK") + raise + + def preparation_worktree(self, run_id: str, fencing_token: int, submitted_path: Path) -> Path: + """Resolve a submitted path only if it still names the claimed project.""" + row = self.connection.execute( + "SELECT r.state,r.phase,r.worktree_path,proj.git_common_dir,c.fencing_token,c.active " + "FROM runs r JOIN projects proj ON proj.id=r.project_id " + "JOIN claims c ON c.run_id=r.id WHERE r.id=?", + (run_id,), + ).fetchone() + if (not row or row["state"] != "queued" or row["phase"] != "preparing" + or not row["active"] or row["fencing_token"] != fencing_token): + raise ConflictError("preparation claim is stale or cancelled") + try: + resolved = submitted_path.resolve(strict=True) + except OSError as exc: + raise ContractError("claimed project worktree is no longer available") from exc + if str(resolved) != row["worktree_path"]: + raise ContractError("project repo_path no longer resolves to the claimed worktree") + if str(git_common_dir(resolved)) != row["git_common_dir"]: + raise ContractError("claimed worktree no longer belongs to the persisted project") + return resolved + + def validate_predecessor(self, run_id: str, fencing_token: int, supersedes_run_id: str | None) -> None: + if supersedes_run_id is None: + return + row = self.connection.execute( + "SELECT r.project_id,r.state,r.phase,c.fencing_token,c.active," + "predecessor.project_id AS predecessor_project,predecessor.state AS predecessor_state " + "FROM runs r JOIN claims c ON c.run_id=r.id " + "LEFT JOIN runs predecessor ON predecessor.id=? WHERE r.id=?", + (supersedes_run_id, run_id), + ).fetchone() + if (not row or row["state"] != "queued" or row["phase"] != "preparing" + or not row["active"] or row["fencing_token"] != fencing_token): + raise ConflictError("preparation claim is stale or cancelled") + if (row["predecessor_project"] != row["project_id"] + or row["predecessor_state"] not in TERMINAL_STATES): + raise ConflictError("superseded run must be terminal and belong to the same project") + + def _terminal_receipt( + self, + run_id: str, + terminal_state: str, + phase: str, + payload: Any, + *, + attempt_id: str | None = None, + now: str | None = None, + ) -> tuple[Path, str, int, str]: + finished_at = now or _utc_now() + receipt = { + "schema_version": 1, + "run_id": run_id, + "state": terminal_state, + "phase": phase, + "attempt_id": attempt_id, + "returncode": None, + "cancelled": terminal_state == "cancelled", + "timed_out": False, + "error": payload if terminal_state == "failed" else None, + "finished_at": finished_at, + } + path, digest, size = self.finalize_artifact( + run_id, "result-receipt.json", canonical_json(receipt).encode(), + ) + return path, digest, size, finished_at + + def _reference_terminal_receipt( + self, + run_id: str, + version: int, + path: Path, + digest: str, + size: int, + now: str, + ) -> int: + artifact_id = str(uuid.uuid4()) + self.connection.execute( + "INSERT INTO artifacts(id,run_id,name,path,sha256,byte_size,created_at) " + "VALUES(?,?,'result-receipt.json',?,?,?,?)", + (artifact_id, run_id, str(path), digest, size, now), + ) + version += 1 + artifact_event = canonical_json({ + "artifact_id": artifact_id, + "name": "result-receipt.json", + "sha256": digest, + "byte_size": size, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'artifact.recorded',?,?)", + (run_id, version, artifact_event, now), + ) + return version + + def complete_preparation( + self, + run_id: str, + fencing_token: int, + mutable_snapshot: Any, + *, + package_path: str | None = None, + package_digest: str | None = None, + supersedes_run_id: str | None = None, + worktree_path: str | None = None, + ) -> int: + snapshot = canonical_json(mutable_snapshot) + resolved_worktree = None + worktree_common = None + if worktree_path is not None: + if not isinstance(worktree_path, str) or not worktree_path or not Path(worktree_path).is_absolute(): + raise ContractError("prepared worktree path must be absolute") + resolved_worktree = str(Path(worktree_path).resolve(strict=True)) + worktree_common = str(git_common_dir(Path(resolved_worktree))) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,c.fencing_token,c.active," + "p.git_common_dir FROM runs r JOIN claims c ON c.run_id=r.id " + "JOIN projects p ON p.id=r.project_id WHERE r.id=?", + (run_id,), + ).fetchone() + if not row or row["state"] != "queued" or row["phase"] != "preparing" or not row["active"] or row["fencing_token"] != fencing_token: + raise ConflictError("preparation claim is stale or cancelled") + if worktree_common is not None and worktree_common != row["git_common_dir"]: + raise ContractError("prepared worktree belongs to a different project") + if supersedes_run_id is not None: + predecessor = self.connection.execute("SELECT project_id,state FROM runs WHERE id=?", (supersedes_run_id,)).fetchone() + project = self.connection.execute("SELECT project_id FROM runs WHERE id=?", (run_id,)).fetchone() + if not predecessor or predecessor["project_id"] != project["project_id"] or predecessor["state"] not in TERMINAL_STATES: + raise ConflictError("superseded run must be terminal and belong to the same project") + version, now = row["version"] + 1, _utc_now() + self._freeze_experiment_assignment( + run_id, fencing_token, version, mutable_snapshot, package_digest, now, + ) + self.connection.execute( + "UPDATE runs SET mutable_snapshot=?,package_path=?,package_digest=?," + "supersedes_run_id=?,worktree_path=COALESCE(?,worktree_path),phase=NULL," + "version=?,updated_at=? WHERE id=?", + (snapshot, package_path, package_digest, supersedes_run_id, + resolved_worktree, version, now, run_id), + ) + self.connection.execute("UPDATE claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.queued','{}',?)", (run_id, version, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def _freeze_experiment_assignment( + self, run_id: str, fencing_token: int, run_version: int, + snapshot: Any, package_digest: str | None, recorded_at: str, + ) -> None: + """Inside complete_preparation's transaction, before any launch is possible.""" + from .experiment_provenance import ( + paired_input_identity, selected_execution_fingerprint, validate_assignment, + ) + from .learning import validate_experiment + + if not isinstance(snapshot, dict): + return + fields = {"experiment_spec", "experiment_assignment"} & set(snapshot) + if not fields: + return + if fields != {"experiment_spec", "experiment_assignment"}: + raise ContractError("experiment preparation requires both specification and assignment") + spec = validate_experiment(snapshot["experiment_spec"]) + project = self.connection.execute( + "SELECT r.project_id,p.git_common_dir FROM runs r " + "JOIN projects p ON p.id=r.project_id WHERE r.id=?", (run_id,), + ).fetchone() + if str(git_common_dir(Path(spec["project_path"]))) != project["git_common_dir"]: + raise ContractError("experiment specification belongs to another project") + assignment = validate_assignment( + snapshot["experiment_assignment"], spec=spec, + project_common_dir=project["git_common_dir"], + ) + identity = paired_input_identity( + snapshot, role=assignment["role"], package_digest=package_digest, + ) + if any(identity[key] != assignment[key] for key in identity): + raise ContractError("experiment paired input does not match its declaration") + selected = snapshot["routing"]["roles"][assignment["role"]]["selected"] + if any(selected[key] != assignment[key] for key in ("profile_id", "profile_sha256")): + raise ContractError("experiment selected profile does not match its arm") + if selected_execution_fingerprint(snapshot, role=assignment["role"]) != assignment["execution_sha256"]: + raise ContractError("experiment native execution does not match its declared arm") + if self.connection.execute( + "SELECT 1 FROM attempts WHERE run_id=? LIMIT 1", (run_id,), + ).fetchone() is not None: + raise ConflictError("experiment assignment must precede every attempt") + spec_json = canonical_json(spec) + spec_sha256 = hashlib.sha256(spec_json.encode()).hexdigest() + existing = self.connection.execute( + "SELECT project_id,spec_json,spec_sha256 FROM experiment_specs WHERE experiment_id=?", + (spec["experiment_id"],), + ).fetchone() + if existing is not None: + if (existing["project_id"] != project["project_id"] + or existing["spec_json"] != spec_json or existing["spec_sha256"] != spec_sha256): + raise ConflictError("experiment specification is already frozen differently") + else: + # An evaluation created before this contract cannot be retroactively + # upgraded into a predeclared trial with the same identifier. + if self.connection.execute( + "SELECT 1 FROM experiments WHERE experiment_id=?", (spec["experiment_id"],), + ).fetchone() is not None: + raise ConflictError("historical experiment cannot acquire new assignments") + self.connection.execute( + "INSERT INTO experiment_specs(experiment_id,project_id,spec_json,spec_sha256,recorded_at) " + "VALUES(?,?,?,?,?)", + (spec["experiment_id"], project["project_id"], spec_json, spec_sha256, recorded_at), + ) + if self.connection.execute( + "SELECT 1 FROM experiment_assignments WHERE run_id=? OR " + "(experiment_id=? AND (outcome_id=? OR (case_id=? AND arm=?)))", + (run_id, spec["experiment_id"], assignment["outcome_id"], + assignment["case_id"], assignment["arm"]), + ).fetchone() is not None: + raise ConflictError("experiment run or arm is already assigned") + payload = canonical_json(assignment) + self.connection.execute( + "INSERT INTO experiment_assignments(run_id,experiment_id,case_id,arm,outcome_id," + "assignment_json,assignment_sha256,snapshot_json,package_digest,frozen_run_version," + "preparation_fencing_token,recorded_at) VALUES(?,?,?,?,?,?,?,?,?,?,?,?)", + (run_id, spec["experiment_id"], assignment["case_id"], assignment["arm"], + assignment["outcome_id"], payload, hashlib.sha256(payload.encode()).hexdigest(), + canonical_json(snapshot), package_digest, run_version, fencing_token, recorded_at), + ) + + @_project_terminal + def fail_preparation( + self, + run_id: str, + fencing_token: int, + error: Any, + *, + mutable_snapshot: Any | None = None, + supersedes_run_id: str | None = None, + terminal_artifacts: list[dict[str, Any]] | None = None, + ) -> int: + encoded = canonical_json(error) + prepared = ( + self._prepare_exact_artifacts( + run_id, terminal_artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS, + ) + if terminal_artifacts is not None else None + ) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.project_id,r.state,r.phase,r.version,c.fencing_token,c.active " + "FROM runs r JOIN claims c ON c.run_id=r.id WHERE r.id=?", + (run_id,), + ).fetchone() + if not row or row["state"] != "queued" or row["phase"] != "preparing" or not row["active"] or row["fencing_token"] != fencing_token: + raise ConflictError("preparation failure is stale or cancelled") + if supersedes_run_id is not None: + predecessor = self.connection.execute( + "SELECT project_id,state FROM runs WHERE id=?", (supersedes_run_id,), + ).fetchone() + if (not predecessor or predecessor["project_id"] != row["project_id"] + or predecessor["state"] not in TERMINAL_STATES): + raise ConflictError("superseded run must be terminal and belong to the same project") + now = _utc_now() + if prepared is None: + path, digest, size, now = self._terminal_receipt( + run_id, "failed", "preparing", error, now=now, + ) + version = self._reference_terminal_receipt( + run_id, row["version"], path, digest, size, now, + ) + else: + version, _ = self._reference_prepared_artifacts( + run_id, row["version"], prepared, + ) + version += 1 + snapshot = canonical_json(mutable_snapshot) if mutable_snapshot is not None else None + self.connection.execute( + "UPDATE runs SET mutable_snapshot=COALESCE(?,mutable_snapshot),supersedes_run_id=?," + "state='failed',phase=NULL,version=?,updated_at=? WHERE id=?", + (snapshot, supersedes_run_id, version, now, run_id), + ) + self.connection.execute("UPDATE claims SET active=0 WHERE run_id=?", (run_id,)) + terminal_payload = dict(error) if isinstance(error, dict) else {"error": error} + terminal_payload["receipt"] = "result-receipt.json" + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.failed',?,?)", + (run_id, version, canonical_json(terminal_payload), now), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + @_project_terminal + def cancel_preparing(self, run_id: str) -> int: + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute("SELECT state,phase,version FROM runs WHERE id=?", (run_id,)).fetchone() + if not row: + raise ContractError("run does not exist") + if row["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return row["version"] + if row["state"] != "queued" or row["phase"] != "preparing": + raise ConflictError("run is no longer preparing") + path, digest, size, now = self._terminal_receipt(run_id, "cancelled", "preparing", None) + version = self._reference_terminal_receipt( + run_id, row["version"], path, digest, size, now, + ) + 1 + self.connection.execute("UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("UPDATE claims SET active=0,fencing_token=fencing_token+1 WHERE run_id=?", (run_id,)) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id, version, canonical_json({"receipt": "result-receipt.json"}), now), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def append_event(self, run_id: str, expected_version: int, event_type: str, payload: Any, state: str | None = None) -> int: + if state is not None: + raise ConflictError("generic events cannot change run state") + encoded = canonical_json(payload) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute("SELECT state,phase,version FROM runs WHERE id=?", (run_id,)).fetchone() + if not row or row["version"] != expected_version: + raise ConflictError("run version changed") + if row["state"] in TERMINAL_STATES: + raise ConflictError("terminal run is immutable") + if row["phase"] is not None: + raise ConflictError("run phase is owned by a fenced operation") + version, now = expected_version + 1, _utc_now() + self.connection.execute("UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,?,?,?)", (run_id, version, event_type, encoded, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def finalize_artifact(self, run_id: str, name: str, content: bytes) -> tuple[Path, str, int]: + if not name or Path(name).name != name: + raise ContractError("artifact name must be a single path component") + row = self.connection.execute("SELECT id FROM runs WHERE id=?", (run_id,)).fetchone() + if not row: + raise ContractError("run does not exist") + digest = hashlib.sha256(content).hexdigest() + directory = self.artifacts / run_id + directory.mkdir(parents=True, exist_ok=True) + name_key = hashlib.sha256(name.encode()).hexdigest()[:16] + destination = directory / f"{digest}.{name_key}.blob" + descriptor, temporary = tempfile.mkstemp(prefix=f".{name}.", dir=directory) + try: + with os.fdopen(descriptor, "wb") as stream: + stream.write(content); stream.flush(); os.fsync(stream.fileno()) + try: + os.link(temporary, destination) + except FileExistsError: + if hashlib.sha256(destination.read_bytes()).hexdigest() != digest: + raise ConflictError("content-addressed artifact path is corrupt") + directory_fd = os.open(directory, os.O_RDONLY) + try: os.fsync(directory_fd) + finally: os.close(directory_fd) + finally: + if os.path.exists(temporary): os.unlink(temporary) + return destination, digest, len(content) + + def reference_artifact(self, run_id: str, name: str, path: Path, expected_sha256: str) -> str: + if not name or Path(name).name != name: + raise ContractError("artifact name must be a single path component") + expected_parent = (self.artifacts / run_id).resolve() + if path.resolve().parent != expected_parent: + raise ContractError("artifact path is outside the run-owned store") + content = path.read_bytes() + actual = hashlib.sha256(content).hexdigest() + if actual != expected_sha256: + raise ConflictError("artifact hash changed before database reference") + artifact_id = str(uuid.uuid4()) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + if not row: + raise ContractError("run does not exist") + if row["state"] in TERMINAL_STATES: + raise ConflictError("terminal run is immutable") + version, now = row["version"] + 1, _utc_now() + self.connection.execute("INSERT INTO artifacts(id,run_id,name,path,sha256,byte_size,created_at) VALUES(?,?,?,?,?,?,?)", (artifact_id, run_id, name, str(path), actual, len(content), _utc_now())) + self.connection.execute("UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, now, run_id)) + event = canonical_json({"artifact_id": artifact_id, "name": name, "sha256": actual, "byte_size": len(content)}) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'artifact.recorded',?,?)", (run_id, version, event, now)) + self.connection.execute("COMMIT") + return artifact_id + except sqlite3.IntegrityError as exc: + self.connection.execute("ROLLBACK") + raise ConflictError("artifact name is already referenced for this run") from exc + except Exception: + self.connection.execute("ROLLBACK") + raise + + def store_artifact(self, run_id: str, name: str, content: bytes) -> str: + path, digest, _ = self.finalize_artifact(run_id, name, content) + return self.reference_artifact(run_id, name, path, digest) + + @staticmethod + def _wall_seconds_from_run(run: sqlite3.Row | dict[str, Any]) -> int | None: + documents = (run.get("mutable_snapshot"), run.get("submitted_request")) \ + if isinstance(run, dict) else (run["mutable_snapshot"], run["submitted_request"]) + for encoded in documents: + if not encoded: + continue + try: + document = json.loads(encoded) + task = document.get("task") if isinstance(document, dict) else None + budget = task.get("budget") if isinstance(task, dict) else None + wall_seconds = budget.get("wall_seconds") if isinstance(budget, dict) else None + except (TypeError, json.JSONDecodeError): + continue + if type(wall_seconds) is int and wall_seconds > 0: + return wall_seconds + return None + + def _execution_elapsed_ms( + self, + run_id: str, + run: sqlite3.Row | dict[str, Any], + current: datetime, + ) -> int: + """Count preflight and worker intervals while excluding saved host waits.""" + now = current.astimezone(timezone.utc) + created_at = _parse_utc(run["created_at"]) + queued = self.connection.execute( + "SELECT created_at FROM events WHERE run_id=? AND type='run.queued' " + "ORDER BY id LIMIT 1", + (run_id,), + ).fetchone() + if queued is not None: + preflight_end = _parse_utc(queued["created_at"]) + elif run["state"] == "queued" and run["phase"] == "preparing": + preflight_end = now + else: + preflight_end = _parse_utc(run["updated_at"]) + elapsed = max(0, int((preflight_end - created_at).total_seconds() * 1000)) + for attempt in self.connection.execute( + "SELECT status,created_at,finished_at FROM attempts WHERE run_id=?", + (run_id,), + ): + started = _parse_utc(attempt["created_at"]) + if (attempt["status"] == "ownership_ambiguous" + or attempt["finished_at"] is None): + finished = now + else: + finished = _parse_utc(attempt["finished_at"]) + elapsed += max(0, int((finished - started).total_seconds() * 1000)) + return elapsed + + def remaining_wall_seconds( + self, + run_id: str, + *, + now: datetime | None = None, + ) -> int | None: + run = self.connection.execute( + "SELECT state,phase,created_at,updated_at,mutable_snapshot,submitted_request " + "FROM runs WHERE id=?", + (run_id,), + ).fetchone() + if run is None: + raise ContractError("run does not exist") + wall_seconds = self._wall_seconds_from_run(run) + current = _authoritative_now(now) + experiment_remaining = self._experiment_remaining_wall_seconds(run_id, current) + if wall_seconds is None: + return experiment_remaining + elapsed_ms = self._execution_elapsed_ms( + run_id, run, current, + ) + remaining = max(0, (wall_seconds * 1000 - elapsed_ms) // 1000) + return remaining if experiment_remaining is None else min(remaining, experiment_remaining) + + def _experiment_remaining_wall_seconds(self, run_id: str, current: datetime) -> int | None: + row = self.connection.execute( + "SELECT s.spec_json,s.recorded_at FROM experiment_assignments a " + "JOIN experiment_specs s ON s.experiment_id=a.experiment_id WHERE a.run_id=?", + (run_id,), + ).fetchone() + if row is None: + return None + wall_seconds = json.loads(row["spec_json"])["budget"]["wall_seconds"] + elapsed_ms = max(0, int((current - _parse_utc(row["recorded_at"])).total_seconds() * 1000)) + return max(0, (wall_seconds * 1000 - elapsed_ms) // 1000) + + def _enforce_attempt_budget( + self, + run_id: str, + run: sqlite3.Row, + ) -> None: + try: + snapshot = json.loads(run["mutable_snapshot"] or "null") + max_invocations = snapshot["task"]["budget"]["max_worker_invocations"] + except (KeyError, TypeError, json.JSONDecodeError): + max_invocations = None + if type(max_invocations) is int: + launched = self.worker_invocations(run_id) + if launched >= max_invocations: + raise BudgetExhausted("worker invocation budget is exhausted") + wall_seconds = self._wall_seconds_from_run(run) + if wall_seconds is not None: + elapsed_ms = self._execution_elapsed_ms( + run_id, run, _authoritative_now(), + ) + if wall_seconds * 1000 - elapsed_ms < 1000: + raise BudgetExhausted("run wall-time budget is exhausted") + + @staticmethod + def _pool_observation(row: sqlite3.Row) -> dict[str, Any]: + return { + "schema_version": 1, + "observation_id": row["observation_id"], + "pool_id": row["pool_id"], + "window_id": row["window_id"], + "applies_to": json.loads(row["applies_to_json"]), + "observed_at": row["observed_at"], + "expires_at": row["expires_at"], + "source": row["source"], + "used": row["used"], + "limit": row["limit_value"], + "unit": row["unit"], + "resets_at": row["resets_at"], + "confidence": row["confidence"], + } + + def _pool_observations(self, pool_id: str) -> list[dict[str, Any]]: + rows = self.connection.execute( + "SELECT observation_id,pool_id,window_id,applies_to_json,observed_at," + "expires_at,source,used,limit_value,unit,resets_at,confidence " + "FROM pool_observations WHERE pool_id=? " + "ORDER BY observed_at,observation_id", + (pool_id,), + ).fetchall() + return [self._pool_observation(row) for row in rows] + + def record_pool_observation( + self, observation: dict[str, Any], *, now: datetime | None = None, + ) -> dict[str, Any]: + """Persist one immutable, replay-safe capacity observation.""" + from .capacity import validate_observation + + current = _authoritative_now(now) + normalized = validate_observation(observation, now=current) + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = self.connection.execute( + "SELECT observation_id,pool_id,window_id,applies_to_json,observed_at," + "expires_at,source,used,limit_value,unit,resets_at,confidence,recorded_at " + "FROM pool_observations WHERE observation_id=?", + (normalized["observation_id"],), + ).fetchone() + if existing is not None: + stored = self._pool_observation(existing) + if stored != normalized: + raise ConflictError( + "capacity observation id was already used with different evidence", + ) + self.connection.execute("COMMIT") + return { + "observation": stored, + "recorded_at": existing["recorded_at"], + "replayed": True, + } + recorded_at = current.isoformat() + self.connection.execute( + "INSERT INTO pool_observations(" + "observation_id,pool_id,window_id,applies_to_json,observed_at,expires_at," + "source,used,limit_value,unit,resets_at,confidence,recorded_at" + ") VALUES(?,?,?,?,?,?,?,?,?,?,?,?,?)", + ( + normalized["observation_id"], normalized["pool_id"], + normalized["window_id"], canonical_json(normalized["applies_to"]), + normalized["observed_at"], normalized["expires_at"], + normalized["source"], normalized["used"], normalized["limit"], + normalized["unit"], normalized["resets_at"], + normalized["confidence"], recorded_at, + ), + ) + self.connection.execute("COMMIT") + return { + "observation": normalized, + "recorded_at": recorded_at, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def capacity_snapshot( + self, + pool_id: str, + *, + target: dict[str, Any] | None = None, + now: datetime | None = None, + ) -> dict[str, Any]: + """Return current evidence and local reservations for one pool/target.""" + from .capacity import derive_pool_capacity + + current = _authoritative_now(now) + in_flight = self.connection.execute( + "SELECT COUNT(*) FROM pool_reservations " + "WHERE pool_id=? AND reconciled_at IS NULL", + (pool_id,), + ).fetchone()[0] + return derive_pool_capacity( + pool_id, + self._pool_observations(pool_id), + target=target, + in_flight=in_flight, + now=current, + ) + + @staticmethod + def _capacity_target(profile: dict[str, Any]) -> dict[str, str] | None: + fields = ("harness", "model_family", "model_id") + if all(isinstance(profile.get(field), str) and profile[field] for field in fields): + return {field: profile[field] for field in fields} + return None + + def _reserve_pool_capacity_locked( + self, + *, + run_id: str, + pool_id: str, + purpose: str, + profile_id: str | None, + target: dict[str, Any] | None, + max_concurrency: int, + unknown_capacity_policy: str, + frozen_status: str | None, + attempt_id: str | None, + now: datetime, + ) -> dict[str, Any]: + from .capacity import derive_pool_capacity + + if purpose not in {"attempt", "qualification", "classifier"}: + raise ContractError("pool reservation purpose is invalid") + if type(max_concurrency) is not int or max_concurrency < 1: + raise ContractError("pool max_concurrency must be a positive integer") + if unknown_capacity_policy not in {"allow_bounded", "block"}: + raise ContractError("pool unknown_capacity_policy is invalid") + if profile_id is not None and (not isinstance(profile_id, str) or not profile_id): + raise ContractError("pool reservation profile_id is invalid") + if frozen_status is not None and frozen_status not in { + "available", "exhausted", "unknown", + }: + raise ContractError("frozen pool capacity status is invalid") + in_flight = self.connection.execute( + "SELECT COUNT(*) FROM pool_reservations " + "WHERE pool_id=? AND reconciled_at IS NULL", + (pool_id,), + ).fetchone()[0] + snapshot = derive_pool_capacity( + pool_id, + self._pool_observations(pool_id), + target=target, + in_flight=in_flight, + now=now, + ) + live_status = snapshot["status"] + status = "exhausted" if "exhausted" in {frozen_status, live_status} else live_status + if status == "exhausted": + raise ConflictError("account pool capacity is exhausted") + if status == "unknown" and unknown_capacity_policy == "block": + raise ConflictError("unknown account pool capacity is blocked") + effective_concurrency = 1 if status == "unknown" else max_concurrency + if in_flight >= effective_concurrency: + raise ConflictError("account pool concurrency is full") + reservation_id = str(uuid.uuid4()) + self.connection.execute( + "INSERT INTO pool_reservations(" + "id,pool_id,run_id,attempt_id,purpose,profile_id,reserved_at" + ") VALUES(?,?,?,?,?,?,?)", + ( + reservation_id, pool_id, run_id, attempt_id, purpose, profile_id, + now.isoformat(), + ), + ) + return { + "schema_version": 1, + "reservation_id": reservation_id, + "pool_id": pool_id, + "run_id": run_id, + "attempt_id": attempt_id, + "purpose": purpose, + "profile_id": profile_id, + "reserved_at": now.isoformat(), + "capacity_status": status, + "capacity_evidence": snapshot, + } + + def reserve_pool_capacity( + self, + run_id: str, + pool_id: str, + purpose: str, + *, + profile_id: str | None = None, + target: dict[str, Any] | None = None, + max_concurrency: int = 1, + unknown_capacity_policy: str = "allow_bounded", + now: datetime | None = None, + ) -> dict[str, Any]: + """Reserve qualification/classifier capacity under one SQLite write lock.""" + if purpose == "attempt": + raise ContractError("attempt capacity is reserved with its attempt") + current = _authoritative_now(now) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,mutable_snapshot FROM runs WHERE id=?", (run_id,), + ).fetchone() + if run is None: + raise ContractError("run does not exist") + if run["state"] in TERMINAL_STATES: + raise ConflictError("terminal run cannot reserve pool capacity") + reservation = self._reserve_pool_capacity_locked( + run_id=run_id, + pool_id=pool_id, + purpose=purpose, + profile_id=profile_id, + target=target, + max_concurrency=max_concurrency, + unknown_capacity_policy=unknown_capacity_policy, + frozen_status=None, + attempt_id=None, + now=current, + ) + self.connection.execute("COMMIT") + return reservation + except Exception: + self.connection.execute("ROLLBACK") + raise + + def reconcile_pool_reservation( + self, + reservation_id: str, + reason: str, + *, + now: datetime | None = None, + ) -> dict[str, Any]: + if not isinstance(reservation_id, str) or not reservation_id: + raise ContractError("pool reservation id is invalid") + if not isinstance(reason, str) or not reason: + raise ContractError("pool reconciliation reason is invalid") + current = _authoritative_now(now) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT pr.*,a.status AS attempt_status FROM pool_reservations pr " + "LEFT JOIN attempts a ON a.id=pr.attempt_id WHERE pr.id=?", + (reservation_id,), + ).fetchone() + if row is None: + raise ContractError("pool reservation does not exist") + if row["reconciled_at"] is not None: + if row["reconcile_reason"] != reason: + raise ConflictError("pool reservation was reconciled differently") + self.connection.execute("COMMIT") + return {**dict(row), "replayed": True} + if row["attempt_id"] is not None and row["attempt_status"] in { + "reserved", "running", "cancelling", "ownership_ambiguous", + }: + raise ConflictError("attempt ownership is not reconciled") + reconciled_at = current.isoformat() + self.connection.execute( + "UPDATE pool_reservations SET reconciled_at=?,reconcile_reason=? " + "WHERE id=? AND reconciled_at IS NULL", + (reconciled_at, reason, reservation_id), + ) + self.connection.execute("COMMIT") + return { + **dict(row), + "reconciled_at": reconciled_at, + "reconcile_reason": reason, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def active_pool_counts(self) -> dict[str, int]: + return { + row["pool_id"]: row["in_flight"] + for row in self.connection.execute( + "SELECT pool_id,COUNT(*) AS in_flight FROM pool_reservations " + "WHERE reconciled_at IS NULL GROUP BY pool_id" + ) + } + + @staticmethod + def _decision_observation_payload( + row: sqlite3.Row, *, action: str, replayed: bool, + ) -> dict[str, Any]: + return { + "run_id": row["run_id"], + "purpose_id": row["purpose_id"], + "mode": row["mode"], + "cache_key": row["cache_key"], + "status": row["status"], + "billable_calls": row["billable_calls"], + "request": json.loads(row["request_json"]), + "response": ( + json.loads(row["response_json"]) + if row["response_json"] is not None else None + ), + "response_sha256": row["response_sha256"], + "usage": ( + json.loads(row["usage_json"]) + if row["usage_json"] is not None else None + ), + "error": row["error"], + "applied": bool(row["applied"]), + "effect": ( + json.loads(row["effect_json"]) + if row["effect_json"] is not None else None + ), + "created_at": row["created_at"], + "updated_at": row["updated_at"], + "recorded_at": row["recorded_at"], + "action": action, + "replayed": replayed, + } + + def _decision_observation_row( + self, run_id: str, purpose_id: str, + ) -> sqlite3.Row | None: + return self.connection.execute( + "SELECT o.run_id,o.purpose_id,o.mode,o.applied,o.effect_json," + "o.recorded_at,c.cache_key,c.request_json,c.status,c.response_json," + "c.response_sha256,c.billable_calls,c.usage_json,c.owner_id,c.error," + "c.created_at,c.updated_at FROM run_decision_observations o " + "JOIN decision_cache c ON c.cache_key=o.cache_key " + "WHERE o.run_id=? AND o.purpose_id=?", + (run_id, purpose_id), + ).fetchone() + + def claim_decision_observation( + self, + run_id: str, + mode: str, + request: dict[str, Any], + owner_id: str, + *, + now: datetime | None = None, + ) -> dict[str, Any]: + """Claim at most one decision call or reuse its content-addressed result.""" + from .decision import decision_cache_key, validate_decision_request + + normalized = validate_decision_request(request) + if mode not in {"shadow", "advisory"}: + raise ContractError("decision observation mode is invalid") + if not isinstance(owner_id, str) or not owner_id: + raise ContractError("decision observation owner is invalid") + purpose_id = normalized["purpose"]["id"] + cache_key = decision_cache_key(normalized) + request_json = canonical_json(normalized) + recorded_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state FROM runs WHERE id=?", (run_id,), + ).fetchone() + if run is None: + raise ContractError("decision observation run does not exist") + if run["state"] in TERMINAL_STATES: + raise ConflictError( + "terminal run cannot claim a decision observation", + ) + existing_link = self._decision_observation_row(run_id, purpose_id) + if existing_link is not None: + if (existing_link["cache_key"] != cache_key + or existing_link["mode"] != mode + or existing_link["request_json"] != request_json): + raise ConflictError( + "run decision purpose was already bound differently", + ) + status = existing_link["status"] + action = "cached" + if status == "reserved": + self.connection.execute( + "UPDATE decision_cache SET owner_id=?,updated_at=? " + "WHERE cache_key=? AND status='reserved' AND billable_calls=0", + (owner_id, recorded_at, cache_key), + ) + action = "claimed" + elif status == "running": + if existing_link["owner_id"] == owner_id: + action = "in_flight" + else: + self.connection.execute( + "UPDATE decision_cache SET status='indeterminate'," + "error=?,updated_at=? WHERE cache_key=? AND status='running'", + ( + "prior launched decision call outcome is unknown", + recorded_at, cache_key, + ), + ) + action = "abstain" + elif status not in {"succeeded", "abstained"}: + action = "abstain" + row = self._decision_observation_row(run_id, purpose_id) + self.connection.execute("COMMIT") + return self._decision_observation_payload( + row, action=action, replayed=True, + ) + + cached = self.connection.execute( + "SELECT * FROM decision_cache WHERE cache_key=?", (cache_key,), + ).fetchone() + if cached is None: + self.connection.execute( + "INSERT INTO decision_cache(cache_key,request_json,status," + "billable_calls,owner_id,created_at,updated_at) " + "VALUES(?,?,'reserved',0,?,?,?)", + ( + cache_key, request_json, owner_id, recorded_at, + recorded_at, + ), + ) + action = "claimed" + replayed = False + else: + if cached["request_json"] != request_json: + raise ConflictError( + "decision cache key has conflicting request evidence", + ) + action = ( + "cached" if cached["status"] in {"succeeded", "abstained"} + else "in_flight" if cached["status"] in {"reserved", "running"} + else "abstain" + ) + replayed = True + self.connection.execute( + "INSERT INTO run_decision_observations(run_id,purpose_id,cache_key," + "mode,applied,effect_json,recorded_at) VALUES(?,?,?,?,0,NULL,?)", + (run_id, purpose_id, cache_key, mode, recorded_at), + ) + row = self._decision_observation_row(run_id, purpose_id) + self.connection.execute("COMMIT") + return self._decision_observation_payload( + row, action=action, replayed=replayed, + ) + except Exception: + self.connection.execute("ROLLBACK") + raise + + def launch_decision_call( + self, cache_key: str, owner_id: str, *, now: datetime | None = None, + ) -> dict[str, Any]: + """Fence a possibly billable call before crossing the adapter boundary.""" + if not isinstance(cache_key, str) or len(cache_key) != 64: + raise ContractError("decision cache key is invalid") + if not isinstance(owner_id, str) or not owner_id: + raise ContractError("decision observation owner is invalid") + launched_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT status,billable_calls,owner_id FROM decision_cache " + "WHERE cache_key=?", (cache_key,), + ).fetchone() + if row is None: + raise ContractError("decision cache entry does not exist") + if (row["status"] != "reserved" or row["billable_calls"] != 0 + or row["owner_id"] != owner_id): + raise ConflictError("decision call is not launchable") + self.connection.execute( + "UPDATE decision_cache SET status='running',billable_calls=1," + "updated_at=? WHERE cache_key=? AND status='reserved' " + "AND billable_calls=0 AND owner_id=?", + (launched_at, cache_key, owner_id), + ) + if self.connection.execute("SELECT changes()").fetchone()[0] != 1: + raise ConflictError("decision call launch fence changed") + self.connection.execute("COMMIT") + return { + "cache_key": cache_key, "status": "running", + "billable_calls": 1, "launched_at": launched_at, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def complete_decision_call( + self, + cache_key: str, + owner_id: str, + response: dict[str, Any], + config: dict[str, Any], + *, + now: datetime | None = None, + ) -> dict[str, Any]: + """Persist a valid typed response, or a redacted invalid verdict.""" + from .decision import validate_decision_response + + row = self.connection.execute( + "SELECT request_json FROM decision_cache WHERE cache_key=?", + (cache_key,), + ).fetchone() + if row is None: + raise ContractError("decision cache entry does not exist") + request = json.loads(row["request_json"]) + error = None + try: + normalized = validate_decision_response(request, response, config) + except ContractError as exc: + normalized = None + error = str(exc) + completed_at = _authoritative_now(now).isoformat() + if normalized is None: + status = "invalid" + response_json = None + response_sha256 = None + usage = { + "source": "unavailable", "billable_requests": 1, + "input_tokens": None, "output_tokens": None, "cost_usd": None, + } + else: + response_json = canonical_json(normalized) + response_sha256 = hashlib.sha256(response_json.encode()).hexdigest() + usage = normalized["usage"] + abstained = ( + normalized["truncation"]["occurred"] + or all( + recommendation["abstain_reason"] is not None + for recommendation in normalized["recommendations"].values() + ) + ) + status = "abstained" if abstained else "succeeded" + usage_json = canonical_json(usage) + self.connection.execute("BEGIN IMMEDIATE") + try: + current = self.connection.execute( + "SELECT status,owner_id,billable_calls FROM decision_cache " + "WHERE cache_key=?", (cache_key,), + ).fetchone() + if (current is None or current["status"] != "running" + or current["owner_id"] != owner_id + or current["billable_calls"] != 1): + raise ConflictError("decision call completion is fenced") + self.connection.execute( + "UPDATE decision_cache SET status=?,response_json=?," + "response_sha256=?,usage_json=?,error=?,updated_at=? " + "WHERE cache_key=? AND status='running' AND owner_id=?", + ( + status, response_json, response_sha256, usage_json, error, + completed_at, cache_key, owner_id, + ), + ) + self.connection.execute("COMMIT") + return { + "cache_key": cache_key, + "status": status, + "response": normalized, + "response_sha256": response_sha256, + "usage": usage, + "error": error, + "billable_calls": 1, + "completed_at": completed_at, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def finish_decision_without_call( + self, + cache_key: str, + owner_id: str, + status: str, + reason: str, + *, + now: datetime | None = None, + ) -> dict[str, Any]: + """Record unavailable/cancelled preprocessing without inventing usage.""" + if status not in {"unavailable", "cancelled"}: + raise ContractError("decision no-call status is invalid") + if not isinstance(reason, str) or not reason: + raise ContractError("decision no-call reason is invalid") + recorded_at = _authoritative_now(now).isoformat() + usage = { + "source": "unavailable", "billable_requests": 0, + "input_tokens": None, "output_tokens": None, "cost_usd": None, + } + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT status,owner_id,billable_calls FROM decision_cache " + "WHERE cache_key=?", (cache_key,), + ).fetchone() + if (row is None or row["status"] != "reserved" + or row["owner_id"] != owner_id or row["billable_calls"] != 0): + raise ConflictError("decision no-call completion is fenced") + self.connection.execute( + "UPDATE decision_cache SET status=?,usage_json=?,error=?,updated_at=? " + "WHERE cache_key=? AND status='reserved' AND owner_id=?", + ( + status, canonical_json(usage), reason, recorded_at, + cache_key, owner_id, + ), + ) + self.connection.execute("COMMIT") + return { + "cache_key": cache_key, "status": status, + "billable_calls": 0, "usage": usage, "error": reason, + "recorded_at": recorded_at, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def cancel_decision_observation( + self, cache_key: str, owner_id: str, *, now: datetime | None = None, + ) -> dict[str, Any]: + """Cancel before launch, or preserve uncertainty after a launched call.""" + cancelled_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT status,billable_calls,owner_id FROM decision_cache " + "WHERE cache_key=?", (cache_key,), + ).fetchone() + if row is None: + raise ContractError("decision cache entry does not exist") + if row["owner_id"] != owner_id: + raise ConflictError("decision cancellation owner changed") + if row["status"] not in {"reserved", "running"}: + self.connection.execute("COMMIT") + return { + "cache_key": cache_key, "status": row["status"], + "billable_calls": row["billable_calls"], "replayed": True, + } + status = "cancelled" if row["status"] == "reserved" else "indeterminate" + reason = ( + "decision call cancelled before launch" + if status == "cancelled" + else "decision call cancelled after launch; outcome is unknown" + ) + usage = { + "source": "unavailable", + "billable_requests": row["billable_calls"], + "input_tokens": None, "output_tokens": None, "cost_usd": None, + } + self.connection.execute( + "UPDATE decision_cache SET status=?,usage_json=?,error=?,updated_at=? " + "WHERE cache_key=? AND status IN ('reserved','running')", + ( + status, canonical_json(usage), reason, cancelled_at, + cache_key, + ), + ) + self.connection.execute("COMMIT") + return { + "cache_key": cache_key, "status": status, + "billable_calls": row["billable_calls"], "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def record_decision_effect( + self, + run_id: str, + purpose_id: str, + cache_key: str, + effect: dict[str, Any], + *, + now: datetime | None = None, + ) -> dict[str, Any]: + """Bind the frozen routing effect to this run without changing cache data.""" + if not isinstance(effect, dict): + raise ContractError("run decision effect must be an object") + effect_json = canonical_json(effect) + applied = bool(effect.get("applied_roles")) + recorded_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT cache_key,applied,effect_json FROM run_decision_observations " + "WHERE run_id=? AND purpose_id=?", + (run_id, purpose_id), + ).fetchone() + if row is None or row["cache_key"] != cache_key: + raise ConflictError("run decision observation binding changed") + if row["effect_json"] is not None: + if (row["effect_json"] != effect_json + or bool(row["applied"]) != applied): + raise ConflictError("run decision effect was recorded differently") + observation = self._decision_observation_row(run_id, purpose_id) + self.connection.execute("COMMIT") + return self._decision_observation_payload( + observation, action="cached", replayed=True, + ) + self.connection.execute( + "UPDATE run_decision_observations SET applied=?,effect_json=?," + "recorded_at=? WHERE run_id=? AND purpose_id=?", + ( + int(applied), effect_json, recorded_at, run_id, purpose_id, + ), + ) + observation = self._decision_observation_row(run_id, purpose_id) + self.connection.execute("COMMIT") + return self._decision_observation_payload( + observation, action="recorded", replayed=False, + ) + except Exception: + self.connection.execute("ROLLBACK") + raise + + def decision_observation( + self, run_id: str, purpose_id: str, + ) -> dict[str, Any] | None: + row = self._decision_observation_row(run_id, purpose_id) + if row is None: + return None + return self._decision_observation_payload( + row, action="read", replayed=True, + ) + + def decision_observations_for_run(self, run_id: str) -> list[dict[str, Any]]: + purposes = self.connection.execute( + "SELECT purpose_id FROM run_decision_observations " + "WHERE run_id=? ORDER BY purpose_id", + (run_id,), + ).fetchall() + return [ + self._decision_observation_payload( + self._decision_observation_row(run_id, row["purpose_id"]), + action="read", replayed=True, + ) + for row in purposes + ] + + def cancel_run_decision_observations( + self, run_id: str, *, now: datetime | None = None, + ) -> list[dict[str, Any]]: + """Fence every pending helper call when its owning run is cancelled.""" + cancelled_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + rows = self.connection.execute( + "SELECT DISTINCT c.cache_key,c.status,c.billable_calls " + "FROM run_decision_observations o JOIN decision_cache c " + "ON c.cache_key=o.cache_key WHERE o.run_id=? " + "AND c.status IN ('reserved','running')", + (run_id,), + ).fetchall() + results = [] + for row in rows: + status = ( + "cancelled" if row["status"] == "reserved" + else "indeterminate" + ) + reason = ( + "owning run cancelled before decision launch" + if status == "cancelled" + else "owning run cancelled after decision launch; outcome is unknown" + ) + usage = { + "source": "unavailable", + "billable_requests": row["billable_calls"], + "input_tokens": None, + "output_tokens": None, + "cost_usd": None, + } + self.connection.execute( + "UPDATE decision_cache SET status=?,usage_json=?,error=?," + "updated_at=? WHERE cache_key=? AND status=?", + ( + status, canonical_json(usage), reason, cancelled_at, + row["cache_key"], row["status"], + ), + ) + results.append({ + "cache_key": row["cache_key"], + "status": status, + "billable_calls": row["billable_calls"], + }) + self.connection.execute("COMMIT") + return results + except Exception: + self.connection.execute("ROLLBACK") + raise + + def project_final_outcome(self, run_id: str) -> dict[str, Any] | None: + """Idempotently drain a new run's durable terminal projection request.""" + from .objective_outcomes import project_outcome + job = self.connection.execute("SELECT completed_outcome_id FROM objective_outcome_jobs WHERE run_id=?", (run_id,)).fetchone() + if job is None or job["completed_outcome_id"] is not None: + return None + self.connection.execute("BEGIN") + try: + run = self.run(run_id) + if run["state"] not in TERMINAL_STATES: + self.connection.execute("COMMIT") + return None + refs, documents = {}, {} + for artifact in self.artifacts_for_run(run_id): + content = Path(artifact["path"]).read_bytes() + if hashlib.sha256(content).hexdigest() != artifact["sha256"]: + raise ConflictError("objective projection artifact integrity is invalid") + refs[artifact["name"]] = f"artifact:{artifact['id']}:{artifact['sha256']}" + if artifact["name"] in {"receipt.json", "result-receipt.json"}: + documents[artifact["name"]] = json.loads(content) + receipt = documents.get("receipt.json", documents.get("result-receipt.json")) + if not isinstance(receipt, dict): + raise ConflictError("objective projection requires a durable terminal receipt") + row = self.connection.execute("SELECT assignment_json FROM experiment_assignments WHERE run_id=?", (run_id,)).fetchone() + assignment = json.loads(row["assignment_json"]) if row else None + outcome = project_outcome(run, receipt, self.attempts_for_run(run_id), refs, assignment) + self.connection.execute("COMMIT") + except Exception: + self.connection.execute("ROLLBACK") + raise + result = self.record_outcome(run_id, outcome, _objective_projection=True) + self.connection.execute("UPDATE objective_outcome_jobs SET completed_outcome_id=? WHERE run_id=? AND completed_outcome_id IS NULL", (outcome["outcome_id"], run_id)) + return result + + def record_outcome( + self, + run_id: str, + outcome: dict[str, Any], + *, + now: datetime | None = None, + _objective_projection: bool = False, + ) -> dict[str, Any]: + """Append one replay-safe final outcome or late correction.""" + from .learning import validate_outcome + + current = _authoritative_now(now) + normalized = validate_outcome(outcome, now=current) + payload = canonical_json(normalized) + digest = hashlib.sha256(payload.encode()).hexdigest() + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = self.connection.execute( + "SELECT run_id,payload_json,recorded_at FROM outcomes WHERE outcome_id=?", + (normalized["outcome_id"],), + ).fetchone() + if existing is not None: + if existing["run_id"] != run_id or existing["payload_json"] != payload: + raise ConflictError( + "outcome id was already used with different evidence", + ) + self.connection.execute("COMMIT") + return { + "run_id": run_id, + "outcome": normalized, + "recorded_at": existing["recorded_at"], + "replayed": True, + } + run = self.connection.execute( + "SELECT state,mutable_snapshot FROM runs WHERE id=?", (run_id,), + ).fetchone() + if run is None: + raise ContractError("run does not exist") + if run["state"] not in TERMINAL_STATES: + raise ConflictError("outcomes require a terminal run") + try: + snapshot = json.loads(run["mutable_snapshot"] or "null") + roles = snapshot["routing"]["roles"].values() + expected_selection_mode = ( + "experimental" + if snapshot.get("experiment_assignment") is not None + else "pinned" + if any(role.get("source") == "override" for role in roles) + else "automatic" + ) + except (KeyError, TypeError, json.JSONDecodeError): + expected_selection_mode = None + if (expected_selection_mode is not None + and normalized["selection_mode"] != expected_selection_mode): + raise ConflictError("outcome selection mode does not match frozen routing") + if normalized["kind"] == "final": + if not _objective_projection and self.connection.execute("SELECT 1 FROM objective_outcome_jobs WHERE run_id=?", (run_id,)).fetchone(): + raise ConflictError("public final outcomes are objective projections; use an explicit late correction") + if normalized["verdict"] != run["state"]: + raise ConflictError("final outcome verdict does not match run state") + else: + corrected = self.connection.execute( + "SELECT run_id,kind,selection_mode,observed_at FROM outcomes " + "WHERE outcome_id=?", + (normalized["corrects_outcome_id"],), + ).fetchone() + if (corrected is None or corrected["run_id"] != run_id + or corrected["kind"] != "final"): + raise ConflictError("late outcome must correct this run's final outcome") + if corrected["selection_mode"] != normalized["selection_mode"]: + raise ConflictError("late outcome selection mode changed") + if datetime.fromisoformat(normalized["observed_at"]) < datetime.fromisoformat( + corrected["observed_at"], + ): + raise ConflictError("late outcome predates the final outcome") + + attempts = { + row["id"]: row + for row in self.connection.execute( + "SELECT id,role,status,output_metadata FROM attempts WHERE run_id=?", + (run_id,), + ) + } + for contribution in normalized["contributions"]: + attempt = attempts.get(contribution["attempt_id"]) + if attempt is None or attempt["role"] != contribution["role"]: + raise ConflictError("outcome contribution does not match run attempt") + if attempt["status"] != "finished": + raise ConflictError("outcome contribution attempt is not finished") + if attempt["output_metadata"]: + try: + metadata = json.loads(attempt["output_metadata"]) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("attempt output metadata is invalid") from exc + if (isinstance(metadata, dict) and metadata.get("failure") is not None + and (contribution["result"] != "failed" + or contribution["independent_success"])): + raise ConflictError( + "failed attempt cannot receive successful contribution credit", + ) + for repair in normalized["lead_repairs"]: + attempt_id = repair["lead_attempt_id"] + if attempt_id is None: + continue + attempt = attempts.get(attempt_id) + if attempt is None or attempt["role"] != "lead" or attempt["status"] != "finished": + raise ConflictError("lead repair does not match a finished lead attempt") + + recorded_at = current.isoformat() + self.connection.execute( + "INSERT INTO outcomes(outcome_id,run_id,kind,verdict,selection_mode," + "observed_at,corrects_outcome_id,payload_json,payload_sha256,recorded_at) " + "VALUES(?,?,?,?,?,?,?,?,?,?)", + ( + normalized["outcome_id"], run_id, normalized["kind"], + normalized["verdict"], normalized["selection_mode"], + normalized["observed_at"], normalized["corrects_outcome_id"], + payload, digest, recorded_at, + ), + ) + self.connection.execute("COMMIT") + return { + "run_id": run_id, + "outcome": normalized, + "recorded_at": recorded_at, + "replayed": False, + } + except sqlite3.IntegrityError as exc: + self.connection.execute("ROLLBACK") + raise ConflictError("run already has a final outcome") from exc + except Exception: + self.connection.execute("ROLLBACK") + raise + + def outcomes_for_run(self, run_id: str) -> list[dict[str, Any]]: + # Repair only the requested run. Corrupt evidence in another pending + # projection must not prevent opening the ledger or observing good runs. + self.project_final_outcome(run_id) + if not self.connection.execute( + "SELECT 1 FROM runs WHERE id=?", (run_id,), + ).fetchone(): + raise ContractError("run does not exist") + return [ + { + "run_id": row["run_id"], + "outcome": json.loads(row["payload_json"]), + "payload_sha256": row["payload_sha256"], + "recorded_at": row["recorded_at"], + } + for row in self.connection.execute( + "SELECT run_id,payload_json,payload_sha256,recorded_at " + "FROM outcomes WHERE run_id=? ORDER BY observed_at,id", + (run_id,), + ) + ] + + def repair_project_outcomes(self, project: Path) -> None: + """Repair before, never inside, a caller's consistent read transaction.""" + if self.connection.in_transaction: + raise ConflictError("project outcome repair requires an independent transaction") + common = str(git_common_dir(project)) + for job in self.connection.execute( + "SELECT j.run_id FROM objective_outcome_jobs j JOIN runs r ON r.id=j.run_id " + "JOIN projects p ON p.id=r.project_id WHERE p.git_common_dir=? " + "AND j.completed_outcome_id IS NULL AND r.state IN ('succeeded','failed','cancelled')", + (common,)).fetchall(): + self.project_final_outcome(job["run_id"]) + + def learning_report( + self, project: Path, *, now: datetime | None = None, _repair_pending: bool = True, + ) -> dict[str, Any]: + """Repair pending public projections and compare with explicit missingness.""" + from .learning import build_comparison_report + + if _repair_pending: + self.repair_project_outcomes(project) + common_dir = git_common_dir(project) + project_row = self.connection.execute( + "SELECT id FROM projects WHERE git_common_dir=?", (str(common_dir),), + ).fetchone() + if project_row is None: + terminal_runs = [] + outcome_records = [] + attempt_profiles = {} + project_id = None + else: + project_id = project_row["id"] + terminal_runs = [ + {"run_id": row["id"], "state": row["state"]} + for row in self.connection.execute( + "SELECT id,state FROM runs WHERE project_id=? " + "AND state IN ('succeeded','failed','cancelled') ORDER BY created_at,id", + (project_id,), + ) + ] + outcome_records = [ + {"run_id": row["run_id"], "outcome": json.loads(row["payload_json"])} + for row in self.connection.execute( + "SELECT o.run_id,o.payload_json FROM outcomes o " + "JOIN runs r ON r.id=o.run_id WHERE r.project_id=? " + "ORDER BY o.observed_at,o.id", + (project_id,), + ) + ] + attempt_profiles = { + row["id"]: row["profile_id"] + for row in self.connection.execute( + "SELECT a.id,a.profile_id FROM attempts a " + "JOIN runs r ON r.id=a.run_id WHERE r.project_id=?", + (project_id,), + ) + } + return build_comparison_report( + project_id=project_id, + project_path=str(project.resolve(strict=True)), + terminal_runs=terminal_runs, + outcome_records=outcome_records, + attempt_profiles=attempt_profiles, + generated_at=_authoritative_now(now).isoformat(), + ) + + def evaluate_learning_experiment( + self, experiment: dict[str, Any], *, now: datetime | None = None, + revision_id: str | None = None, + previous_evaluation_sha256: str | None = None, + ) -> dict[str, Any]: + """Persist an original evaluation or an explicit append-only review.""" + from .experiment_eligibility import current_evidence, saved_evaluation + from .experiment_provenance import require_sha256 + from .learning import evaluate_experiment, validate_experiment + + if (revision_id is None) != (previous_evaluation_sha256 is None): + raise ContractError("evaluation revision id and predecessor must be supplied together") + if revision_id is not None: + if not isinstance(revision_id, str) or not revision_id.strip() or len(revision_id) > 128: + raise ContractError("evaluation revision id is invalid") + require_sha256(previous_evaluation_sha256, "previous evaluation") + spec = validate_experiment(experiment) + project_path = Path(spec["project_path"]).resolve(strict=True) + # V2 hashes the exact predeclared specification. Resolving a symlink + # for project lookup must not rewrite that declaration after launch. + if spec["schema_version"] == 1: + spec = {**spec, "project_path": str(project_path)} + spec_json = canonical_json(spec) + spec_sha256 = hashlib.sha256(spec_json.encode()).hexdigest() + current = _authoritative_now(now) + common_dir = git_common_dir(project_path) + for assigned in self.connection.execute( + "SELECT run_id FROM experiment_assignments WHERE experiment_id=?", + (spec["experiment_id"],)).fetchall(): + self.project_final_outcome(assigned["run_id"]) + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = saved_evaluation(self.connection, spec["experiment_id"]) + if existing is not None and existing["spec_json"] != spec_json: + raise ConflictError( + "experiment id was already used with a different specification", + ) + if revision_id is not None: + if existing is None or spec["schema_version"] != 2: + raise ContractError("explicit evaluation review requires an original v2 evaluation") + revision = self.connection.execute( + "SELECT experiment_id,previous_evaluation_sha256,evaluation_sha256 " + "FROM experiment_evaluation_revisions WHERE revision_id=?", (revision_id,), + ).fetchone() + if revision is not None: + if (revision["experiment_id"] != spec["experiment_id"] + or revision["previous_evaluation_sha256"] != previous_evaluation_sha256): + raise ConflictError("evaluation revision id was already used with different evidence") + existing = saved_evaluation(self.connection, spec["experiment_id"], revision["evaluation_sha256"]) + else: + if existing["evaluation_sha256"] != previous_evaluation_sha256: + raise ConflictError("evaluation revision predecessor is no longer the latest evaluation") + existing = None + if existing is not None: + eligibility = current_evidence(self.connection, existing, now=current) + if spec["schema_version"] == 2 and not eligibility["eligible"]: + raise ContractError("experiment evidence changed; saved evaluation is stale and requires explicit review/revision: " + ", ".join(eligibility["reasons"])) + self.connection.execute("COMMIT") + return { + "experiment": spec, + "evaluation": existing["evaluation"], + "evaluation_sha256": existing["evaluation_sha256"], + "recorded_at": existing["recorded_at"], + "revision_id": existing["revision_id"], + "previous_evaluation_sha256": existing["previous_evaluation_sha256"], + "eligibility": eligibility, + "replayed": True, + } + project = self.connection.execute( + "SELECT id FROM projects WHERE git_common_dir=?", (str(common_dir),), + ).fetchone() + project_id = project["id"] if project is not None else None + chains: dict[str, dict[str, Any]] = {} + if spec["schema_version"] == 2: + from .experiment_evidence import read_experiment_chains + + chains = read_experiment_chains( + self.connection, spec=spec, project_id=project_id, + project_common_dir=str(common_dir), evaluated_at=current.isoformat(), + ) + else: + # Historical v1 decoding remains separate from v2 authority. + records = [] + if project_id is not None: + records = self.connection.execute( + "SELECT o.payload_json FROM outcomes o " + "JOIN runs r ON r.id=o.run_id WHERE r.project_id=? " + "ORDER BY o.observed_at,o.id", + (project_id,), + ).fetchall() + corrections: dict[str, list[dict[str, Any]]] = {} + for record in records: + outcome = json.loads(record["payload_json"]) + if outcome["kind"] == "final": + chains[outcome["outcome_id"]] = { + "final": outcome, "late_corrections": [], + } + else: + corrections.setdefault( + outcome["corrects_outcome_id"], [], + ).append(outcome) + for outcome_id, history in corrections.items(): + if outcome_id in chains: + chains[outcome_id]["late_corrections"] = history + evaluation = evaluate_experiment( + spec, chains, evaluated_at=current.isoformat(), + project_common_dir=str(common_dir), + ) + evaluation_json = canonical_json(evaluation) + evaluation_sha256 = hashlib.sha256(evaluation_json.encode()).hexdigest() + recorded_at = current.isoformat() + if revision_id is None: + self.connection.execute( + "INSERT INTO experiments(experiment_id,project_id,project_path,spec_json," + "spec_sha256,evaluation_json,evaluation_sha256,verdict,recorded_at) " + "VALUES(?,?,?,?,?,?,?,?,?)", + ( + spec["experiment_id"], project_id, spec["project_path"], + spec_json, spec_sha256, evaluation_json, evaluation_sha256, + evaluation["verdict"], recorded_at, + ), + ) + else: + if saved_evaluation(self.connection, spec["experiment_id"], evaluation_sha256) is not None: + raise ConflictError("evaluation review must create a distinct immutable receipt") + self.connection.execute( + "INSERT INTO experiment_evaluation_revisions(revision_id,experiment_id," + "previous_evaluation_sha256,evaluation_json,evaluation_sha256,verdict,recorded_at) " + "VALUES(?,?,?,?,?,?,?)", + (revision_id, spec["experiment_id"], previous_evaluation_sha256, + evaluation_json, evaluation_sha256, evaluation["verdict"], recorded_at), + ) + saved = saved_evaluation(self.connection, spec["experiment_id"], evaluation_sha256) + eligibility = current_evidence(self.connection, saved, now=current) + self.connection.execute("COMMIT") + return { + "experiment": spec, + "evaluation": evaluation, + "evaluation_sha256": evaluation_sha256, + "recorded_at": recorded_at, + "revision_id": revision_id, + "previous_evaluation_sha256": previous_evaluation_sha256, + "eligibility": eligibility, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def learning_proposal_inputs( + self, project: Path, *, now: datetime | None = None, + ) -> dict[str, Any]: + """Read a consistent report and the latest frozen project experiment.""" + from .experiment_eligibility import current_evidence, saved_evaluation + project_path = project.resolve(strict=True) + current = _authoritative_now(now) + self.repair_project_outcomes(project_path) + self.connection.execute("BEGIN") + try: + report = self.learning_report(project_path, now=current, _repair_pending=False) + if report["project_id"] is None: + row = self.connection.execute( + "SELECT experiment_id FROM experiments WHERE project_path=? " + "ORDER BY recorded_at DESC,id DESC LIMIT 1", + (str(project_path),), + ).fetchone() + else: + row = self.connection.execute( + "SELECT experiment_id FROM experiments WHERE project_id=? OR " + "(project_id IS NULL AND project_path=?) " + "ORDER BY recorded_at DESC,id DESC LIMIT 1", + (report["project_id"], str(project_path)), + ).fetchone() + experiment = None + if row is not None: + saved = saved_evaluation(self.connection, row["experiment_id"]) + experiment = { + "experiment": saved["spec"], "spec_sha256": saved["spec_sha256"], + "evaluation": saved["evaluation"], "evaluation_sha256": saved["evaluation_sha256"], + "recorded_at": saved["recorded_at"], + "eligibility": current_evidence(self.connection, saved, now=current), + } + self.connection.execute("COMMIT") + return {"report": report, "experiment": experiment} + except Exception: + self.connection.execute("ROLLBACK") + raise + + @staticmethod + def _profile_record(profile: dict[str, Any]) -> tuple[str, str]: + from .lifecycle import profile_fingerprint + + payload = canonical_json(profile) + return payload, profile_fingerprint(profile) + + def _insert_concrete_profile( + self, profile: dict[str, Any], recorded_at: str, + ) -> tuple[str, str]: + payload, digest = self._profile_record(profile) + existing = self.connection.execute( + "SELECT profile_json,profile_sha256 FROM concrete_profiles " + "WHERE profile_id=?", + (profile["id"],), + ).fetchone() + if existing is not None: + if (existing["profile_json"] != payload + or existing["profile_sha256"] != digest): + raise ConflictError( + "profile id was already used with different concrete settings", + ) + return payload, digest + self.connection.execute( + "INSERT INTO concrete_profiles(profile_id,profile_json,profile_sha256," + "recorded_at) VALUES(?,?,?,?)", + (profile["id"], payload, digest, recorded_at), + ) + return payload, digest + + def _insert_profile_template( + self, template: dict[str, Any], recorded_at: str, + ) -> tuple[str, str]: + payload = canonical_json(template) + digest = hashlib.sha256(payload.encode()).hexdigest() + existing = self.connection.execute( + "SELECT payload_json,payload_sha256 FROM profile_templates " + "WHERE template_id=?", + (template["template_id"],), + ).fetchone() + if existing is not None: + if (existing["payload_json"] != payload + or existing["payload_sha256"] != digest): + raise ConflictError( + "template id was already used with a different policy", + ) + return payload, digest + self.connection.execute( + "INSERT INTO profile_templates(template_id,alias,update_mode,policy_id," + "policy_version,payload_json,payload_sha256,recorded_at) " + "VALUES(?,?,?,?,?,?,?,?)", + ( + template["template_id"], template["alias"], + template["update_mode"], template["policy"]["id"], + template["policy"]["version"], payload, digest, recorded_at, + ), + ) + return payload, digest + + def register_profile_template( + self, template: dict[str, Any], *, now: datetime | None = None, + ) -> dict[str, Any]: + """Register an immutable reviewed lifecycle template.""" + from .lifecycle import validate_profile_template + + normalized = validate_profile_template(template) + recorded_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + before = self.connection.execute( + "SELECT 1 FROM profile_templates WHERE template_id=?", + (normalized["template_id"],), + ).fetchone() + _, digest = self._insert_profile_template(normalized, recorded_at) + self.connection.execute("COMMIT") + return { + "template": normalized, + "template_sha256": digest, + "recorded_at": recorded_at, + "replayed": before is not None, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def bootstrap_profile_binding( + self, + template: dict[str, Any], + profile: dict[str, Any], + *, + version: int, + now: datetime | None = None, + ) -> dict[str, Any]: + """Import one reviewed baseline binding without fabricating qualification.""" + from .lifecycle import ( + profile_template_violation, + validate_profile_template, + ) + + normalized_template = validate_profile_template(template) + profile_payload, profile_digest = self._profile_record(profile) + profile = json.loads(profile_payload) + if type(version) is not int or version < 1: + raise ContractError("baseline binding version must be positive") + violation = profile_template_violation(profile, normalized_template) + if violation is not None: + raise ContractError(f"baseline profile violates template: {violation}") + if profile["quality_status"] != "proven": + raise ContractError("baseline profile must already be proven") + recorded_at = _authoritative_now(now).isoformat() + alias = normalized_template["alias"] + self.connection.execute("BEGIN IMMEDIATE") + try: + self._insert_profile_template(normalized_template, recorded_at) + self._insert_concrete_profile(profile, recorded_at) + existing = self.connection.execute( + "SELECT template_id,profile_id,qualification_id,version,updated_at " + "FROM profile_bindings WHERE alias=?", + (alias,), + ).fetchone() + if existing is not None: + if (existing["template_id"] != normalized_template["template_id"] + or existing["profile_id"] != profile["id"] + or existing["qualification_id"] is not None + or existing["version"] != version): + raise ConflictError( + "baseline alias is already bound differently", + ) + self.connection.execute("COMMIT") + return { + "binding": dict(existing), + "profile_sha256": profile_digest, + "replayed": True, + } + self.connection.execute( + "INSERT INTO profile_bindings(alias,template_id,profile_id," + "qualification_id,version,updated_at) VALUES(?,?,?,NULL,?,?)", + ( + alias, normalized_template["template_id"], profile["id"], + version, recorded_at, + ), + ) + self.connection.execute( + "INSERT INTO profile_binding_versions(alias,version,template_id," + "profile_id,qualification_id,decision_id,recorded_at) " + "VALUES(?,?,?,?,NULL,NULL,?)", + ( + alias, version, normalized_template["template_id"], + profile["id"], recorded_at, + ), + ) + self.connection.execute("COMMIT") + return { + "binding": { + "alias": alias, + "template_id": normalized_template["template_id"], + "profile_id": profile["id"], + "qualification_id": None, + "version": version, + "updated_at": recorded_at, + }, + "profile_sha256": profile_digest, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def _qualification_evidence( + self, record: dict[str, Any], *, now: datetime, + ) -> tuple[list[str], dict[str, Any] | None]: + """The same current-evidence and context checks for every consumer.""" + from .experiment_eligibility import require_current_evidence + from .lifecycle import qualification_gate_failures, validate_profile_template, profile_fingerprint + + template_row = self.connection.execute( + "SELECT payload_json,payload_sha256 FROM profile_templates WHERE template_id=?", + (record["template_id"],), + ).fetchone() + if template_row is None: + raise ContractError("qualification template is not registered") + if hashlib.sha256(template_row["payload_json"].encode()).hexdigest() != template_row["payload_sha256"]: + raise ContractError("qualification template hash is invalid") + template = validate_profile_template(json.loads(template_row["payload_json"])) + failures = qualification_gate_failures(record, template) + eligibility = None + if record["experiment_id"] is not None: + experiment, eligibility = require_current_evidence( + self.connection, record["experiment_id"], record["evaluation_sha256"], now=now, + ) + spec, evaluation = experiment["spec"], experiment["evaluation"] + variable = spec["variable"] + if (variable["alias"] != record["alias"] + or variable["candidate_profile_id"] != record["candidate_profile"]["id"] + or variable["candidate_profile_sha256"] != profile_fingerprint(record["candidate_profile"])): + raise ContractError("qualification candidate fingerprint does not match the tested experiment") + contexts = self.connection.execute( + "SELECT assignment_json,snapshot_json FROM experiment_assignments " + "WHERE experiment_id=? AND arm='candidate'", (record["experiment_id"],), + ).fetchall() + if not contexts or any( + json.loads(row["assignment_json"])["role"] != variable["role"] + or json.loads(row["snapshot_json"])["task"]["task_class"] != record["task_class"] + for row in contexts): + raise ContractError("qualification role/task-class context does not match actual frozen tasks") + metrics = evaluation["metrics"] + escaped = sum(metrics[split]["candidate_escaped_defects"] for split in ("evaluation", "held_out")) + if (record["measured"]["evaluation_pairs"] != metrics["evaluation"]["available_pairs"] + or record["measured"]["held_out_pairs"] != metrics["held_out"]["available_pairs"] + or record["measured"]["critical_defects"] != escaped): + raise ContractError("qualification measurements do not match saved evaluation") + # Saved evaluations currently contain no verified paired whole-run + # latency or usage measurement. UTC attempt timestamps are not + # monotonic latency, and partial native token reports are not a + # complete paired usage measure. Preserve unknown rather than + # letting caller-supplied ratios grant lifecycle authority. + if any(record["measured"][field] is not None for field in ("latency_ratio", "usage_ratio")): + raise ContractError("qualification measurements contain unmeasured latency/usage ratios") + if evaluation["verdict"] != "promotion_proposal": + failures.append("experiment_did_not_propose_promotion") + failures = sorted(set(failures)) + if record["verdict"] == "qualified" and failures: + raise ContractError("qualification gate did not pass: " + ", ".join(failures)) + return failures, eligibility + + def _require_current_qualification( + self, qualification_id: str, *, now: datetime, + ) -> dict[str, Any]: + from .lifecycle import validate_qualification + + row = self.connection.execute( + "SELECT * FROM qualification_runs WHERE qualification_id=?", (qualification_id,), + ).fetchone() + if row is None or row["verdict"] != "qualified": + raise ContractError("binding change requires a qualified candidate") + if hashlib.sha256(row["payload_json"].encode()).hexdigest() != row["payload_sha256"]: + raise ContractError("saved qualification hash is invalid") + record = validate_qualification(json.loads(row["payload_json"])) + if any(row[key] != record[key] for key in ( + "qualification_id", "alias", "template_id", "experiment_id", "evaluation_sha256", "verdict")) or row["profile_id"] != record["candidate_profile"]["id"]: + raise ContractError("saved qualification identity is invalid") + failures, _ = self._qualification_evidence(record, now=now) + if failures or canonical_json(failures) != row["gate_failures_json"]: + raise ContractError("saved qualification gates are inconsistent") + profile_row = self.connection.execute( + "SELECT profile_json,profile_sha256 FROM concrete_profiles WHERE profile_id=?", (row["profile_id"],), + ).fetchone() + payload, digest = self._profile_record(record["candidate_profile"]) + if profile_row is None or profile_row["profile_json"] != payload or profile_row["profile_sha256"] != digest: + raise ContractError("saved qualification concrete profile differs from tested evidence") + return record + + def record_profile_qualification( + self, qualification: dict[str, Any], *, now: datetime | None = None, + ) -> dict[str, Any]: + """Persist bounded qualification evidence after checking saved evaluation.""" + from .lifecycle import validate_qualification + + record = validate_qualification(qualification) + payload = canonical_json(record) + digest = hashlib.sha256(payload.encode()).hexdigest() + recorded_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = self.connection.execute( + "SELECT payload_json,payload_sha256,gate_failures_json,recorded_at " + "FROM qualification_runs WHERE qualification_id=?", + (record["qualification_id"],), + ).fetchone() + if existing is not None: + if existing["payload_json"] != payload: + raise ConflictError( + "qualification id was already used with different evidence", + ) + failures, eligibility = self._qualification_evidence(record, now=_authoritative_now(now)) + if (existing["payload_sha256"] != digest + or canonical_json(failures) != existing["gate_failures_json"]): + raise ContractError("saved qualification evidence hash/gates are invalid") + self.connection.execute("COMMIT") + return { + "qualification": record, + "qualification_sha256": existing["payload_sha256"], + "gate_failures": json.loads(existing["gate_failures_json"]), + "recorded_at": existing["recorded_at"], + "eligibility": eligibility, + "replayed": True, + } + failures, eligibility = self._qualification_evidence(record, now=_authoritative_now(now)) + self._insert_concrete_profile( + record["candidate_profile"], recorded_at, + ) + self.connection.execute( + "INSERT INTO qualification_runs(qualification_id,alias,template_id," + "profile_id,experiment_id,evaluation_sha256,verdict,payload_json," + "payload_sha256,gate_failures_json,recorded_at) " + "VALUES(?,?,?,?,?,?,?,?,?,?,?)", + ( + record["qualification_id"], record["alias"], + record["template_id"], record["candidate_profile"]["id"], + record["experiment_id"], record["evaluation_sha256"], + record["verdict"], payload, digest, + canonical_json(failures), recorded_at, + ), + ) + self.connection.execute("COMMIT") + return { + "qualification": record, + "qualification_sha256": digest, + "gate_failures": failures, + "recorded_at": recorded_at, + "eligibility": eligibility, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def profile_binding(self, alias: str) -> dict[str, Any] | None: + if not isinstance(alias, str) or not alias: + raise ContractError("profile binding alias is invalid") + row = self.connection.execute( + "SELECT b.alias,b.template_id,b.profile_id,b.qualification_id,b.version," + "b.updated_at,p.profile_json,p.profile_sha256,t.payload_json AS template_json," + "t.payload_sha256 AS template_sha256 FROM profile_bindings b " + "JOIN concrete_profiles p ON p.profile_id=b.profile_id " + "JOIN profile_templates t ON t.template_id=b.template_id " + "WHERE b.alias=?", + (alias,), + ).fetchone() + if row is None: + return None + return { + "alias": row["alias"], + "template_id": row["template_id"], + "profile_id": row["profile_id"], + "qualification_id": row["qualification_id"], + "version": row["version"], + "updated_at": row["updated_at"], + "profile": json.loads(row["profile_json"]), + "profile_sha256": row["profile_sha256"], + "template": json.loads(row["template_json"]), + "template_sha256": row["template_sha256"], + } + + def change_profile_binding( + self, change: dict[str, Any], *, now: datetime | None = None, + ) -> dict[str, Any]: + """CAS-promote or roll back one alias and persist its decision receipt.""" + from .lifecycle import ( + guarded_change_violation, + validate_binding_change, + ) + + request = validate_binding_change(change) + request_json = canonical_json(request) + request_sha256 = hashlib.sha256(request_json.encode()).hexdigest() + recorded_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = self.connection.execute( + "SELECT request_sha256,receipt_json,receipt_sha256,recorded_at " + "FROM binding_decisions WHERE decision_id=?", + (request["decision_id"],), + ).fetchone() + if existing is not None: + if existing["request_sha256"] != request_sha256: + raise ConflictError( + "decision id was already used with a different request", + ) + self.connection.execute("COMMIT") + return { + "receipt": json.loads(existing["receipt_json"]), + "receipt_sha256": existing["receipt_sha256"], + "recorded_at": existing["recorded_at"], + "replayed": True, + } + current = self.connection.execute( + "SELECT b.template_id,b.profile_id,b.qualification_id,b.version," + "p.profile_json,p.profile_sha256,t.payload_json,t.payload_sha256 " + "FROM profile_bindings b " + "JOIN concrete_profiles p ON p.profile_id=b.profile_id " + "JOIN profile_templates t ON t.template_id=b.template_id " + "WHERE b.alias=?", + (request["alias"],), + ).fetchone() + if current is None: + raise ContractError("profile binding does not exist") + if current["version"] != request["expected_binding_version"]: + raise ConflictError("profile binding version changed") + current_template = json.loads(current["payload_json"]) + current_profile = json.loads(current["profile_json"]) + if (request["actor"] == "guarded_auto" + and current_template["update_mode"] != "guarded_auto"): + raise ContractError( + "guarded automatic promotion is not enabled", + ) + qualification_payload = None + rollback_evaluation = None + if request["action"] == "promote": + qualification = self.connection.execute( + "SELECT q.alias,q.template_id,q.profile_id,q.verdict," + "q.payload_json,q.payload_sha256,q.gate_failures_json," + "p.profile_json,p.profile_sha256,t.payload_json AS template_json," + "t.payload_sha256 AS template_sha256 FROM qualification_runs q " + "JOIN concrete_profiles p ON p.profile_id=q.profile_id " + "JOIN profile_templates t ON t.template_id=q.template_id " + "WHERE q.qualification_id=?", + (request["qualification_id"],), + ).fetchone() + if (qualification is None or qualification["alias"] != request["alias"] + or qualification["verdict"] != "qualified" + or json.loads(qualification["gate_failures_json"])): + raise ContractError( + "binding promotion requires a qualified candidate", + ) + target_template = json.loads(qualification["template_json"]) + target_profile = json.loads(qualification["profile_json"]) + target_template_id = qualification["template_id"] + target_profile_sha256 = qualification["profile_sha256"] + target_template_sha256 = qualification["template_sha256"] + target_qualification_id = request["qualification_id"] + qualification_payload = self._require_current_qualification( + request["qualification_id"], now=_authoritative_now(now), + ) + from .experiment_eligibility import saved_evaluation + + tested = saved_evaluation( + self.connection, qualification_payload["experiment_id"], + qualification_payload["evaluation_sha256"], + )["spec"]["variable"] + if (tested["control_profile_id"] != current["profile_id"] + or tested["control_profile_sha256"] != current["profile_sha256"]): + raise ContractError("promotion evidence does not compare the current incumbent") + if (request["actor"] == "guarded_auto" + and target_template_id != current["template_id"]): + raise ContractError( + "guarded automation cannot change lifecycle policy", + ) + guarded_violation = ( + guarded_change_violation(current_profile, target_profile) + if request["actor"] == "guarded_auto" else None + ) + if guarded_violation is not None: + raise ContractError(guarded_violation) + else: + rollback = request["rollback_target"] + target = self.connection.execute( + "SELECT v.template_id,v.profile_id,v.qualification_id," + "p.profile_json,p.profile_sha256,t.payload_json AS template_json," + "t.payload_sha256 AS template_sha256 FROM profile_binding_versions v " + "JOIN concrete_profiles p ON p.profile_id=v.profile_id " + "JOIN profile_templates t ON t.template_id=v.template_id " + "WHERE v.alias=? AND v.version=?", + (request["alias"], rollback["binding_version"]), + ).fetchone() + if (target is None or target["profile_id"] != rollback["profile_id"] + or rollback["binding_version"] >= current["version"]): + raise ContractError("rollback target is not a prior binding") + target_profile = json.loads(target["profile_json"]) + if target_profile["quality_status"] == "suspended": + raise ContractError("rollback target is suspended") + target_template = json.loads(target["template_json"]) + target_template_id = target["template_id"] + target_profile_sha256 = target["profile_sha256"] + target_template_sha256 = target["template_sha256"] + target_qualification_id = target["qualification_id"] + if target_qualification_id is not None: + qualification_payload = self._require_current_qualification( + target_qualification_id, now=_authoritative_now(now), + ) + if (request["actor"] == "guarded_auto" + and target_template_id != current["template_id"]): + raise ContractError( + "guarded automation cannot change lifecycle policy", + ) + from .experiment_eligibility import require_current_evidence + + regression, regression_eligibility = require_current_evidence( + self.connection, request["experiment_id"], request["evaluation_sha256"], + now=_authoritative_now(now), + ) + if regression["verdict"] != "no_change": + raise ContractError( + "rollback requires saved no-change regression evidence", + ) + regression_spec = regression["spec"] + regression_evaluation = regression["evaluation"] + variable = regression_spec["variable"] + if (variable["alias"] != request["alias"] + or variable["candidate_profile_id"] != current["profile_id"] + or variable["control_profile_id"] != target_profile["id"] + or variable["candidate_profile_sha256"] != current["profile_sha256"] + or variable["control_profile_sha256"] != target_profile_sha256): + raise ContractError( + "rollback experiment does not compare the active and target profiles", + ) + if any( + regression_evaluation["metrics"][split]["available_pairs"] < regression_spec["gate"][minimum] + for split, minimum in (("evaluation", "min_evaluation_pairs"), ("held_out", "min_held_out_pairs"))): + raise ContractError("rollback requires complete evaluation and held-out regression pairs") + rollback_evaluation = { + "experiment_id": request["experiment_id"], + "evaluation_sha256": request["evaluation_sha256"], + "verdict": regression["verdict"], + "reasons": regression_evaluation["reasons"], + "metrics": regression_evaluation["metrics"], + "failures": regression_evaluation["failures"], + "eligibility": regression_eligibility, + } + if target_profile["id"] == current["profile_id"]: + raise ConflictError("binding already targets the requested profile") + new_version = current["version"] + 1 + receipt = { + "schema_version": 1, + "decision_id": request["decision_id"], + "action": request["action"], + "alias": request["alias"], + "actor": request["actor"], + "reason": request["reason"], + "evidence_refs": request["evidence_refs"], + "from": { + "binding_version": current["version"], + "profile_id": current["profile_id"], + "profile_sha256": current["profile_sha256"], + "template_id": current["template_id"], + "template_sha256": current["payload_sha256"], + }, + "to": { + "binding_version": new_version, + "profile_id": target_profile["id"], + "profile_sha256": target_profile_sha256, + "template_id": target_template_id, + "template_sha256": target_template_sha256, + }, + "qualification_id": target_qualification_id, + "qualification": qualification_payload, + "policy": target_template["policy"], + "rollback_target": { + "profile_id": current["profile_id"], + "binding_version": current["version"], + }, + "requested_rollback_target": ( + request["rollback_target"] + if request["action"] == "rollback" else None + ), + "rollback_evaluation": rollback_evaluation, + "effective_at": recorded_at, + "affects_new_runs_only": True, + } + receipt_json = canonical_json(receipt) + receipt_sha256 = hashlib.sha256(receipt_json.encode()).hexdigest() + self.connection.execute( + "UPDATE profile_bindings SET template_id=?,profile_id=?," + "qualification_id=?,version=?,updated_at=? WHERE alias=? AND version=?", + ( + target_template_id, target_profile["id"], + target_qualification_id, new_version, recorded_at, + request["alias"], current["version"], + ), + ) + if self.connection.execute("SELECT changes()").fetchone()[0] != 1: + raise ConflictError("profile binding version changed") + self.connection.execute( + "INSERT INTO profile_binding_versions(alias,version,template_id," + "profile_id,qualification_id,decision_id,recorded_at) " + "VALUES(?,?,?,?,?,?,?)", + ( + request["alias"], new_version, target_template_id, + target_profile["id"], target_qualification_id, + request["decision_id"], recorded_at, + ), + ) + self.connection.execute( + "INSERT INTO binding_decisions(decision_id,alias,action,actor," + "from_version,to_version,from_profile_id,to_profile_id," + "qualification_id,request_sha256,receipt_json,receipt_sha256," + "recorded_at) VALUES(?,?,?,?,?,?,?,?,?,?,?,?,?)", + ( + request["decision_id"], request["alias"], request["action"], + request["actor"], current["version"], new_version, + current["profile_id"], target_profile["id"], + target_qualification_id, request_sha256, receipt_json, + receipt_sha256, recorded_at, + ), + ) + self.connection.execute("COMMIT") + return { + "receipt": receipt, + "receipt_sha256": receipt_sha256, + "recorded_at": recorded_at, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def fallback_unavailable_profile_binding( + self, change: dict[str, Any], *, now: datetime | None = None, + ) -> dict[str, Any]: + """Roll back an unavailable incumbent to the latest safe predecessor.""" + from .lifecycle import ( + guarded_change_violation, + profile_template_violation, + validate_catalog_fallback, + ) + + request = validate_catalog_fallback(change) + request_json = canonical_json(request) + request_sha256 = hashlib.sha256(request_json.encode()).hexdigest() + recorded_at = _authoritative_now(now).isoformat() + catalog_change = request["catalog_change"] + catalog_change_sha256 = hashlib.sha256( + canonical_json(catalog_change).encode(), + ).hexdigest() + unavailable = set(catalog_change["unavailable_profile_ids"]) + removed_models = set(catalog_change["removed_model_ids"]) + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = self.connection.execute( + "SELECT request_sha256,receipt_json,receipt_sha256,recorded_at " + "FROM binding_decisions WHERE decision_id=?", + (request["decision_id"],), + ).fetchone() + if existing is not None: + if existing["request_sha256"] != request_sha256: + raise ConflictError( + "decision id was already used with a different request", + ) + self.connection.execute("COMMIT") + return { + "receipt": json.loads(existing["receipt_json"]), + "receipt_sha256": existing["receipt_sha256"], + "recorded_at": existing["recorded_at"], + "replayed": True, + } + current = self.connection.execute( + "SELECT b.template_id,b.profile_id,b.qualification_id,b.version," + "p.profile_json,p.profile_sha256,t.payload_json AS template_json," + "t.payload_sha256 AS template_sha256 FROM profile_bindings b " + "JOIN concrete_profiles p ON p.profile_id=b.profile_id " + "JOIN profile_templates t ON t.template_id=b.template_id " + "WHERE b.alias=?", + (request["alias"],), + ).fetchone() + if current is None: + raise ContractError("profile binding does not exist") + if current["version"] != request["expected_binding_version"]: + raise ConflictError("profile binding version changed") + current_profile = json.loads(current["profile_json"]) + current_template = json.loads(current["template_json"]) + if (current_profile["id"] not in unavailable + or current_profile["harness"] != catalog_change["harness"] + or current_profile["model_id"] not in removed_models): + raise ContractError( + "catalog evidence does not prove the incumbent unavailable", + ) + if (request["actor"] == "guarded_auto" + and current_template["update_mode"] != "guarded_auto"): + raise ContractError( + "guarded automatic promotion is not enabled", + ) + + target = None + target_profile = None + qualification_payload = None + for candidate in self.connection.execute( + "SELECT v.version,v.template_id,v.profile_id,v.qualification_id," + "p.profile_json,p.profile_sha256,t.payload_json AS template_json," + "t.payload_sha256 AS template_sha256,q.verdict AS qualification_verdict," + "q.payload_json AS qualification_json," + "q.gate_failures_json AS qualification_failures " + "FROM profile_binding_versions v " + "JOIN concrete_profiles p ON p.profile_id=v.profile_id " + "JOIN profile_templates t ON t.template_id=v.template_id " + "LEFT JOIN qualification_runs q " + "ON q.qualification_id=v.qualification_id " + "WHERE v.alias=? AND v.version list[dict[str, Any]]: + if not isinstance(alias, str) or not alias: + raise ContractError("profile binding alias is invalid") + return [ + { + "receipt": json.loads(row["receipt_json"]), + "receipt_sha256": row["receipt_sha256"], + "recorded_at": row["recorded_at"], + } + for row in self.connection.execute( + "SELECT receipt_json,receipt_sha256,recorded_at " + "FROM binding_decisions WHERE alias=? ORDER BY to_version", + (alias,), + ) + ] + + def effective_profile_registry( + self, + profiles_payload: bytes | str, + policy_payload: bytes | str, + ) -> dict[str, Any]: + """Overlay policy-matched local bindings without editing project files.""" + from .router import _strict_json + from .validation import validate_policy, validate_profile_registry + + registry, source_sha256 = _strict_json( + profiles_payload, "profiles file", + ) + policy, policy_sha256 = _strict_json(policy_payload, "policy file") + validate_profile_registry(registry) + validate_policy(policy) + aliases = sorted({ + reference["id"] + for candidates in policy["roles"].values() + for reference in candidates + if reference["kind"] == "alias" + }) + if not aliases: + return { + "profiles_payload": profiles_payload, + "source_sha256": source_sha256, + "policy_sha256": policy_sha256, + "lifecycle_bindings": [], + } + placeholders = ",".join("?" for _ in aliases) + rows = self.connection.execute( + "SELECT b.alias,b.profile_id,b.version,b.qualification_id,b.updated_at," + "p.profile_json,p.profile_sha256,t.template_id,t.payload_sha256," + "t.policy_id,t.policy_version FROM profile_bindings b " + "JOIN concrete_profiles p ON p.profile_id=b.profile_id " + "JOIN profile_templates t ON t.template_id=b.template_id " + f"WHERE b.alias IN ({placeholders}) ORDER BY b.alias", + tuple(aliases), + ).fetchall() + applicable = [ + row for row in rows + if (row["policy_id"] == policy["id"] + and row["policy_version"] == policy["version"]) + ] + if not applicable: + return { + "profiles_payload": profiles_payload, + "source_sha256": source_sha256, + "policy_sha256": policy_sha256, + "lifecycle_bindings": [], + } + effective = json.loads(canonical_json(registry)) + profiles = {profile["id"]: profile for profile in effective["profiles"]} + lifecycle_bindings = [] + for row in applicable: + profile = json.loads(row["profile_json"]) + existing = profiles.get(profile["id"]) + if existing is not None and canonical_json(existing) != canonical_json(profile): + raise ConflictError( + "runtime profile conflicts with the project profile registry", + ) + if existing is None: + effective["profiles"].append(profile) + profiles[profile["id"]] = profile + effective["bindings"][row["alias"]] = { + "profile_id": row["profile_id"], + "version": row["version"], + } + lifecycle_bindings.append({ + "alias": row["alias"], + "profile_id": row["profile_id"], + "profile_sha256": row["profile_sha256"], + "version": row["version"], + "template_id": row["template_id"], + "template_sha256": row["payload_sha256"], + "qualification_id": row["qualification_id"], + "updated_at": row["updated_at"], + }) + effective["profiles"].sort(key=lambda profile: profile["id"]) + return { + "profiles_payload": (canonical_json(effective) + "\n").encode(), + "source_sha256": source_sha256, + "policy_sha256": policy_sha256, + "lifecycle_bindings": lifecycle_bindings, + } + + def worker_invocations(self, run_id: str) -> int: + """Count attempts whose durable runner actually crossed the launch fence.""" + return self.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=? AND pid IS NOT NULL", + (run_id,), + ).fetchone()[0] + + def _enforce_experiment_budget(self, run_id: str) -> None: + """Shared experiment fence; caller holds the reservation write lock.""" + row = self.connection.execute( + "SELECT s.spec_json,s.recorded_at FROM experiment_assignments a " + "JOIN experiment_specs s ON s.experiment_id=a.experiment_id WHERE a.run_id=?", + (run_id,), + ).fetchone() + if row is None: + return + spec = json.loads(row["spec_json"]) + # Count every durable reservation, including failed/fallback/revision + # slots and concurrent not-yet-launched work. Never trust caller counters. + consumed = self.connection.execute( + "SELECT COUNT(*) FROM attempts t JOIN experiment_assignments a ON a.run_id=t.run_id " + "WHERE a.experiment_id=?", (spec["experiment_id"],), + ).fetchone()[0] + if consumed >= spec["budget"]["max_worker_invocations"]: + raise BudgetExhausted("experiment worker reservation budget is exhausted") + if self._experiment_remaining_wall_seconds(run_id, _authoritative_now()) == 0: + raise BudgetExhausted("experiment wall-time budget is exhausted") + + def reserve_attempt( + self, + run_id: str, + expected_version: int, + owner_id: str, + package_digest: str, + role: str = "worker", + *, + account_pool_id: str | None = None, + profile_id: str | None = None, + profile_index: int | None = None, + ) -> AttemptReservation: + if not owner_id or not package_digest: + raise ContractError("supervisor owner and package digest are required") + if role not in {"worker", "implementer", "reviewer", "lead", "researcher", "proposer_a", "proposer_b", "critic"}: + raise ContractError("attempt role is invalid") + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT project_id,worktree_path,state,phase,version,created_at,updated_at," + "mutable_snapshot,submitted_request FROM runs WHERE id=?", + (run_id,), + ).fetchone() + if not run or run["version"] != expected_version or run["state"] != "queued" or run["phase"] is not None: + raise ConflictError("run is not available for supervisor claim") + active = self.connection.execute("SELECT 1 FROM supervisor_claims WHERE run_id=? AND active=1", (run_id,)).fetchone() + if active: + raise ConflictError("run already has a supervisor claim") + if not run["worktree_path"]: + raise ContractError("run has no canonical worktree identity") + from .experiment_evidence import verify_prelaunch_snapshot + + try: + current_snapshot = json.loads(run["mutable_snapshot"] or "{}") + except (TypeError, ValueError) as exc: + raise ContractError("frozen run snapshot is invalid") from exc + verify_prelaunch_snapshot( + self.connection, run_id=run_id, snapshot=current_snapshot, + package_digest=package_digest, + ) + self._enforce_attempt_budget(run_id, run) + self._enforce_experiment_budget(run_id) + if account_pool_id is not None: + if not isinstance(account_pool_id, str) or not account_pool_id: + raise ContractError("attempt account pool is invalid") + try: + snapshot = json.loads(run["mutable_snapshot"]) + routed_role = snapshot["routing"]["roles"][role] + candidates = [ + routed_role["selected"], *routed_role.get("fallbacks", []), + ] + if profile_id is None and profile_index is None: + selected = candidates[0] + elif (type(profile_index) is int + and 0 <= profile_index < len(candidates) + and profile_id == candidates[profile_index]["profile_id"]): + selected = candidates[profile_index] + else: + raise ConflictError( + "attempt profile does not match frozen routing order" + ) + capacity = snapshot["routing"]["capacity"][account_pool_id] + profile_capacity = capacity.get("profiles", {}).get( + selected.get("profile_id"), capacity, + ) + expected_pool = selected["profile"]["account_pool_id"] + max_concurrency = capacity["max_concurrency"] + capacity_status = profile_capacity.get("status", "available") + unknown_policy = capacity.get( + "unknown_capacity_policy", "allow_bounded", + ) + except ConflictError: + raise + except (IndexError, KeyError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError( + "frozen account-pool reservation is invalid" + ) from exc + if (expected_pool != account_pool_id + or type(max_concurrency) is not int + or max_concurrency < 1 + or capacity_status not in { + "available", "exhausted", "unknown", + } + or unknown_policy not in {"allow_bounded", "block"}): + raise ConflictError( + "attempt account pool does not match frozen routing" + ) + if capacity_status == "exhausted": + raise ConflictError("account pool capacity is exhausted") + if (capacity_status == "unknown" + and unknown_policy == "block"): + raise ConflictError("unknown account pool capacity is blocked") + target = self._capacity_target(selected["profile"]) + elif profile_id is not None or profile_index is not None: + raise ContractError("attempt profile requires an account pool") + old = self.connection.execute("SELECT COALESCE(MAX(fencing_token),0) FROM supervisor_claims WHERE run_id=?", (run_id,)).fetchone()[0] + supervisor_token, attempt_id, attempt_token = old + 1, str(uuid.uuid4()), uuid.uuid4().hex + now, version = _utc_now(), expected_version + 1 + self.connection.execute("INSERT OR REPLACE INTO supervisor_claims(run_id,owner_id,fencing_token,package_digest,heartbeat_at,active) VALUES(?,?,?,?,?,1)", (run_id, owner_id, supervisor_token, package_digest, now)) + self.connection.execute( + "INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token," + "status,heartbeat_at,package_digest,created_at,role,account_pool_id," + "profile_id,profile_index) VALUES(?,?,?,?,?,'reserved',?,?,?,?,?,?,?)", + (attempt_id, run_id, run["project_id"], run["worktree_path"], + attempt_token, now, package_digest, now, role, account_pool_id, + profile_id, profile_index), + ) + if account_pool_id is not None: + self._reserve_pool_capacity_locked( + run_id=run_id, + pool_id=account_pool_id, + purpose="attempt", + profile_id=profile_id, + target=target, + max_concurrency=max_concurrency, + unknown_capacity_policy=unknown_policy, + frozen_status=capacity_status, + attempt_id=attempt_id, + now=datetime.fromisoformat(now), + ) + self.connection.execute("UPDATE runs SET phase='launching',version=?,updated_at=? WHERE id=?", (version, now, run_id)) + event = canonical_json({"attempt_id": attempt_id, "supervisor_token": supervisor_token}) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'supervisor.claimed',?,?)", (run_id, version, event, now)) + self.connection.execute("COMMIT") + return AttemptReservation(run_id, attempt_id, attempt_token, supervisor_token, version) + except sqlite3.IntegrityError as exc: + self.connection.execute("ROLLBACK") + raise ConflictError("worktree already has an active writer") from exc + except Exception: + self.connection.execute("ROLLBACK") + raise + + def mark_attempt_running(self, reservation: AttemptReservation, pid: int, pgid: int, process_start_id: str, durable_paths: dict[str, str] | None = None) -> int: + if any(type(value) is not int or value <= 0 for value in (pid, pgid)) or not process_start_id: + raise ContractError("valid process identity is required") + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute("SELECT r.version,r.phase,a.status,a.attempt_token,s.fencing_token,s.active FROM runs r JOIN attempts a ON a.run_id=r.id JOIN supervisor_claims s ON s.run_id=r.id WHERE r.id=? AND a.id=?", (reservation.run_id, reservation.attempt_id)).fetchone() + if not row or row["version"] != reservation.version or row["phase"] != "launching" or row["status"] != "reserved" or row["attempt_token"] != reservation.attempt_token or row["fencing_token"] != reservation.supervisor_token or not row["active"]: + raise ConflictError("attempt reservation is stale") + version, now = row["version"] + 1, _utc_now() + durable_paths = durable_paths or {} + self.connection.execute("UPDATE attempts SET status='running',pid=?,pgid=?,process_start_id=?,heartbeat_at=?,stdout_spool=?,stderr_spool=?,stdout_meta=?,stderr_meta=?,exit_record=?,child_record=? WHERE id=?", (pid, pgid, process_start_id, now, durable_paths.get("stdout_spool"), durable_paths.get("stderr_spool"), durable_paths.get("stdout_meta"), durable_paths.get("stderr_meta"), durable_paths.get("exit_record"), durable_paths.get("child_record"), reservation.attempt_id)) + self.connection.execute("UPDATE runs SET state='running',phase=NULL,version=?,updated_at=? WHERE id=?", (version, now, reservation.run_id)) + event = canonical_json({"attempt_id": reservation.attempt_id, "pid": pid, "pgid": pgid, "process_start_id": process_start_id}) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.running',?,?)", (reservation.run_id, version, event, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def heartbeat_attempt(self, run_id: str, attempt_token: str, supervisor_token: int) -> None: + now = _utc_now() + self.connection.execute("BEGIN IMMEDIATE") + try: + attempt = self.connection.execute("UPDATE attempts SET heartbeat_at=? WHERE run_id=? AND attempt_token=? AND status IN ('running','cancelling')", (now, run_id, attempt_token)).rowcount + claim = self.connection.execute("UPDATE supervisor_claims SET heartbeat_at=? WHERE run_id=? AND fencing_token=? AND active=1", (now, run_id, supervisor_token)).rowcount + if attempt != 1 or claim != 1: + raise ConflictError("attempt heartbeat is fenced") + self.connection.execute("COMMIT") + except Exception: + self.connection.execute("ROLLBACK") + raise + + def request_cancel(self, run_id: str) -> tuple[int, dict[str, Any] | None]: + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + if not run: + raise ContractError("run does not exist") + if run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["version"], None + if run["state"] == "cancelling": + attempt = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND status='cancelling' ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() + self.connection.execute("COMMIT") + return run["version"], dict(attempt) if attempt else None + if run["state"] != "running": + raise ConflictError("run is not cancellable by the supervisor") + attempt = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND status='running'", (run_id,)).fetchone() + if not attempt: + raise ConflictError("running run has no active attempt") + version, now = run["version"] + 1, _utc_now() + self.connection.execute("UPDATE runs SET state='cancelling',version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("UPDATE attempts SET status='cancelling',heartbeat_at=? WHERE id=?", (now, attempt["id"])) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.cancelling','{}',?)", (run_id, version, now)) + self.connection.execute("COMMIT") + return version, dict(attempt) + except Exception: + self.connection.execute("ROLLBACK") + raise + + @_project_terminal + def finish_attempt(self, run_id: str, attempt_token: str, terminal_state: str, payload: Any) -> int: + if terminal_state not in TERMINAL_STATES: + raise ContractError("invalid terminal state") + encoded = canonical_json(payload) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + attempt = self.connection.execute("SELECT id,status FROM attempts WHERE run_id=? AND attempt_token=?", (run_id, attempt_token)).fetchone() + if not run or not attempt or attempt["status"] not in {"running", "cancelling"}: + raise ConflictError("attempt completion is fenced") + if run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["version"] + version, now = run["version"] + 1, _utc_now() + self.connection.execute("UPDATE attempts SET status='finished',finished_at=? WHERE id=?", (now, attempt["id"])) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) + terminal_state = "cancelled" if run["state"] == "cancelling" else terminal_state + self.connection.execute("UPDATE runs SET state=?,phase=NULL,version=?,updated_at=? WHERE id=?", (terminal_state, version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,?,?,?)", (run_id, version, f"run.{terminal_state}", encoded, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def record_attempt_output(self, run_id: str, attempt_token: str, stdout_id: str, stderr_id: str, metadata: Any) -> int: + encoded = canonical_json(metadata) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + attempt = self.connection.execute("SELECT id,status FROM attempts WHERE run_id=? AND attempt_token=?", (run_id, attempt_token)).fetchone() + if not run or not attempt or attempt["status"] not in {"running", "cancelling"} or run["state"] in TERMINAL_STATES: + raise ConflictError("attempt output is fenced") + version, now = run["version"] + 1, _utc_now() + self.connection.execute("UPDATE attempts SET stdout_artifact_id=?,stderr_artifact_id=?,output_metadata=? WHERE id=?", (stdout_id, stderr_id, encoded, attempt["id"])) + self.connection.execute("UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'attempt.output',?,?)", (run_id, version, encoded, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def _prepare_durable_artifacts( + self, + run_id: str, + artifacts: list[dict[str, Any]], + *, + require_result_receipt: bool, + ) -> tuple[list[tuple[str, str, str, int]], str, str]: + expected_parent = (self.artifacts / run_id).resolve() + prepared = [] + names = set() + for artifact in artifacts: + if not isinstance(artifact, dict): + raise ContractError("durable artifacts must be objects") + name = artifact.get("name") + path = Path(artifact.get("path", "")) + digest = artifact.get("sha256") + size = artifact.get("byte_size") + if not name or Path(name).name != name or name in names: + raise ContractError("durable artifact names must be unique path components") + names.add(name) + if path.resolve().parent != expected_parent: + raise ContractError("durable artifact path is outside the run-owned store") + content = path.read_bytes() + if hashlib.sha256(content).hexdigest() != digest or len(content) != size: + raise ConflictError("durable artifact changed before database import") + prepared.append((name, str(path), digest, size)) + stdout_names = [name for name in names if name.endswith(".stdout")] + stderr_names = [name for name in names if name.endswith(".stderr")] + if (len(stdout_names) != 1 or len(stderr_names) != 1 + or (require_result_receipt and "result-receipt.json" not in names)): + requirement = "stdout, stderr, and result receipt" if require_result_receipt else "stdout and stderr" + raise ContractError(f"durable import requires exactly one {requirement}") + return prepared, stdout_names[0], stderr_names[0] + + def _prepare_exact_artifacts( + self, + run_id: str, + artifacts: list[dict[str, Any]], + expected_names: frozenset[str], + ) -> list[tuple[str, str, str, int]]: + """Verify a complete named artifact set before entering a write transaction.""" + if not isinstance(artifacts, list): + raise ContractError("artifacts must be an array") + expected_parent = (self.artifacts / run_id).resolve() + prepared = [] + names = set() + for artifact in artifacts: + if not isinstance(artifact, dict): + raise ContractError("artifacts must be objects") + name = artifact.get("name") + path = Path(artifact.get("path", "")) + digest = artifact.get("sha256") + size = artifact.get("byte_size") + if (not isinstance(name, str) or not name or Path(name).name != name + or name in names): + raise ContractError("artifact names must be unique path components") + names.add(name) + if path.resolve().parent != expected_parent: + raise ContractError("artifact path is outside the run-owned store") + content = path.read_bytes() + if (not isinstance(digest, str) + or type(size) is not int + or size < 0 + or hashlib.sha256(content).hexdigest() != digest + or len(content) != size): + raise ConflictError("artifact changed before database import") + prepared.append((name, str(path), digest, size)) + if names != set(expected_names): + raise ContractError("terminal artifact set is incomplete or unexpected") + return prepared + + def _reference_prepared_artifacts( + self, + run_id: str, + version: int, + prepared: list[tuple[str, str, str, int]], + ) -> tuple[int, dict[str, str]]: + artifact_ids = {} + for name, path, digest, size in prepared: + existing = self.connection.execute( + "SELECT id,path,sha256,byte_size FROM artifacts WHERE run_id=? AND name=?", + (run_id, name), + ).fetchone() + if existing: + if (existing["path"] != path or existing["sha256"] != digest + or existing["byte_size"] != size): + raise ConflictError("durable artifact conflicts with an existing reference") + artifact_ids[name] = existing["id"] + continue + artifact_id, now = str(uuid.uuid4()), _utc_now() + self.connection.execute( + "INSERT INTO artifacts(id,run_id,name,path,sha256,byte_size,created_at) " + "VALUES(?,?,?,?,?,?,?)", + (artifact_id, run_id, name, path, digest, size, now), + ) + artifact_ids[name] = artifact_id + version += 1 + event = canonical_json({ + "artifact_id": artifact_id, + "name": name, + "sha256": digest, + "byte_size": size, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'artifact.recorded',?,?)", + (run_id, version, event, now), + ) + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + return version, artifact_ids + + def _record_prepared_output( + self, + run_id: str, + version: int, + attempt: sqlite3.Row, + artifact_ids: dict[str, str], + stdout_name: str, + stderr_name: str, + encoded_metadata: str, + ) -> int: + stdout_id, stderr_id = artifact_ids[stdout_name], artifact_ids[stderr_name] + if attempt["output_metadata"] is None: + now = _utc_now() + version += 1 + self.connection.execute( + "UPDATE attempts SET stdout_artifact_id=?,stderr_artifact_id=?," + "output_metadata=? WHERE id=?", + (stdout_id, stderr_id, encoded_metadata, attempt["id"]), + ) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'attempt.output',?,?)", + (run_id, version, encoded_metadata, now), + ) + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + elif (attempt["stdout_artifact_id"] != stdout_id + or attempt["stderr_artifact_id"] != stderr_id + or attempt["output_metadata"] != encoded_metadata): + raise ConflictError("durable output conflicts with its prior import") + return version + + @_project_terminal + def commit_durable_import(self, run_id: str, attempt_token: str, artifacts: list[dict[str, Any]], metadata: Any, terminal_state: str, payload: Any) -> str: + """Atomically import one durable receipt, or observe its prior import. + + Content-addressed files are finalized before this call. All database + references, output projection fields, and terminal state then cross a + single write fence so competing recovery processes cannot partially + import or downgrade a valid completion. + """ + if terminal_state not in TERMINAL_STATES: + raise ContractError("invalid terminal state") + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts( + run_id, artifacts, require_result_receipt=True, + ) + encoded_metadata, encoded_payload = canonical_json(metadata), canonical_json(payload) + + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + attempt = self.connection.execute( + "SELECT id,status,stdout_artifact_id,stderr_artifact_id,output_metadata FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("durable import is fenced") + if attempt["status"] == "finished" and run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["state"] + if attempt["status"] not in {"running", "cancelling"} or run["state"] not in {"running", "cancelling"}: + raise ConflictError("durable import is fenced") + + version, artifact_ids = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version = self._record_prepared_output( + run_id, + version, + attempt, + artifact_ids, + stdout_name, + stderr_name, + encoded_metadata, + ) + + now = _utc_now(); version += 1 + effective_state = "cancelled" if run["state"] == "cancelling" else terminal_state + self.connection.execute("UPDATE attempts SET status='finished',finished_at=? WHERE id=?", (now, attempt["id"])) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute( + "UPDATE runs SET state=?,phase=NULL,version=?,updated_at=? WHERE id=?", + (effective_state, version, now, run_id), + ) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,?,?,?)", + (run_id, version, f"run.{effective_state}", encoded_payload, now), + ) + self.connection.execute("COMMIT") + return effective_state + except Exception: + self.connection.execute("ROLLBACK") + raise + + def commit_durable_fallback( + self, + run_id: str, + attempt_token: str, + artifacts: list[dict[str, Any]], + metadata: Any, + error: dict[str, Any], + ) -> str: + """Record one failed profile attempt and queue its frozen fallback.""" + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts( + run_id, artifacts, require_result_receipt=False, + ) + enriched_metadata = dict(metadata) + enriched_metadata["failure"] = json.loads(canonical_json(error)) + encoded_metadata = canonical_json(enriched_metadata) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status,stdout_artifact_id,stderr_artifact_id,output_metadata " + "FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("durable fallback import is fenced") + if (attempt["status"] == "finished" and run["state"] == "queued" + and run["phase"] is None): + self.connection.execute("COMMIT") + return "queued" + if (attempt["status"] != "running" or run["state"] != "running" + or run["phase"] is not None): + raise ConflictError("durable fallback import is fenced") + version, artifact_ids = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version = self._record_prepared_output( + run_id, + version, + attempt, + artifact_ids, + stdout_name, + stderr_name, + encoded_metadata, + ) + now, version = _utc_now(), version + 1 + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='queued',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = dict(error) + payload["attempt_id"] = attempt["id"] + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.fallback_queued',?,?)", + (run_id, version, canonical_json(payload), now), + ) + self.connection.execute("COMMIT") + return "queued" + except Exception: + self.connection.execute("ROLLBACK") + raise + + def commit_delivery_candidate( + self, + run_id: str, + attempt_token: str, + artifacts: list[dict[str, Any]], + metadata: Any, + mutable_snapshot: dict[str, Any], + review_worktree_path: str, + candidate: dict[str, Any], + ) -> str: + """Atomically import one implementation and queue its frozen candidate.""" + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts( + run_id, artifacts, require_result_receipt=False, + ) + encoded_metadata = canonical_json(metadata) + encoded_snapshot = canonical_json(mutable_snapshot) + if (not isinstance(candidate, dict) + or mutable_snapshot.get("candidate") != candidate): + raise ContractError("delivery candidate differs from its saved snapshot") + expected_lengths = { + "candidate_sha256": 64, + "commit_oid": 40, + "patch_sha256": 64, + } + for field, expected_length in expected_lengths.items(): + value = candidate.get(field) + if (not isinstance(value, str) or len(value) != expected_length + or any(character not in "0123456789abcdef" for character in value)): + raise ContractError(f"delivery candidate {field} is invalid") + try: + resolved_worktree = str(Path(review_worktree_path).resolve(strict=True)) + worktree_common = str(git_common_dir(Path(resolved_worktree))) + except OSError as exc: + raise ContractError("delivery review worktree is unavailable") from exc + + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT r.state,r.phase,r.version,r.mutable_snapshot,p.git_common_dir " + "FROM runs r JOIN projects p ON p.id=r.project_id WHERE r.id=?", + (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status,role,stdout_artifact_id,stderr_artifact_id," + "output_metadata FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("delivery candidate import is fenced") + if (attempt["status"] == "finished" and run["state"] == "queued" + and run["phase"] is None + and run["mutable_snapshot"] == encoded_snapshot): + self.connection.execute("COMMIT") + return "candidate_ready" + if (attempt["status"] != "running" or attempt["role"] != "implementer" + or run["state"] != "running" or run["phase"] is not None): + raise ConflictError("delivery candidate import is fenced") + if worktree_common != run["git_common_dir"]: + raise ContractError("delivery review worktree belongs to another project") + version, artifact_ids = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version = self._record_prepared_output( + run_id, + version, + attempt, + artifact_ids, + stdout_name, + stderr_name, + encoded_metadata, + ) + now, version = _utc_now(), version + 1 + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET mutable_snapshot=?,worktree_path=?,state='queued'," + "phase=NULL,version=?,updated_at=? WHERE id=?", + (encoded_snapshot, resolved_worktree, version, now, run_id), + ) + payload = canonical_json({ + "attempt_id": attempt["id"], + "candidate_sha256": candidate["candidate_sha256"], + "commit_oid": candidate["commit_oid"], + "patch_sha256": candidate["patch_sha256"], + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'delivery.candidate_ready',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return "candidate_ready" + except Exception: + self.connection.execute("ROLLBACK") + raise + + def commit_durable_handoff( + self, + run_id: str, + attempt_token: str, + artifacts: list[dict[str, Any]], + metadata: Any, + packet: dict[str, Any], + *, mutable_snapshot: dict[str, Any] | None = None, + ) -> str: + """Atomically import one completed attempt and publish its host packet.""" + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts( + run_id, artifacts, require_result_receipt=False, + ) + encoded_metadata = canonical_json(metadata) + frozen_packet = json.loads(canonical_json(packet)) + packet_artifacts = frozen_packet.get("artifacts") + if not isinstance(packet_artifacts, list) or not packet_artifacts: + raise ContractError("handoff packet must reference evidence artifacts") + prepared_hashes = {name: digest for name, _, digest, _ in prepared} + referenced_names = set() + for reference in packet_artifacts: + if not isinstance(reference, dict) or set(reference) != {"name", "sha256"}: + raise ContractError("handoff artifact references are invalid") + name, digest = reference["name"], reference["sha256"] + if (not isinstance(name, str) or name in referenced_names + or prepared_hashes.get(name) != digest): + raise ContractError("handoff artifact does not match durable evidence") + referenced_names.add(name) + + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status,stdout_artifact_id,stderr_artifact_id,output_metadata " + "FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("durable handoff import is fenced") + if attempt["status"] == "finished" and run["state"] == "awaiting_host": + self.connection.execute("COMMIT") + return "awaiting_host" + if (attempt["status"] != "running" or run["state"] != "running" + or run["phase"] is not None): + raise ConflictError("durable handoff import is fenced") + if mutable_snapshot is not None: + from .council_runtime import verify_origin + current_snapshot = json.loads(self.run(run_id)["mutable_snapshot"]) + if (current_snapshot.get("task", {}).get("workflow") != "council-decision" + or attempt["id"] is None + or self.connection.execute("SELECT role FROM attempts WHERE id=?", (attempt["id"],)).fetchone()[0] != "critic"): + raise ConflictError("Council handoff projection is fenced") + verify_origin(self, run_id, current_snapshot) + verify_origin(self, run_id, mutable_snapshot) + old = current_snapshot["council_state"] + new = mutable_snapshot["council_state"] + if (old["next_role"] != "critic" or set(old["documents"]) != {"proposer_a", "proposer_b"} + or set(new["documents"]) != {"proposer_a", "proposer_b", "critic"} + or new["next_role"] != "lead" + or {key: value for key, value in new["documents"].items() if key != "critic"} != old["documents"] + or {key: value for key, value in new["artifacts"].items() if key != "critic"} != old["artifacts"]): + raise ConflictError("Council critic import changed finalized proposals or skipped the barrier") + self.connection.execute("UPDATE runs SET mutable_snapshot=? WHERE id=?", (canonical_json(mutable_snapshot), run_id)) + + version, artifact_ids = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version = self._record_prepared_output( + run_id, + version, + attempt, + artifact_ids, + stdout_name, + stderr_name, + encoded_metadata, + ) + for reference in packet_artifacts: + reference["artifact_id"] = artifact_ids[reference["name"]] + packet_json = canonical_json(frozen_packet) + packet_sha256 = hashlib.sha256(packet_json.encode()).hexdigest() + if self.connection.execute( + "SELECT 1 FROM handoffs WHERE run_id=? AND status IN ('open','submitted')", + (run_id,), + ).fetchone(): + raise ConflictError("run already has a pending handoff") + sequence = self.connection.execute( + "SELECT COALESCE(MAX(sequence),0)+1 FROM handoffs WHERE run_id=?", + (run_id,), + ).fetchone()[0] + handoff_id, now = str(uuid.uuid4()), _utc_now() + from .reports import build_handoff_reports, handoff_report_names + handoff_reports = build_handoff_reports( + run_id=run_id, + handoff_id=handoff_id, + sequence=sequence, + packet=frozen_packet, + packet_sha256=packet_sha256, + created_at=now, + ) + report_artifacts = [] + for name, content in handoff_reports.items(): + path, digest, size = self.finalize_artifact(run_id, name, content) + report_artifacts.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + prepared_reports = self._prepare_exact_artifacts( + run_id, + report_artifacts, + frozenset(handoff_report_names(sequence)), + ) + version, _ = self._reference_prepared_artifacts( + run_id, version, prepared_reports, + ) + version += 1 + self.connection.execute( + "INSERT INTO handoffs(id,run_id,sequence,packet_json,packet_sha256,status," + "created_run_version,created_at) VALUES(?,?,?,?,?,'open',?,?)", + (handoff_id, run_id, sequence, packet_json, packet_sha256, version, now), + ) + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='awaiting_host',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff_id, + "packet_sha256": packet_sha256, + "sequence": sequence, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.awaiting_host',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return "awaiting_host" + except Exception: + self.connection.execute("ROLLBACK") + raise + + def commit_council_stage(self, run_id: str, attempt_token: str, artifacts: list, + metadata: dict, snapshot: dict, role: str) -> str: + """Import one sealed proposal under the existing attempt fence.""" + from .council_runtime import verify_origin + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts(run_id, artifacts, require_result_receipt=False) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.run(run_id) + attempt = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND attempt_token=?", (run_id, attempt_token)).fetchone() + if (not attempt or run["state"] != "running" or run["phase"] is not None + or attempt["status"] != "running" or attempt["role"] != role + or role not in {"proposer_a", "proposer_b"}): + raise ConflictError("Council proposal import is fenced") + prior = json.loads(run["mutable_snapshot"]) + verify_origin(self, run_id, prior) + verify_origin(self, run_id, snapshot) + if prior["council_state"]["next_role"] != role or role in prior["council_state"]["documents"]: + raise ConflictError("Council proposal stage is stale") + old_docs = prior["council_state"]["documents"] + new_docs = snapshot["council_state"]["documents"] + if (set(new_docs) != set(old_docs) | {role} + or {k: v for k, v in new_docs.items() if k != role} != old_docs + or snapshot["council_state"]["next_role"] != ("proposer_b" if role == "proposer_a" else "critic") + or {key: value for key, value in snapshot["council_state"]["artifacts"].items() if key != role} != prior["council_state"].get("artifacts", {})): + raise ConflictError("Council proposal import changed another finalized proposal") + version, ids = self._reference_prepared_artifacts(run_id, run["version"], prepared) + version = self._record_prepared_output(run_id, version, attempt, ids, stdout_name, stderr_name, canonical_json(metadata)) + now, version = _utc_now(), version + 1 + self.connection.execute("UPDATE attempts SET status='finished',finished_at=? WHERE id=?", (now, attempt["id"])) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute("UPDATE runs SET mutable_snapshot=?,state='queued',phase=NULL,version=?,updated_at=? WHERE id=?", + (canonical_json(snapshot), version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'council.proposal_finalized',?,?)", + (run_id, version, canonical_json({"role": role, "attempt_id": attempt["id"]}), now)) + self.connection.execute("COMMIT") + return "queued" + except Exception: + self.connection.execute("ROLLBACK") + raise + + def queue_headless_lead( + self, + run_id: str, + expected_version: int, + ) -> dict[str, Any]: + """Pause an open handoff only long enough to launch its configured lead.""" + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT state,phase,version,mutable_snapshot FROM runs WHERE id=?", + (run_id,), + ).fetchone() + handoff = self.connection.execute( + "SELECT id,status,sequence FROM handoffs WHERE run_id=? " + "ORDER BY sequence DESC LIMIT 1", + (run_id,), + ).fetchone() + if not row or not handoff: + raise ConflictError("headless lead handoff is missing") + if row["version"] != expected_version: + raise ConflictError("headless lead run version changed") + try: + snapshot = json.loads(row["mutable_snapshot"]) + budget = snapshot["task"]["budget"]["max_worker_invocations"] + lead_mode = snapshot["task"]["lead"]["mode"] + except (TypeError, KeyError, json.JSONDecodeError) as exc: + raise ConflictError("frozen headless lead configuration is invalid") from exc + if lead_mode != "headless" or type(budget) is not int or budget < 1: + raise ConflictError("run is not configured for a headless lead") + if (row["state"] != "awaiting_host" or row["phase"] is not None + or handoff["status"] != "open"): + raise ConflictError("headless lead handoff is not queueable") + active = self.connection.execute( + "SELECT 1 FROM supervisor_claims WHERE run_id=? AND active=1", + (run_id,), + ).fetchone() + if active: + raise ConflictError("headless lead already has an active supervisor") + invocations = self.worker_invocations(run_id) + if invocations >= budget: + self.connection.execute("COMMIT") + return { + "action": "budget_exhausted", + "version": row["version"], + "worker_invocations": invocations, + } + now, version = _utc_now(), row["version"] + 1 + self.connection.execute( + "UPDATE runs SET state='queued',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff["id"], + "sequence": handoff["sequence"], + "worker_invocations": invocations, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.headless_lead_queued',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return { + "action": "queued", + "version": version, + "worker_invocations": invocations, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def commit_headless_lead( + self, + run_id: str, + attempt_token: str, + artifacts: list[dict[str, Any]], + metadata: Any, + ) -> str: + """Import one valid headless-lead result and reopen its frozen handoff.""" + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts( + run_id, artifacts, require_result_receipt=False, + ) + encoded_metadata = canonical_json(metadata) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status,role,stdout_artifact_id,stderr_artifact_id," + "output_metadata FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + handoff = self.connection.execute( + "SELECT id,status,sequence FROM handoffs WHERE run_id=? " + "ORDER BY sequence DESC LIMIT 1", + (run_id,), + ).fetchone() + if not run or not attempt or not handoff: + raise ConflictError("headless lead import is fenced") + if (attempt["status"] == "finished" + and run["state"] == "awaiting_host" + and handoff["status"] == "open"): + self.connection.execute("COMMIT") + return "awaiting_host" + if (attempt["status"] != "running" or attempt["role"] != "lead" + or run["state"] != "running" or run["phase"] is not None + or handoff["status"] != "open"): + raise ConflictError("headless lead import is fenced") + evidence_name = f"lead-attempt-{attempt['id']}.json" + if evidence_name not in {item[0] for item in prepared}: + raise ContractError("headless lead evidence artifact is missing") + version, artifact_ids = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version = self._record_prepared_output( + run_id, + version, + attempt, + artifact_ids, + stdout_name, + stderr_name, + encoded_metadata, + ) + now, version = _utc_now(), version + 1 + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='awaiting_host',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "attempt_id": attempt["id"], + "handoff_id": handoff["id"], + "evidence": evidence_name, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.headless_lead_ready',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return "awaiting_host" + except Exception: + self.connection.execute("ROLLBACK") + raise + + def block_recovery(self, run_id: str, attempt_token: str, reason: str, *, release_writer: bool = False) -> int: + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + attempt = self.connection.execute("SELECT id,status FROM attempts WHERE run_id=? AND attempt_token=?", (run_id, attempt_token)).fetchone() + if not run or not attempt or attempt["status"] not in {"running", "cancelling"}: + raise ConflictError("recovery disposition is fenced") + version, now = run["version"] + 1, _utc_now() + status = "recovery_required" if release_writer else "ownership_ambiguous" + self.connection.execute("UPDATE attempts SET status=?,finished_at=? WHERE id=?", (status, now, attempt["id"])) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute("UPDATE runs SET state='blocked',phase='recovery_required',version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.blocked',?,?)", (run_id, version, canonical_json({"reason": reason, "next_action": "RECOVERY_REQUIRED"}), now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + @_project_terminal + def recover_unstarted_attempt( + self, run_id: str, attempt_token: str, reason: str, *, terminal_artifacts: list | None = None, + ) -> tuple[int, str]: + """Recover a dead gated runner that never published a child identity. + + The supervisor must first establish that the runner is dead and the + atomic child record is absent. The inner gate cannot open before that + record is durable, so no worker command can have executed in this case. + """ + prepared = (self._prepare_exact_artifacts(run_id, terminal_artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS) + if terminal_artifacts is not None else None) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status,child_record FROM attempts " + "WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if (not run or not attempt or run["state"] not in {"running", "cancelling"} + or run["phase"] is not None + or attempt["status"] not in {"running", "cancelling"}): + raise ConflictError("unstarted attempt recovery is fenced") + if not attempt["child_record"]: + raise ConflictError("unstarted attempt has no durable child-record path") + if Path(attempt["child_record"]).exists() or Path(attempt["child_record"]).is_symlink(): + raise ConflictError("unstarted attempt now has a child identity record") + now = _utc_now() + if run["state"] == "cancelling": + if prepared is not None: + version, _ = self._reference_prepared_artifacts(run_id, run["version"], prepared) + version += 1 + else: + path, digest, size, receipt_time = self._terminal_receipt( + run_id, "cancelled", "cancelling", None, attempt_id=attempt["id"], now=now, + ) + version = self._reference_terminal_receipt( + run_id, run["version"], path, digest, size, receipt_time, + ) + 1 + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "attempt_id": attempt["id"], + "reason": reason, + "receipt": "result-receipt.json", + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id, version, payload, now), + ) + disposition = "cancelled" + else: + version = run["version"] + 1 + self.connection.execute( + "UPDATE attempts SET status='recovery_required',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='queued',phase=NULL,version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.unstarted_attempt_recovered',?,?)", + (run_id, version, canonical_json({ + "attempt_id": attempt["id"], "reason": reason, + }), now), + ) + disposition = "requeued" + self.connection.execute("COMMIT") + return version, disposition + except Exception: + self.connection.execute("ROLLBACK") + raise + + def request_recovery_cancel(self, run_id: str, attempt_token: str) -> int: + """Persist cancellation intent for an ownerless blocked attempt.""" + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("recovery cancellation is fenced") + if run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["version"] + if run["state"] == "cancelling" and attempt["status"] == "cancelling": + self.connection.execute("COMMIT") + return run["version"] + if (run["state"] != "blocked" or run["phase"] != "recovery_required" + or attempt["status"] not in {"ownership_ambiguous", "recovery_required"}): + raise ConflictError("run is not awaiting recovery cancellation") + version, now = run["version"] + 1, _utc_now() + self.connection.execute( + "UPDATE attempts SET status='cancelling',heartbeat_at=?,finished_at=NULL WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE runs SET state='cancelling',phase='recovery_cleanup'," + "version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({"attempt_id": attempt["id"]}) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelling',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + @_project_terminal + def finish_recovery_cancel( + self, run_id: str, attempt_token: str, reason: str, + ) -> int: + """Finish an ownerless cancellation only after absence is confirmed.""" + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("recovery cancellation completion is fenced") + if run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["version"] + if (run["state"] != "cancelling" or run["phase"] != "recovery_cleanup" + or attempt["status"] != "cancelling"): + raise ConflictError("recovery cancellation is not active") + now = _utc_now() + path, digest, size, receipt_time = self._terminal_receipt( + run_id, + "cancelled", + "recovery_cleanup", + None, + attempt_id=attempt["id"], + now=now, + ) + version = self._reference_terminal_receipt( + run_id, run["version"], path, digest, size, receipt_time, + ) + 1 + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "attempt_id": attempt["id"], + "reason": reason, + "receipt": "result-receipt.json", + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def fail_launch(self, reservation: AttemptReservation, reason: str) -> int: + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT phase,version FROM runs WHERE id=?", (reservation.run_id,)).fetchone() + attempt = self.connection.execute("SELECT status,attempt_token FROM attempts WHERE id=?", (reservation.attempt_id,)).fetchone() + if not run or run["phase"] != "launching" or not attempt or attempt["status"] != "reserved" or attempt["attempt_token"] != reservation.attempt_token: + raise ConflictError("launch failure disposition is fenced") + version, now = run["version"] + 1, _utc_now() + self.connection.execute("UPDATE attempts SET status='recovery_required',finished_at=? WHERE id=?", (now, reservation.attempt_id)) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=? AND fencing_token=?", (reservation.run_id, reservation.supervisor_token)) + self.connection.execute("UPDATE runs SET state='blocked',phase='recovery_required',version=?,updated_at=? WHERE id=?", (version, now, reservation.run_id)) + payload = canonical_json({"reason": reason, "next_action": "RECOVERY_REQUIRED"}) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.blocked',?,?)", (reservation.run_id, version, payload, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def recover_launching(self, run_id: str, expected_version: int) -> int: + """Fence an abandoned pre-gate reservation and make the run launchable.""" + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status FROM attempts WHERE run_id=? ORDER BY created_at DESC LIMIT 1", + (run_id,), + ).fetchone() + claim = self.connection.execute( + "SELECT active FROM supervisor_claims WHERE run_id=?", (run_id,), + ).fetchone() + if (not run or run["state"] != "queued" or run["phase"] != "launching" + or run["version"] != expected_version or not attempt + or attempt["status"] != "reserved" or not claim or not claim["active"]): + raise ConflictError("launch reservation is no longer recoverable") + version, now = expected_version + 1, _utc_now() + self.connection.execute( + "UPDATE attempts SET status='recovery_required',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET phase=NULL,version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "attempt_id": attempt["id"], + "reason": "abandoned gated launch reservation", + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.launch_recovered',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def active_attempt(self, run_id: str) -> dict[str, Any] | None: + row = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND status IN ('reserved','running','cancelling','ownership_ambiguous') ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() + return dict(row) if row else None + + @staticmethod + def _handoff_claim_from_row(row: sqlite3.Row, action: str) -> HandoffClaim: + return HandoffClaim( + run_id=row["run_id"], + handoff_id=row["handoff_id"], + owner_id=row["owner_id"], + fencing_token=row["fencing_token"], + expires_at=row["lease_expires_at"], + run_version=row["run_version"], + action=action, + ) + + @classmethod + def _handoff_snapshot_from_rows( + cls, + run_id: str, + run: sqlite3.Row, + handoff: sqlite3.Row | None, + claim_row: sqlite3.Row | None, + ) -> HandoffSnapshot | None: + if handoff is None: + return None + packet_json = handoff["packet_json"] + if hashlib.sha256(packet_json.encode()).hexdigest() != handoff["packet_sha256"]: + raise ConflictError("persisted handoff packet hash does not match") + try: + packet = json.loads(packet_json) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("persisted handoff packet is invalid") from exc + if not isinstance(packet, dict) or canonical_json(packet) != packet_json: + raise ConflictError("persisted handoff packet is not canonical finite JSON") + claim = cls._handoff_claim_from_row(claim_row, "current") if claim_row else None + return HandoffSnapshot( + run_id=run_id, + run_state=run["state"], + run_phase=run["phase"], + run_version=run["version"], + handoff_id=handoff["id"], + sequence=handoff["sequence"], + status=handoff["status"], + packet=packet, + packet_sha256=handoff["packet_sha256"], + created_run_version=handoff["created_run_version"], + submitted_run_version=handoff["submitted_run_version"], + created_at=handoff["created_at"], + closed_at=handoff["closed_at"], + claim=claim, + ) + + def publish_handoff( + self, + run_id: str, + expected_version: int, + attempt_token: str, + supervisor_token: int, + packet: dict[str, Any], + *, + now: datetime | None = None, + ) -> HandoffSnapshot: + """Release a reconciled writer and publish one immutable host packet. + + Process absence is established by the owning supervisor before it calls + this storage transition, just as it is before an ordinary attempt + completion. The persisted attempt and supervisor fences ensure a stale + coordinator cannot publish after ownership has moved. + """ + if (type(expected_version) is not int or expected_version < 1 + or not isinstance(attempt_token, str) or not attempt_token + or type(supervisor_token) is not int + or supervisor_token < 1 or not isinstance(packet, dict)): + raise ContractError("handoff publication requires valid ownership and packet fields") + packet_json = canonical_json(packet) + packet_sha256 = hashlib.sha256(packet_json.encode()).hexdigest() + self.connection.execute("BEGIN IMMEDIATE") + try: + timestamp = _authoritative_now(now).isoformat() + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,a.id AS attempt_id,a.status AS attempt_status," + "s.fencing_token AS supervisor_token,s.active AS supervisor_active " + "FROM runs r JOIN attempts a ON a.run_id=r.id " + "JOIN supervisor_claims s ON s.run_id=r.id " + "WHERE r.id=? AND a.attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if (not row or row["state"] != "running" or row["phase"] is not None + or row["version"] != expected_version + or row["attempt_status"] != "running" + or not row["supervisor_active"] + or row["supervisor_token"] != supervisor_token): + raise ConflictError("handoff publication is fenced") + if self.connection.execute( + "SELECT 1 FROM handoffs WHERE run_id=? AND status IN ('open','submitted')", + (run_id,), + ).fetchone(): + raise ConflictError("run already has a pending handoff") + sequence = self.connection.execute( + "SELECT COALESCE(MAX(sequence),0)+1 FROM handoffs WHERE run_id=?", (run_id,), + ).fetchone()[0] + handoff_id, version = str(uuid.uuid4()), expected_version + 1 + self.connection.execute( + "INSERT INTO handoffs(id,run_id,sequence,packet_json,packet_sha256,status," + "created_run_version,created_at) VALUES(?,?,?,?,?,'open',?,?)", + (handoff_id, run_id, sequence, packet_json, packet_sha256, version, timestamp), + ) + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (timestamp, row["attempt_id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=? AND fencing_token=?", + (run_id, supervisor_token), + ) + self.connection.execute( + "UPDATE runs SET state='awaiting_host',phase=NULL,version=?,updated_at=? WHERE id=?", + (version, timestamp, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff_id, + "packet_sha256": packet_sha256, + "sequence": sequence, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.awaiting_host',?,?)", + (run_id, version, payload, timestamp), + ) + self.connection.execute("COMMIT") + except Exception: + self.connection.execute("ROLLBACK") + raise + snapshot = self.handoff_snapshot(run_id) + if snapshot is None: # Defensive: the just-committed projection must exist. + raise ConflictError("published handoff is missing") + return snapshot + + def handoff_snapshot(self, run_id: str) -> HandoffSnapshot | None: + self.connection.execute("BEGIN") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + if not run: + raise ContractError("run does not exist") + handoff = self.connection.execute( + "SELECT * FROM handoffs WHERE run_id=? ORDER BY sequence DESC LIMIT 1", (run_id,), + ).fetchone() + claim_row = None + if handoff: + claim_row = self.connection.execute( + "SELECT run_id,handoff_id,owner_id,fencing_token,lease_expires_at," + "? AS run_version FROM claims WHERE run_id=? AND kind='host' " + "AND handoff_id=? AND active=1", + (run["version"], run_id, handoff["id"]), + ).fetchone() + snapshot = self._handoff_snapshot_from_rows( + run_id, run, handoff, claim_row, + ) + self.connection.execute("COMMIT") + return snapshot + except Exception: + self.connection.execute("ROLLBACK") + raise + + def handoff_snapshot_by_id( + self, run_id: str, handoff_id: str, + ) -> HandoffSnapshot: + if not isinstance(handoff_id, str) or not handoff_id: + raise ContractError("handoff id is required") + self.connection.execute("BEGIN") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + if not run: + raise ContractError("run does not exist") + handoff = self.connection.execute( + "SELECT * FROM handoffs WHERE id=? AND run_id=?", (handoff_id, run_id), + ).fetchone() + if not handoff: + raise ConflictError("handoff belongs to a different run") + claim_row = self.connection.execute( + "SELECT run_id,handoff_id,owner_id,fencing_token,lease_expires_at," + "? AS run_version FROM claims WHERE run_id=? AND kind='host' " + "AND handoff_id=? AND active=1", + (run["version"], run_id, handoff_id), + ).fetchone() + snapshot = self._handoff_snapshot_from_rows( + run_id, run, handoff, claim_row, + ) + if snapshot is None: # Defensive: handoff was selected above. + raise ConflictError("handoff is missing") + self.connection.execute("COMMIT") + return snapshot + except Exception: + self.connection.execute("ROLLBACK") + raise + + def branch_review_history(self, run_id: str) -> list[dict[str, Any]]: + """Return canonical recorded handoffs and decisions in workflow order.""" + if not self.connection.execute( + "SELECT 1 FROM runs WHERE id=?", (run_id,), + ).fetchone(): + raise ContractError("run does not exist") + rows = self.connection.execute( + "SELECT h.id AS handoff_id,h.sequence,h.packet_json,h.packet_sha256," + "s.submission_id,s.submission_hash,s.disposition,s.decision_json," + "s.recorded_run_version " + "FROM handoffs h JOIN handoff_submissions s ON s.handoff_id=h.id " + "WHERE h.run_id=? AND s.outcome='recorded' ORDER BY h.sequence", + (run_id,), + ).fetchall() + history = [] + for row in rows: + try: + packet = json.loads(row["packet_json"]) + decision = json.loads(row["decision_json"]) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("persisted branch review history is invalid") from exc + if (not isinstance(packet, dict) + or canonical_json(packet) != row["packet_json"] + or hashlib.sha256(row["packet_json"].encode()).hexdigest() + != row["packet_sha256"] + or not isinstance(decision, dict) + or canonical_json(decision) != row["decision_json"]): + raise ConflictError("persisted branch review history hash is invalid") + submission_id, submission_hash, disposition, _, _ = ( + self._validated_handoff_decision(run_id, decision) + ) + if (submission_id != row["submission_id"] + or submission_hash != row["submission_hash"] + or disposition != row["disposition"]): + raise ConflictError("persisted branch review decision is inconsistent") + history.append({ + "handoff_id": row["handoff_id"], + "sequence": row["sequence"], + "packet": packet, + "packet_sha256": row["packet_sha256"], + "decision": decision, + "recorded_run_version": row["recorded_run_version"], + }) + return history + + def recorded_handoff_submission( + self, run_id: str, handoff_id: str, + ) -> dict[str, Any] | None: + for entry in self.branch_review_history(run_id): + if entry["handoff_id"] == handoff_id: + return entry + return None + + def _terminal_finish_for_claim( + self, run_id: str, row: sqlite3.Row, + ) -> dict[str, Any] | None: + """Only the exact latest acquisition event proves terminal authority.""" + if (row["kind"] != "host" or not row["active"] + or row["owner_id"] != "terminal-operator" + or row["claim_handoff_id"] != row["handoff_id"]): + return None + event = self.connection.execute( + "SELECT type,payload FROM events WHERE run_id=? " + "AND type IN ('handoff.acquired','handoff.taken_over','handoff.renewed') " + "ORDER BY id DESC LIMIT 1", (run_id,), + ).fetchone() + if event is None: + return None + try: + payload = json.loads(event["payload"]) + if not isinstance(payload, dict) or "terminal_finish" not in payload: + return None + if (set(payload) != {"action", "expires_at", "fencing_token", "handoff_id", "owner_id", "terminal_finish"} + or canonical_json(payload) != event["payload"]): + raise ConflictError("persisted terminal finish marker is invalid") + marker = payload["terminal_finish"] + if (not isinstance(marker, dict) + or set(marker) != {"schema_version", "packet_sha256", "decision"} + or type(marker["schema_version"]) is not int + or marker["schema_version"] != 1): + raise ConflictError("persisted terminal finish intent is invalid") + # A later app claim or renewal invalidates older terminal authority, + # including when it happens to use the same human-readable owner. + if (type(payload["fencing_token"]) is not int + or payload["fencing_token"] != row["fencing_token"] + or payload["expires_at"] != row["lease_expires_at"] + or payload["handoff_id"] != row["handoff_id"] + or payload["owner_id"] != row["owner_id"] + or event["type"] != f"handoff.{payload['action']}" + or marker["packet_sha256"] != row["packet_sha256"]): + return None + self._validated_handoff_decision(run_id, marker["decision"]) + if marker["decision"]["submission_id"] != f"terminal-{row['handoff_id']}": + raise ConflictError("persisted terminal finish targets a different handoff") + return marker["decision"] + except (TypeError, ValueError, ContractError) as exc: + raise ConflictError("persisted terminal finish intent is invalid") from exc + + def terminal_finish_decision( + self, run_id: str, handoff_id: str, + ) -> dict[str, Any] | None: + """Read a pending guided choice without exposing or inventing a claim.""" + row = self.connection.execute( + "SELECT h.id AS handoff_id,h.packet_sha256,c.kind,c.owner_id," + "c.fencing_token,c.active,c.handoff_id AS claim_handoff_id,c.lease_expires_at " + "FROM runs r JOIN handoffs h ON h.run_id=r.id JOIN claims c ON c.run_id=r.id " + "WHERE r.id=? AND h.id=? AND r.state='awaiting_host' AND r.phase IS NULL " + "AND h.status='open'", (run_id, handoff_id), + ).fetchone() + return self._terminal_finish_for_claim(run_id, row) if row is not None else None + + def claim_handoff( + self, + run_id: str, + expected_version: int, + owner_id: str, + prior_claim: HandoffClaim | None = None, + *, + now: datetime | None = None, + initial_only: bool = False, + terminal_decision: dict[str, Any] | None = None, + ) -> HandoffClaim: + if (type(expected_version) is not int or expected_version < 1 + or not isinstance(owner_id, str) or not owner_id): + raise ContractError("handoff claim requires a run version and owner") + if prior_claim is not None and not isinstance(prior_claim, HandoffClaim): + raise ContractError("prior handoff claim is invalid") + if terminal_decision is not None and ( + not initial_only or prior_claim is not None + or owner_id != "terminal-operator"): + raise ContractError("terminal finish requires its own initial guided claim") + self.connection.execute("BEGIN IMMEDIATE") + try: + current = _authoritative_now(now) + timestamp = current.isoformat() + expires_at = (current + timedelta(seconds=HOST_LEASE_SECONDS)).isoformat() + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,h.id AS handoff_id,h.status,h.packet_sha256," + "c.kind,c.owner_id,c.fencing_token,c.active,c.handoff_id AS claim_handoff_id," + "c.lease_expires_at " + "FROM runs r JOIN handoffs h ON h.run_id=r.id " + "JOIN claims c ON c.run_id=r.id " + "WHERE r.id=? ORDER BY h.sequence DESC LIMIT 1", + (run_id,), + ).fetchone() + if not row: + raise ContractError("run has no handoff") + if (row["state"] != "awaiting_host" or row["phase"] is not None + or row["status"] != "open" or row["version"] != expected_version): + raise ConflictError("handoff is not claimable at that run version") + terminal_marker = None + if terminal_decision is not None: + self._validated_handoff_decision(run_id, terminal_decision) + if terminal_decision["submission_id"] != f"terminal-{row['handoff_id']}": + raise ConflictError("terminal finish decision targets a different handoff") + if (not terminal_decision["reason"].strip() + or len(terminal_decision["reason"]) > 2000): + raise ContractError("terminal finish requires a bounded non-empty reason") + terminal_marker = { + "schema_version": 1, "packet_sha256": row["packet_sha256"], + "decision": json.loads(canonical_json(terminal_decision)), + } + prior_host_claim = row["kind"] == "host" and row["claim_handoff_id"] == row["handoff_id"] + if initial_only and prior_host_claim: + saved = self._terminal_finish_for_claim(run_id, row) if terminal_marker is not None else None + if saved is None: + raise ConflictError("handoff already has a host claim; use its saved claim to complete or renew it") + if canonical_json(saved) != canonical_json(terminal_decision): + raise ConflictError("a different terminal finish intent is pending; retry its exact saved decision") + live = bool( + row["kind"] == "host" and row["active"] + and row["claim_handoff_id"] == row["handoff_id"] + and row["lease_expires_at"] + and current < _parse_utc(row["lease_expires_at"]) + ) + if initial_only and prior_host_claim and live: + # Exact live recovery does not renew, mutate or increment a + # fence. The durable event retains the capability after a crash. + self.connection.execute("COMMIT") + return HandoffClaim( + run_id, row["handoff_id"], owner_id, row["fencing_token"], + row["lease_expires_at"], expected_version, "current", + ) + if prior_claim is not None: + if (not live or prior_claim.run_id != run_id + or prior_claim.handoff_id != row["handoff_id"] + or prior_claim.owner_id != owner_id + or row["owner_id"] != owner_id + or prior_claim.fencing_token != row["fencing_token"] + or prior_claim.expires_at != row["lease_expires_at"]): + raise ConflictError("handoff renewal claim is stale or expired") + token, action = row["fencing_token"], "renewed" + self.connection.execute( + "UPDATE claims SET lease_expires_at=?,renewed_at=? WHERE run_id=?", + (expires_at, timestamp, run_id), + ) + else: + if live: + raise ConflictError("handoff already has a live claim") + token = row["fencing_token"] + 1 + action = "taken_over" if prior_host_claim else "acquired" + self.connection.execute( + "UPDATE claims SET kind='host',fencing_token=?,owner_id=?,active=1," + "claimed_at=?,handoff_id=?,lease_expires_at=?,renewed_at=NULL WHERE run_id=?", + (token, owner_id, timestamp, row["handoff_id"], expires_at, run_id), + ) + version = expected_version + 1 + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, timestamp, run_id), + ) + payload = canonical_json({ + "action": action, + "expires_at": expires_at, + "fencing_token": token, + "handoff_id": row["handoff_id"], + "owner_id": owner_id, + **({"terminal_finish": terminal_marker} if terminal_marker is not None else {}), + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,?,?,?)", + (run_id, version, f"handoff.{action}", payload, timestamp), + ) + self.connection.execute("COMMIT") + return HandoffClaim( + run_id=run_id, + handoff_id=row["handoff_id"], + owner_id=owner_id, + fencing_token=token, + expires_at=expires_at, + run_version=version, + action=action, + ) + except Exception: + self.connection.execute("ROLLBACK") + raise + + def _validated_handoff_decision( + self, run_id: str, decision: dict[str, Any], + ) -> tuple[str, str, str, str, list[dict[str, str]]]: + expected_fields = { + "schema_version", "submission_id", "submission_hash", + "disposition", "reason", "evidence_refs", + } + snapshot = json.loads(self.run(run_id)["mutable_snapshot"] or "{}") + if snapshot.get("task", {}).get("workflow") == "council-decision": + expected_fields.add("council_choice") + if not isinstance(decision, dict) or set(decision) != expected_fields: + raise ContractError("handoff decision fields are invalid") + if decision["schema_version"] != 1 or type(decision["schema_version"]) is not int: + raise ContractError("handoff decision schema_version is invalid") + submission_id = decision["submission_id"] + disposition = decision["disposition"] + reason = decision["reason"] + evidence_refs = decision["evidence_refs"] + if not isinstance(submission_id, str) or not submission_id: + raise ContractError("handoff submission_id is required") + if disposition not in {"accept", "revise", "reject"}: + raise ContractError("handoff disposition is invalid") + if not isinstance(reason, str) or (disposition == "revise" and not reason): + raise ContractError("handoff decision reason is invalid") + if not isinstance(evidence_refs, list): + raise ContractError("handoff evidence_refs must be an array") + normalized_refs = [] + for evidence in evidence_refs: + if (not isinstance(evidence, dict) or set(evidence) != {"artifact_id", "sha256"} + or not isinstance(evidence["artifact_id"], str) + or not evidence["artifact_id"] + or not isinstance(evidence["sha256"], str)): + raise ContractError("handoff evidence reference is invalid") + artifact = self.connection.execute( + "SELECT sha256 FROM artifacts WHERE id=? AND run_id=?", + (evidence["artifact_id"], run_id), + ).fetchone() + if not artifact or artifact["sha256"] != evidence["sha256"]: + raise ContractError("handoff evidence does not match a run artifact") + normalized_refs.append({ + "artifact_id": evidence["artifact_id"], "sha256": evidence["sha256"], + }) + body = {key: decision[key] for key in expected_fields if key != "submission_hash"} + expected_hash = request_hash(body) + if decision["submission_hash"] != expected_hash: + raise ContractError("handoff submission hash does not match the decision") + return submission_id, expected_hash, disposition, reason, normalized_refs + + def record_handoff_submission( + self, + run_id: str, + claim: HandoffClaim, + decision: dict[str, Any], + *, + now: datetime | None = None, + ) -> HandoffSubmission: + if not isinstance(claim, HandoffClaim) or claim.run_id != run_id: + raise ContractError("handoff completion claim is invalid") + submission_id, submission_hash, disposition, _, evidence_refs = ( + self._validated_handoff_decision(run_id, decision) + ) + decision_json = canonical_json(decision) + evidence_json = canonical_json(evidence_refs) + rejection = None + result = None + self.connection.execute("BEGIN IMMEDIATE") + try: + current = _authoritative_now(now) + timestamp = current.isoformat() + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + if not run: + raise ContractError("run does not exist") + handoff = self.connection.execute( + "SELECT * FROM handoffs WHERE id=? AND run_id=?", (claim.handoff_id, run_id), + ).fetchone() + if not handoff: + raise ConflictError("handoff completion targets a different run") + prior = self.connection.execute( + "SELECT * FROM handoff_submissions WHERE handoff_id=? AND submission_id=? " + "AND submission_hash=? ORDER BY created_at LIMIT 1", + (claim.handoff_id, submission_id, submission_hash), + ).fetchone() + if prior: + if prior["outcome"] == "recorded": + self.connection.execute("COMMIT") + return HandoffSubmission( + run_id=run_id, + handoff_id=claim.handoff_id, + submission_id=submission_id, + submission_hash=submission_hash, + disposition=prior["disposition"], + recorded_run_version=prior["recorded_run_version"], + replayed=True, + ) + rejection = prior["rejection_code"] or "handoff_submission_rejected" + reused_id = self.connection.execute( + "SELECT 1 FROM handoff_submissions WHERE handoff_id=? AND submission_id=? " + "AND submission_hash<>?", + (claim.handoff_id, submission_id, submission_hash), + ).fetchone() + if reused_id and rejection is None: + rejection = "submission_id_reused" + persisted_claim = self.connection.execute( + "SELECT c.kind,c.owner_id,c.fencing_token,c.active,c.lease_expires_at," + "c.handoff_id AS claim_handoff_id,h.id AS handoff_id,h.packet_sha256 " + "FROM claims c JOIN handoffs h ON h.run_id=c.run_id " + "WHERE c.run_id=? AND h.id=?", (run_id, claim.handoff_id), + ).fetchone() + recovered_rejection = None + if (prior is not None and prior["outcome"] == "rejected" + and isinstance(prior["id"], str) and prior["id"] + and prior["rejection_code"] == "expired_claim" + and claim.owner_id == "terminal-operator" + and prior["owner_id"] == claim.owner_id + and type(prior["fencing_token"]) is int + and prior["fencing_token"] < claim.fencing_token + and prior["decision_json"] == decision_json + and prior["evidence_refs_json"] == evidence_json + and prior["disposition"] == disposition + and type(prior["recorded_run_version"]) is int + and prior["recorded_run_version"] < run["version"] + and reused_id is None + and persisted_claim is not None + and claim.expires_at == persisted_claim["lease_expires_at"]): + saved = self._terminal_finish_for_claim(run_id, persisted_claim) + if saved is not None and canonical_json(saved) == decision_json: + rejected_event = self.connection.execute( + "SELECT run_id,run_version,type,payload,created_at FROM events " + "WHERE run_id=? AND run_version=?", + (run_id, prior["recorded_run_version"]), + ).fetchone() + expected_payload = canonical_json({ + "handoff_id": prior["handoff_id"], + "reason": "expired_claim", + "submission_hash": prior["submission_hash"], + "submission_id": prior["submission_id"], + }) + if (rejected_event is None + or rejected_event["run_id"] != run_id + or rejected_event["run_version"] != prior["recorded_run_version"] + or rejected_event["type"] != "handoff.completion_rejected" + or rejected_event["created_at"] != prior["created_at"] + or rejected_event["payload"] != expected_payload): + raise ConflictError("expired terminal finish rejection audit is invalid") + # The row is an operational projection with a unique + # decision identity. Preserve its complete rejected state + # in the append-only log before changing that projection. + if _parse_utc(prior["created_at"]) > current: + raise ConflictError("expired terminal finish rejection timestamp is invalid") + recovered_rejection = dict(prior) + rejection = None + if rejection is None: + if run["state"] in TERMINAL_STATES: + rejection = "terminal_run" + elif (run["state"] != "awaiting_host" or run["phase"] is not None + or handoff["status"] != "open"): + rejection = "handoff_not_open" + elif (not persisted_claim or persisted_claim["kind"] != "host" + or not persisted_claim["active"] + or persisted_claim["claim_handoff_id"] != claim.handoff_id + or persisted_claim["owner_id"] != claim.owner_id + or persisted_claim["fencing_token"] != claim.fencing_token): + rejection = "stale_claim" + elif (not persisted_claim["lease_expires_at"] + or current >= _parse_utc(persisted_claim["lease_expires_at"])): + rejection = "expired_claim" + if rejection is None: + version = run["version"] + 1 + if recovered_rejection is not None: + payload = canonical_json({ + "disposition": disposition, + "handoff_id": claim.handoff_id, + "fencing_token": claim.fencing_token, + "submission_hash": submission_hash, + "submission_id": submission_id, + "rejected_submission": recovered_rejection, + "rejected_submission_sha256": request_hash(recovered_rejection), + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'handoff.completion_recovered',?,?)", + (run_id, version, payload, timestamp), + ) + version += 1 + updated = self.connection.execute( + "UPDATE handoff_submissions SET owner_id=?,fencing_token=?,outcome='recorded'," + "rejection_code=NULL,recorded_run_version=?,created_at=? " + "WHERE id=? AND outcome='rejected' AND rejection_code='expired_claim'", + (claim.owner_id, claim.fencing_token, version, timestamp, prior["id"]), + ) + if updated.rowcount != 1: + raise ConflictError("expired terminal finish projection changed") + else: + self.connection.execute( + "INSERT INTO handoff_submissions(id,handoff_id,submission_id,submission_hash," + "owner_id,fencing_token,disposition,decision_json,evidence_refs_json,outcome," + "rejection_code,recorded_run_version,created_at) " + "VALUES(?,?,?,?,?,?,?,?,?,'recorded',NULL,?,?)", + (str(uuid.uuid4()), claim.handoff_id, submission_id, submission_hash, + claim.owner_id, claim.fencing_token, disposition, decision_json, + evidence_json, version, timestamp), + ) + self.connection.execute( + "UPDATE handoffs SET status='submitted',submitted_run_version=?,closed_at=? " + "WHERE id=?", (version, timestamp, claim.handoff_id), + ) + self.connection.execute( + "UPDATE claims SET active=0 WHERE run_id=? AND handoff_id=?", + (run_id, claim.handoff_id), + ) + self.connection.execute( + "UPDATE runs SET phase='handoff_submitted',version=?,updated_at=? WHERE id=?", + (version, timestamp, run_id), + ) + payload = canonical_json({ + "disposition": disposition, + "handoff_id": claim.handoff_id, + "submission_hash": submission_hash, + "submission_id": submission_id, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'handoff.submitted',?,?)", + (run_id, version, payload, timestamp), + ) + result = HandoffSubmission( + run_id=run_id, + handoff_id=claim.handoff_id, + submission_id=submission_id, + submission_hash=submission_hash, + disposition=disposition, + recorded_run_version=version, + replayed=False, + ) + else: + existing_rejection = self.connection.execute( + "SELECT recorded_run_version FROM handoff_submissions WHERE handoff_id=? " + "AND submission_id=? AND submission_hash=? AND outcome='rejected'", + (claim.handoff_id, submission_id, submission_hash), + ).fetchone() + if not existing_rejection: + terminal_audit = run["state"] in TERMINAL_STATES + version = run["version"] if terminal_audit else run["version"] + 1 + self.connection.execute( + "INSERT INTO handoff_submissions(id,handoff_id,submission_id,submission_hash," + "owner_id,fencing_token,disposition,decision_json,evidence_refs_json,outcome," + "rejection_code,recorded_run_version,created_at) " + "VALUES(?,?,?,?,?,?,?,?,?,'rejected',?,?,?)", + (str(uuid.uuid4()), claim.handoff_id, submission_id, submission_hash, + claim.owner_id, claim.fencing_token, disposition, decision_json, + evidence_json, rejection, version, timestamp), + ) + if not terminal_audit: + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", + (version, timestamp, run_id), + ) + payload = canonical_json({ + "handoff_id": claim.handoff_id, + "reason": rejection, + "submission_hash": submission_hash, + "submission_id": submission_id, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'handoff.completion_rejected',?,?)", + (run_id, version, payload, timestamp), + ) + self.connection.execute("COMMIT") + except Exception: + self.connection.execute("ROLLBACK") + raise + if rejection is not None: + raise ConflictError(f"handoff completion rejected: {rejection}") + if result is None: + raise ConflictError("handoff completion was not recorded") + return result + + def requeue_review_revision( + self, + run_id: str, + handoff_id: str, + submission_id: str, + submission_hash: str, + ) -> dict[str, Any]: + """Consume one saved revise decision and requeue within both task budgets.""" + if not all( + isinstance(value, str) and value + for value in (handoff_id, submission_id, submission_hash) + ): + raise ContractError("revision requeue identifiers are invalid") + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,r.mutable_snapshot,h.status,h.sequence," + "s.disposition FROM runs r JOIN handoffs h ON h.run_id=r.id " + "JOIN handoff_submissions s ON s.handoff_id=h.id " + "WHERE r.id=? AND h.id=? AND s.submission_id=? " + "AND s.submission_hash=? AND s.outcome='recorded'", + (run_id, handoff_id, submission_id, submission_hash), + ).fetchone() + if not row: + raise ConflictError("recorded revision submission is missing") + if row["disposition"] != "revise": + raise ConflictError("handoff submission is not a revision request") + latest_sequence = self.connection.execute( + "SELECT MAX(sequence) FROM handoffs WHERE run_id=?", (run_id,), + ).fetchone()[0] + if row["status"] == "consumed": + action = ( + "requeued" + if row["sequence"] == latest_sequence + and row["state"] == "queued" + and row["phase"] is None + else "already_advanced" + ) + self.connection.execute("COMMIT") + return { + "action": action, + "version": row["version"], + "replayed": True, + } + if (row["state"] != "awaiting_host" + or row["phase"] != "handoff_submitted" + or row["status"] != "submitted" + or row["sequence"] != latest_sequence): + raise ConflictError("revision handoff is no longer current") + try: + snapshot = json.loads(row["mutable_snapshot"]) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("frozen review snapshot is invalid") from exc + if (not isinstance(snapshot, dict) + or canonical_json(snapshot) != row["mutable_snapshot"]): + raise ConflictError("frozen review snapshot is not canonical") + try: + budget = snapshot["task"]["budget"] + max_revisions = budget["max_revisions"] + max_invocations = budget["max_worker_invocations"] + lead_mode = snapshot["task"]["lead"]["mode"] + except (KeyError, TypeError) as exc: + raise ConflictError("frozen review budget is missing") from exc + if (type(max_revisions) is not int or max_revisions < 0 + or type(max_invocations) is not int or max_invocations < 1): + raise ConflictError("frozen review budget is invalid") + revisions = self.connection.execute( + "SELECT COUNT(*) FROM handoff_submissions s " + "JOIN handoffs h ON h.id=s.handoff_id " + "WHERE h.run_id=? AND s.outcome='recorded' " + "AND s.disposition='revise' AND h.sequence<=?", + (run_id, row["sequence"]), + ).fetchone()[0] + invocations = self.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=?", (run_id,), + ).fetchone()[0] + required_invocations = 2 if lead_mode == "headless" else 1 + if (revisions > max_revisions + or invocations + required_invocations > max_invocations): + self.connection.execute("COMMIT") + return { + "action": "budget_exhausted", + "version": row["version"], + "replayed": False, + "revisions_requested": revisions, + "worker_invocations": invocations, + } + now, version = _utc_now(), row["version"] + 1 + self.connection.execute( + "UPDATE handoffs SET status='consumed',closed_at=? WHERE id=?", + (now, handoff_id), + ) + self.connection.execute( + "UPDATE runs SET state='queued',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff_id, + "submission_id": submission_id, + "revisions_requested": revisions, + "worker_invocations": invocations, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.revision_queued',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return { + "action": "requeued", + "version": version, + "replayed": False, + "revisions_requested": revisions, + "worker_invocations": invocations, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def requeue_delivery_revision( + self, + run_id: str, + handoff_id: str, + submission_id: str, + submission_hash: str, + ) -> dict[str, Any]: + """Consume a delivery revise decision and atomically return to its writer.""" + if not all( + isinstance(value, str) and value + for value in (handoff_id, submission_id, submission_hash) + ): + raise ContractError("delivery revision identifiers are invalid") + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,r.mutable_snapshot,h.status,h.sequence," + "h.packet_json,s.disposition,s.decision_json " + "FROM runs r JOIN handoffs h ON h.run_id=r.id " + "JOIN handoff_submissions s ON s.handoff_id=h.id " + "WHERE r.id=? AND h.id=? AND s.submission_id=? " + "AND s.submission_hash=? AND s.outcome='recorded'", + (run_id, handoff_id, submission_id, submission_hash), + ).fetchone() + if not row: + raise ConflictError("recorded delivery revision is missing") + if row["disposition"] != "revise": + raise ConflictError("delivery submission is not a revision request") + latest_sequence = self.connection.execute( + "SELECT MAX(sequence) FROM handoffs WHERE run_id=?", (run_id,), + ).fetchone()[0] + if row["status"] == "consumed": + action = ( + "requeued" + if row["sequence"] == latest_sequence + and row["state"] == "queued" + and row["phase"] is None + else "already_advanced" + ) + self.connection.execute("COMMIT") + return { + "action": action, + "version": row["version"], + "replayed": True, + } + if (row["state"] != "awaiting_host" + or row["phase"] != "handoff_submitted" + or row["status"] != "submitted" + or row["sequence"] != latest_sequence): + raise ConflictError("delivery revision handoff is no longer current") + try: + snapshot = json.loads(row["mutable_snapshot"]) + packet = json.loads(row["packet_json"]) + decision = json.loads(row["decision_json"]) + task = snapshot["task"] + budget = task["budget"] + candidate = snapshot["candidate"] + delivery = snapshot["delivery_workspace"] + except (KeyError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError("frozen delivery revision is invalid") from exc + if (not isinstance(snapshot, dict) + or canonical_json(snapshot) != row["mutable_snapshot"] + or task.get("workflow") != "issue-delivery" + or packet.get("workflow") != "issue-delivery" + or packet.get("candidate_sha256") + != candidate.get("candidate_sha256") + or decision.get("submission_id") != submission_id + or decision.get("submission_hash") != submission_hash): + raise ConflictError("frozen delivery revision changed its evidence") + max_revisions = budget.get("max_revisions") + max_invocations = budget.get("max_worker_invocations") + lead_mode = task.get("lead", {}).get("mode") + if (type(max_revisions) is not int or max_revisions < 0 + or type(max_invocations) is not int or max_invocations < 1 + or lead_mode not in {"host", "headless"}): + raise ConflictError("frozen delivery budget is invalid") + revisions = self.connection.execute( + "SELECT COUNT(*) FROM handoff_submissions s " + "JOIN handoffs h ON h.id=s.handoff_id " + "WHERE h.run_id=? AND s.outcome='recorded' " + "AND s.disposition='revise' AND h.sequence<=?", + (run_id, row["sequence"]), + ).fetchone()[0] + invocations = self.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=?", (run_id,), + ).fetchone()[0] + required_invocations = 3 if lead_mode == "headless" else 2 + wall_exhausted = self.remaining_wall_seconds(run_id) == 0 + if (revisions > max_revisions + or invocations + required_invocations > max_invocations + or wall_exhausted): + self.connection.execute("COMMIT") + return { + "action": "budget_exhausted", + "version": row["version"], + "replayed": False, + "revisions_requested": revisions, + "worker_invocations": invocations, + "wall_exhausted": wall_exhausted, + } + new_snapshot = json.loads(canonical_json(snapshot)) + review_fixture = new_snapshot.pop("internal_review_fixture", None) + if review_fixture is not None: + new_snapshot["pending_review_fixture"] = { + field: review_fixture[field] + for field in ("verdict", "summary", "findings") + } + revision_request = { + "handoff_id": handoff_id, + "sequence": row["sequence"], + "submission_id": submission_id, + "submission_hash": submission_hash, + "previous_candidate_sha256": candidate["candidate_sha256"], + "reason": decision["reason"], + "review": packet["review"], + "checks": packet["checks"], + "evidence_refs": decision["evidence_refs"], + } + new_snapshot["revision_request"] = revision_request + for field in ("candidate", "workspace", "check_workspace"): + new_snapshot.pop(field, None) + encoded_snapshot = canonical_json(new_snapshot) + delivery_path = str(Path(delivery["path"]).resolve(strict=True)) + now, version = _utc_now(), row["version"] + 1 + self.connection.execute( + "UPDATE handoffs SET status='consumed',closed_at=? WHERE id=?", + (now, handoff_id), + ) + self.connection.execute( + "UPDATE runs SET mutable_snapshot=?,worktree_path=?,state='queued'," + "phase=NULL,version=?,updated_at=? WHERE id=?", + (encoded_snapshot, delivery_path, version, now, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff_id, + "submission_id": submission_id, + "previous_candidate_sha256": candidate["candidate_sha256"], + "revisions_requested": revisions, + "worker_invocations": invocations, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'delivery.revision_queued',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return { + "action": "requeued", + "version": version, + "replayed": False, + "revisions_requested": revisions, + "worker_invocations": invocations, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + @_project_terminal + def complete_handoff_terminal( + self, + run_id: str, + handoff_id: str, + submission_id: str, + submission_hash: str, + artifacts: list[dict[str, Any]], + terminal_state: str, + payload: dict[str, Any], + ) -> dict[str, Any]: + """Atomically publish M3 terminal reports and consume the host handoff.""" + if terminal_state not in {"succeeded", "failed"}: + raise ContractError("branch review terminal state is invalid") + if not all( + isinstance(value, str) and value + for value in (handoff_id, submission_id, submission_hash) + ): + raise ContractError("terminal handoff identifiers are invalid") + prepared = self._prepare_exact_artifacts( + run_id, artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS, + ) + encoded_payload = canonical_json(payload) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,h.status,h.sequence,s.disposition " + "FROM runs r JOIN handoffs h ON h.run_id=r.id " + "JOIN handoff_submissions s ON s.handoff_id=h.id " + "WHERE r.id=? AND h.id=? AND s.submission_id=? " + "AND s.submission_hash=? AND s.outcome='recorded'", + (run_id, handoff_id, submission_id, submission_hash), + ).fetchone() + if not row: + raise ConflictError("recorded terminal submission is missing") + expected_state = "succeeded" if row["disposition"] == "accept" else "failed" + if terminal_state != expected_state: + raise ContractError("terminal state contradicts the lead disposition") + if row["state"] in TERMINAL_STATES: + if row["state"] != terminal_state or row["status"] != "consumed": + raise ConflictError("terminal handoff was completed differently") + self.connection.execute("COMMIT") + return { + "state": row["state"], + "version": row["version"], + "replayed": True, + } + latest_sequence = self.connection.execute( + "SELECT MAX(sequence) FROM handoffs WHERE run_id=?", (run_id,), + ).fetchone()[0] + if (row["state"] != "awaiting_host" + or row["phase"] != "handoff_submitted" + or row["status"] != "submitted" + or row["sequence"] != latest_sequence): + raise ConflictError("terminal handoff is no longer current") + version, _ = self._reference_prepared_artifacts( + run_id, row["version"], prepared, + ) + now, version = _utc_now(), version + 1 + self.connection.execute( + "UPDATE handoffs SET status='consumed',closed_at=? WHERE id=?", + (now, handoff_id), + ) + self.connection.execute( + "UPDATE claims SET active=0 WHERE run_id=? AND handoff_id=?", + (run_id, handoff_id), + ) + self.connection.execute( + "UPDATE runs SET state=?,phase=NULL,version=?,updated_at=? WHERE id=?", + (terminal_state, version, now, run_id), + ) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,?,?,?)", + (run_id, version, f"run.{terminal_state}", encoded_payload, now), + ) + self.connection.execute("COMMIT") + return {"state": terminal_state, "version": version, "replayed": False} + except Exception: + self.connection.execute("ROLLBACK") + raise + + @_project_terminal + def cancel_host_wait( + self, + run_id: str, + *, + now: datetime | None = None, + terminal_artifacts: list[dict[str, Any]] | None = None, + ) -> int: + prepared = ( + self._prepare_exact_artifacts( + run_id, terminal_artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS, + ) + if terminal_artifacts is not None else None + ) + self.connection.execute("BEGIN IMMEDIATE") + try: + timestamp = _authoritative_now(now).isoformat() + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + if not run: + raise ContractError("run does not exist") + if run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["version"] + handoff = self.connection.execute( + "SELECT id,status FROM handoffs WHERE run_id=? ORDER BY sequence DESC LIMIT 1", + (run_id,), + ).fetchone() + if (run["state"] != "awaiting_host" or not handoff + or handoff["status"] not in {"open", "submitted"} + or run["phase"] not in {None, "handoff_submitted"}): + raise ConflictError("run is not awaiting a cancellable host handoff") + if prepared is None: + path, digest, size, receipt_time = self._terminal_receipt( + run_id, "cancelled", "awaiting_host", None, now=timestamp, + ) + version = self._reference_terminal_receipt( + run_id, run["version"], path, digest, size, receipt_time, + ) + else: + version, _ = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version += 1 + self.connection.execute( + "UPDATE handoffs SET status='cancelled',closed_at=? WHERE id=?", + (timestamp, handoff["id"]), + ) + self.connection.execute( + "UPDATE claims SET active=0,fencing_token=fencing_token+1 " + "WHERE run_id=? AND kind='host' AND handoff_id=?", + (run_id, handoff["id"]), + ) + self.connection.execute( + "UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?", + (version, timestamp, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff["id"], + "receipt": "result-receipt.json", + "terminal_reports": prepared is not None, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id, version, payload, timestamp), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + @_project_terminal + def fail_queued_budget( + self, + run_id: str, + expected_version: int, + error: dict[str, Any], + terminal_artifacts: list[dict[str, Any]], + ) -> int: + """Fail a paused workflow before another worker launch can consume budget.""" + prepared = self._prepare_exact_artifacts( + run_id, terminal_artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS, + ) + encoded_error = canonical_json(error) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + if (not run or run["state"] not in {"queued", "awaiting_host"} + or run["phase"] is not None + or run["version"] != expected_version): + raise ConflictError("budget exhaustion is no longer current") + if self.connection.execute( + "SELECT 1 FROM supervisor_claims WHERE run_id=? AND active=1", + (run_id,), + ).fetchone(): + raise ConflictError("budget exhaustion raced with a supervisor") + version, _ = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + now, version = _utc_now(), version + 1 + self.connection.execute( + "UPDATE handoffs SET status='consumed',closed_at=? WHERE run_id=? " + "AND status IN ('open','submitted')", + (now, run_id), + ) + self.connection.execute( + "UPDATE claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='failed',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = dict(error) + payload["receipt"] = "result-receipt.json" + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.failed',?,?)", + (run_id, version, canonical_json(payload), now), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + @_project_terminal + def cancel_council_queued(self, run_id: str, expected_version: int, artifacts: list) -> int: + prepared = self._prepare_exact_artifacts(run_id, artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.run(run_id) + snapshot = json.loads(run["mutable_snapshot"]) + if (run["version"] != expected_version or run["state"] != "queued" or run["phase"] is not None + or snapshot["task"]["workflow"] != "council-decision"): + raise ConflictError("Council queued cancellation is fenced") + if self.connection.execute("SELECT 1 FROM supervisor_claims WHERE run_id=? AND active=1", (run_id,)).fetchone(): + raise ConflictError("Council queued cancellation raced with an owned launcher") + version, _ = self._reference_prepared_artifacts(run_id, run["version"], prepared) + now, version = _utc_now(), version + 1 + self.connection.execute("UPDATE attempts SET status='recovery_required',finished_at=? WHERE run_id=? AND status='reserved'", (now, run_id)) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute("UPDATE claims SET active=0,fencing_token=fencing_token+1 WHERE run_id=?", (run_id,)) + self.connection.execute("UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.cancelled',?,?)", + (run_id, version, canonical_json({"receipt": "result-receipt.json", "workflow": "council-decision"}), now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + @_project_terminal + def cancel_queued(self, run_id: str) -> int: + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute("SELECT state,phase,version FROM runs WHERE id=?", (run_id,)).fetchone() + if not row: + raise ContractError("run does not exist") + if row["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT"); return row["version"] + if row["state"] != "queued" or row["phase"] is not None: + raise ConflictError("queued run is owned by another operation") + path,digest,size,now=self._terminal_receipt(run_id,"cancelled","queued",None) + version=self._reference_terminal_receipt(run_id,row["version"],path,digest,size,now)+1 + self.connection.execute("UPDATE runs SET state='cancelled',version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id,version,canonical_json({"receipt":"result-receipt.json"}),now), + ) + self.connection.execute("COMMIT"); return version + except Exception: + self.connection.execute("ROLLBACK"); raise + + @_project_terminal + def cancel_launching(self, run_id: str) -> int: + self.connection.execute("BEGIN IMMEDIATE") + try: + row=self.connection.execute("SELECT state,phase,version FROM runs WHERE id=?",(run_id,)).fetchone() + if not row or row["state"]!="queued" or row["phase"]!="launching": raise ConflictError("run is not launching") + path,digest,size,now=self._terminal_receipt(run_id,"cancelled","launching",None) + version=self._reference_terminal_receipt(run_id,row["version"],path,digest,size,now)+1 + self.connection.execute("UPDATE attempts SET status='recovery_required',finished_at=? WHERE run_id=? AND status='reserved'",(now,run_id)) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?",(run_id,)) + self.connection.execute("UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?",(version,now,run_id)) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id,version,canonical_json({"receipt":"result-receipt.json"}),now), + ) + self.connection.execute("COMMIT"); return version + except Exception: self.connection.execute("ROLLBACK"); raise + + def events_page(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str, Any]: + if type(after) is not int or after < 0 or type(limit) is not int or not 1 <= limit <= 1000: + raise ContractError("event cursor/limit is invalid") + if not self.connection.execute("SELECT 1 FROM runs WHERE id=?", (run_id,)).fetchone(): + raise ContractError("run does not exist") + rows = self.connection.execute("SELECT id,run_version,type,payload,created_at FROM events WHERE run_id=? AND id>? ORDER BY id LIMIT ?", (run_id, after, limit + 1)).fetchall() + page, more = rows[:limit], len(rows) > limit + consumed = page[-1]["id"] if page else after + return {"events": [{**dict(row), "payload": json.loads(row["payload"])} for row in page], "next_cursor": consumed, "has_more": more} + + def events_for_run(self, run_id: str) -> list[dict[str, Any]]: + if not self.connection.execute( + "SELECT 1 FROM runs WHERE id=?", (run_id,), + ).fetchone(): + raise ContractError("run does not exist") + rows = self.connection.execute( + "SELECT id,run_version,type,payload,created_at FROM events " + "WHERE run_id=? ORDER BY id", + (run_id,), + ).fetchall() + events = [] + for row in rows: + try: + payload = json.loads(row["payload"]) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("persisted event payload is invalid") from exc + if canonical_json(payload) != row["payload"]: + raise ConflictError("persisted event payload is not canonical") + events.append({**dict(row), "payload": payload}) + return events + + def status_snapshot( + self, run_id: str, + ) -> tuple[dict[str, Any], dict[str, Any] | None, HandoffSnapshot | None]: + self.connection.execute("BEGIN") + try: + run = self.connection.execute("SELECT * FROM runs WHERE id=?", (run_id,)).fetchone() + if not run: + raise ContractError("run does not exist") + attempt = self.connection.execute("SELECT * FROM attempts WHERE run_id=? ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() + handoff = self.connection.execute( + "SELECT * FROM handoffs WHERE run_id=? ORDER BY sequence DESC LIMIT 1", + (run_id,), + ).fetchone() + claim_row = None + if handoff: + claim_row = self.connection.execute( + "SELECT run_id,handoff_id,owner_id,fencing_token,lease_expires_at," + "? AS run_version FROM claims WHERE run_id=? AND kind='host' " + "AND handoff_id=? AND active=1", + (run["version"], run_id, handoff["id"]), + ).fetchone() + handoff_snapshot = self._handoff_snapshot_from_rows( + run_id, run, handoff, claim_row, + ) + self.connection.execute("COMMIT") + return dict(run), dict(attempt) if attempt else None, handoff_snapshot + except Exception: + self.connection.execute("ROLLBACK") + raise + + def result_snapshot(self, run_id: str) -> tuple[dict[str, Any], list[dict[str, Any]]]: + self.connection.execute("BEGIN") + try: + run=self.connection.execute("SELECT * FROM runs WHERE id=?",(run_id,)).fetchone() + if not run: raise ContractError("run does not exist") + artifacts=self.connection.execute("SELECT id,name,path,sha256,byte_size,created_at FROM artifacts WHERE run_id=? ORDER BY created_at,id",(run_id,)).fetchall() + self.connection.execute("COMMIT"); return dict(run),[dict(row) for row in artifacts] + except Exception: self.connection.execute("ROLLBACK"); raise + + def artifacts_for_run(self, run_id: str) -> list[dict[str, Any]]: + return [dict(row) for row in self.connection.execute("SELECT id,name,path,sha256,byte_size,created_at FROM artifacts WHERE run_id=? ORDER BY created_at,id", (run_id,))] + + def artifact_named(self, run_id: str, name: str) -> dict[str, Any] | None: + row=self.connection.execute("SELECT id,name,path,sha256,byte_size,created_at FROM artifacts WHERE run_id=? AND name=?",(run_id,name)).fetchone() + return dict(row) if row else None + + def attempt(self, run_id: str) -> dict[str, Any] | None: + row = self.connection.execute("SELECT * FROM attempts WHERE run_id=? ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() + return dict(row) if row else None + + def attempts_for_run(self, run_id: str) -> list[dict[str, Any]]: + return [ + dict(row) for row in self.connection.execute( + "SELECT * FROM attempts WHERE run_id=? ORDER BY created_at,id", + (run_id,), + ) + ] + + def run(self, run_id: str) -> dict[str, Any]: + row = self.connection.execute("SELECT * FROM runs WHERE id=?", (run_id,)).fetchone() + if not row: raise ContractError("run does not exist") + return dict(row) diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py new file mode 100644 index 0000000..5534c27 --- /dev/null +++ b/plugin/core/src/devsquad/supervisor.py @@ -0,0 +1,1042 @@ +"""Bounded process-group supervision for persisted M2 attempts.""" + +from __future__ import annotations + +from dataclasses import dataclass +import ctypes +from datetime import datetime, timezone +import hashlib +import os +import signal +import stat +import subprocess +import sys +import threading +import time +from typing import Any, BinaryIO +from pathlib import Path +import json + +from .contracts import ContractError, LaunchSpec +from .claude_identity import strict_json, validate_failure +from .reports import build_early_terminal_reports +from .store import AttemptReservation, ConflictError, Store, canonical_json +from .workflows import ( + decode_branch_review_evidence, + decode_headless_lead_evidence, + review_mode, + require_independent_delivery_review, + validate_implementation_evidence, + validate_review_document, +) +from .workspaces import ( + freeze_delivery_candidate, + prepare_check_workspace, + prepare_review_workspace, + repo_relative_config, +) + + +def _require_attempt_profile( + evidence: dict[str, Any], snapshot: dict[str, Any], + attempt: dict[str, Any], role: str, +) -> None: + """A permitted fallback is not proof that this invocation used it.""" + try: + routed = snapshot["routing"]["roles"][role] + candidates = [routed["selected"], *routed["fallbacks"]] + index = attempt["profile_index"] + if type(index) is not int or not 0 <= index < len(candidates): + raise ContractError("durable attempt profile index is invalid") + selected = candidates[index] + if (selected["profile_id"] != attempt["profile_id"] + or canonical_json(evidence["attempt"]["selected_profile"]) + != canonical_json(selected)): + raise ContractError("imported evidence differs from the actual attempt profile") + except (KeyError, TypeError, IndexError) as exc: + raise ContractError("durable attempt profile binding is missing") from exc + + +def _open_stdin_artifact(path: str) -> BinaryIO: + candidate = Path(path) + if candidate.is_symlink(): + raise ContractError("stdin artifact must be a regular non-symlink file") + flags = os.O_RDONLY + if hasattr(os, "O_NOFOLLOW"): + flags |= os.O_NOFOLLOW + try: + descriptor = os.open(candidate, flags) + except OSError as exc: + raise ContractError("stdin artifact must be a readable regular non-symlink file") from exc + try: + if not stat.S_ISREG(os.fstat(descriptor).st_mode): + raise ContractError("stdin artifact must be a regular non-symlink file") + return os.fdopen(descriptor, "rb") + except Exception: + os.close(descriptor) + raise + + +def process_start_identity(pid: int) -> str | None: + if sys.platform.startswith("linux"): + try: + fields = open(f"/proc/{pid}/stat", encoding="ascii").read().rsplit(") ", 1)[1].split() + return f"linux-start-ticks:{fields[19]}" + except (OSError, IndexError): + return None + if sys.platform == "darwin": + class ProcBsdInfo(ctypes.Structure): + _fields_ = [("prefix", ctypes.c_byte * 120), ("start_sec", ctypes.c_uint64), ("start_usec", ctypes.c_uint64)] + info = ProcBsdInfo() + try: + function = ctypes.CDLL("/usr/lib/libproc.dylib").proc_pidinfo + function.argtypes = [ctypes.c_int, ctypes.c_int, ctypes.c_uint64, ctypes.c_void_p, ctypes.c_int] + function.restype = ctypes.c_int + copied = function(pid, 3, 0, ctypes.byref(info), ctypes.sizeof(info)) + except OSError: + return None + return f"darwin-start:{info.start_sec}:{info.start_usec}" if copied == ctypes.sizeof(info) else None + return None + + +def inspect_process(pid: int, pgid: int, expected_start: str) -> str: + observed = process_start_identity(pid) + if observed is None: + try: + os.killpg(pgid, 0) + except ProcessLookupError: + return "dead" + except PermissionError: + return "ambiguous" + try: + if not _live_group_exists(pgid): + return "dead" + except RuntimeError: + pass + return "ambiguous" + try: + observed_pgid = os.getpgid(pid) + except (ProcessLookupError, PermissionError): + return "ambiguous" + return "live" if observed == expected_start and observed_pgid == pgid else "ambiguous" + + +def _live_group_exists(pgid: int) -> bool: + result = subprocess.run(["/bin/ps", "-axo", "pgid=,stat="], text=True, capture_output=True, check=False) + if result.returncode != 0: + raise RuntimeError("process-group inventory failed") + parsed = 0 + for line in result.stdout.splitlines(): + fields = line.split() + if len(fields) < 2 or not fields[0].isdigit(): + raise RuntimeError("process-group inventory was malformed") + parsed += 1 + if int(fields[0]) == pgid and not fields[1].startswith("Z"): + return True + if parsed == 0: + raise RuntimeError("process-group inventory was empty") + return False + + +def _read_child_identity(path: Path) -> tuple[int, int, str]: + if path.is_symlink() or not path.is_file(): + raise ValueError("child identity record is not a regular file") + value = json.loads(path.read_text()) + if not isinstance(value, dict) or set(value) != {"pid", "pgid", "process_start_id"}: + raise ValueError("child identity record fields are invalid") + if (type(value["pid"]) is not int or value["pid"] <= 0 + or type(value["pgid"]) is not int or value["pgid"] <= 0 + or not isinstance(value["process_start_id"], str) + or not value["process_start_id"]): + raise ValueError("child identity record values are invalid") + return value["pid"], value["pgid"], value["process_start_id"] + + +class BoundedDrain: + def __init__(self, stream: BinaryIO, limit: int): + self.stream, self.limit = stream, limit + self.content = bytearray() + self.total_bytes = 0 + self.digest = hashlib.sha256() + self.thread = threading.Thread(target=self._run, daemon=True) + + def _run(self) -> None: + while True: + chunk = self.stream.read(65536) + if not chunk: + return + self.total_bytes += len(chunk) + self.digest.update(chunk) + remaining = self.limit - len(self.content) + if remaining > 0: + self.content.extend(chunk[:remaining]) + + def start(self) -> None: + self.thread.start() + + def finish(self) -> dict[str, Any]: + self.thread.join(timeout=5) + if self.thread.is_alive(): + raise RuntimeError("output drain did not finish") + self.stream.close() + return {"total_bytes": self.total_bytes, "captured_bytes": len(self.content), "truncated": self.total_bytes > len(self.content), "full_sha256": self.digest.hexdigest()} + + +@dataclass +class RunningAttempt: + reservation: AttemptReservation + process: subprocess.Popen[bytes] + start_identity: str + stdout: BoundedDrain + stderr: BoundedDrain + + +@dataclass +class DurableAttempt: + reservation: AttemptReservation + process: subprocess.Popen[bytes] + paths: dict[str, str] + + +class Supervisor: + def __init__(self, store: Store, *, output_limit: int = 1024 * 1024, grace_seconds: float = 5.0): + if output_limit <= 0 or grace_seconds < 0: + raise ContractError("supervisor bounds must be positive") + self.store, self.output_limit, self.grace_seconds = store, output_limit, grace_seconds + + def launch(self, run_id: str, expected_version: int, spec: LaunchSpec, owner_id: str, package_digest: str) -> RunningAttempt: + reservation = self.store.reserve_attempt( + run_id, + expected_version, + owner_id, + package_digest, + account_pool_id=spec.requested.account_pool, + ) + environment = os.environ.copy(); environment.update(spec.environment) + try: + stdin_stream: Any = subprocess.DEVNULL + if spec.stdin_path is not None: + stdin_stream = _open_stdin_artifact(spec.stdin_path) + try: + process = subprocess.Popen(list(spec.argv), cwd=spec.cwd, env=environment, stdin=stdin_stream, stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True) + finally: + if stdin_stream is not subprocess.DEVNULL: + stdin_stream.close() + except Exception: + self.store.fail_launch(reservation, "process spawn failed") + raise + assert process.stdout is not None and process.stderr is not None + started = process_start_identity(process.pid) + if started is None: + returncode = process.poll() + if returncode is None: + try: os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: pass + process.wait(timeout=2) + process.stdout.close(); process.stderr.close() + self.store.fail_launch(reservation, "live child had no strong process identity") + raise RuntimeError("live child identity could not be persisted") + started = f"exited-before-observation:{returncode}:non-signalable" + pgid = process.pid + else: + pgid = os.getpgid(process.pid) + try: + self.store.mark_attempt_running(reservation, process.pid, pgid, started) + except Exception: + if process.poll() is None: + try: os.killpg(pgid, signal.SIGKILL) + except ProcessLookupError: pass + process.wait(timeout=2) + process.stdout.close(); process.stderr.close() + self.store.fail_launch(reservation, "process identity could not be persisted") + raise + stdout, stderr = BoundedDrain(process.stdout, self.output_limit), BoundedDrain(process.stderr, self.output_limit) + stdout.start(); stderr.start() + return RunningAttempt(reservation, process, started, stdout, stderr) + + def launch_durable( + self, + run_id: str, + expected_version: int, + spec: LaunchSpec, + owner_id: str, + package_digest: str, + *, + role: str = "worker", + profile_id: str | None = None, + profile_index: int | None = None, + ) -> DurableAttempt: + reservation = self.store.reserve_attempt( + run_id, + expected_version, + owner_id, + package_digest, + role, + account_pool_id=spec.requested.account_pool, + profile_id=profile_id, + profile_index=profile_index, + ) + directory = self.store.artifacts / run_id / f".{reservation.attempt_id}.spool" + directory.mkdir(parents=True, exist_ok=False) + paths = {name: str(directory / filename) for name, filename in { + "stdout_spool":"stdout.capture", "stderr_spool":"stderr.capture", "stdout_meta":"stdout.meta.json", "stderr_meta":"stderr.meta.json", "exit_record":"exit.json", "child_record":"child.json"}.items()} + gate_read, gate_write = os.pipe() + process = None + stdin_stream = None + identity_committed = False + gate_released = False + try: + if spec.stdin_path is not None: + stdin_stream = _open_stdin_artifact(spec.stdin_path) + command=[sys.executable,"-P","-m","devsquad.attempt_runner","--gate-fd",str(gate_read), + "--database",str(self.store.database),"--artifacts",str(self.store.artifacts), + "--run-id",run_id,"--attempt-token",reservation.attempt_token, + "--supervisor-token",str(reservation.supervisor_token),"--stdout",paths["stdout_spool"], + "--stderr",paths["stderr_spool"],"--exit-record",paths["exit_record"], + "--child-record",paths["child_record"],"--limit",str(self.output_limit), + "--timeout",str(spec.timeout_seconds),"--grace",str(self.grace_seconds),"--",*spec.argv] + if stdin_stream is not None: + command[4:4] = ["--stdin-fd", str(stdin_stream.fileno())] + environment=os.environ.copy(); environment.update(spec.environment) + passed_fds=(gate_read,) if stdin_stream is None else (gate_read,stdin_stream.fileno()) + process=subprocess.Popen(command,cwd=spec.cwd,env=environment,stdin=subprocess.DEVNULL,stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL,pass_fds=passed_fds,start_new_session=True) + if stdin_stream is not None: + stdin_stream.close(); stdin_stream=None + os.close(gate_read) + started=process_start_identity(process.pid) + if started is None: raise RuntimeError("gated child has no strong process identity") + self.store.mark_attempt_running(reservation,process.pid,process.pid,started,paths) + identity_committed = True + self._release_runner_gate(gate_write) + gate_released = True + os.close(gate_write) + return DurableAttempt(reservation,process,paths) + except Exception: + if stdin_stream is not None: + stdin_stream.close() + try: os.close(gate_write) + except OSError: pass + for fd in (gate_read,): + try: os.close(fd) + except OSError: pass + if process and process.poll() is None: + try: os.killpg(process.pid,signal.SIGKILL) + except ProcessLookupError: pass + process.wait(timeout=2) + if not identity_committed: + self.store.fail_launch(reservation,"durable gated launch failed") + elif not gate_released: + self._recover_unstarted_attempt( + run_id, reservation.attempt_token, + "runner gate could not be released", + ) + else: + self.store.block_recovery( + run_id, reservation.attempt_token, + "coordinator failed after releasing the runner gate", + ) + raise + + @staticmethod + def _release_runner_gate(gate_write: int) -> None: + os.write(gate_write, b"1") + + def wait_durable(self, handle: DurableAttempt, timeout_seconds: float) -> int: + deadline=time.monotonic()+timeout_seconds + self.grace_seconds + 5 + while handle.process.poll() is None and time.monotonic() tuple[int, str]: + artifacts = None + run = self.store.run(run_id) + snapshot = json.loads(run["mutable_snapshot"]) + if run["state"] == "cancelling" and snapshot.get("task", {}).get("workflow") == "council-decision": + from .council_runtime import prepared_reports + artifacts = prepared_reports(self.store, run_id, snapshot, "cancelled") + return self.store.recover_unstarted_attempt(run_id, attempt_token, reason, terminal_artifacts=artifacts) + + def _commit_review_handoff( + self, + run_id: str, + attempt: dict[str, Any], + stream_artifacts: list[dict[str, Any]], + metadata: dict[str, Any], + snapshot: dict[str, Any], + stdout: bytes, + ) -> str: + evidence = decode_branch_review_evidence(stdout, snapshot) + _require_attempt_profile(evidence, snapshot, attempt, "reviewer") + suffix = attempt["id"] + documents = { + f"review-{suffix}.json": evidence["review"], + f"checks-{suffix}.json": { + "schema_version": 1, + "candidate_sha256": evidence["candidate_sha256"], + "target_oid": evidence["target_oid"], + "results": evidence["checks"], + }, + f"evaluation-{suffix}.json": evidence["evaluation"], + f"review-attempt-{suffix}.json": evidence["attempt"], + } + artifacts = list(stream_artifacts) + evidence_references = [] + for name, document in documents.items(): + content = (canonical_json(document) + "\n").encode() + path, digest, size = self.store.finalize_artifact(run_id, name, content) + artifacts.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + evidence_references.append({"name": name, "sha256": digest}) + packet = { + "schema_version": 1, + "workflow": evidence["workflow"], + "candidate_sha256": evidence["candidate_sha256"], + "base_oid": evidence["base_oid"], + "target_oid": evidence["target_oid"], + "review": evidence["review"], + "checks": evidence["checks"], + "evaluation": evidence["evaluation"], + "attempt_id": attempt["id"], + "attempt": evidence["attempt"], + "artifacts": evidence_references, + "instructions": ( + "Inspect the bound review and check evidence, then submit exactly one " + "accept, revise, or reject disposition." + ), + } + return self.store.commit_durable_handoff( + run_id, + attempt["attempt_token"], + artifacts, + metadata, + packet, + ) + + def _commit_delivery_candidate( + self, + run_id: str, + attempt: dict[str, Any], + stream_artifacts: list[dict[str, Any]], + metadata: dict[str, Any], + snapshot: dict[str, Any], + stdout: bytes, + ) -> str: + evidence = validate_implementation_evidence( + strict_json(stdout), snapshot, + ) + _require_attempt_profile(evidence, snapshot, attempt, "implementer") + task = snapshot["task"] + delivery = snapshot["delivery_workspace"] + source_repo = Path(task["project"]["repo_path"]).resolve(strict=True) + workspace = Path(delivery["path"]).resolve(strict=True) + iterations = list(snapshot.get("delivery_iterations", [])) + iteration = len(iterations) + 1 + parent_oid = ( + iterations[-1]["candidate"]["commit_oid"] if iterations + else delivery["baseline_oid"] + ) + candidate, patch = freeze_delivery_candidate( + source_repo, + workspace, + delivery["baseline_oid"], + task["scope"]["write_paths"], + run_id, + parent_oid=parent_oid, + ) + project_id = self.store.run(run_id)["project_id"] + scope_paths = tuple(dict.fromkeys( + task["scope"]["read_paths"] + task["scope"]["write_paths"] + )) + config_paths = ( + () + if "profiles" in task["routing"] + else tuple( + repo_relative_config(source_repo, task["routing"][label], label) + for label in ("profiles_file", "policy_file") + ) + ) + review_workspace = prepare_review_workspace( + source_repo, + self.store.database.parent, + project_id, + run_id, + delivery["baseline_oid"], + candidate["commit_oid"], + scope_paths, + required_clean_paths=config_paths, + candidate_sha256=candidate["candidate_sha256"], + workspace_name=( + "review-worktree" if iteration == 1 + else f"review-worktree-{iteration}" + ), + ) + check_workspace = prepare_check_workspace( + source_repo, + self.store.database.parent, + project_id, + run_id, + candidate["commit_oid"], + scope_paths, + required_clean_paths=config_paths, + workspace_name=( + "check-worktree" if iteration == 1 + else f"check-worktree-{iteration}" + ), + ) + candidate_record = { + **candidate, + "iteration": iteration, + "patch_artifact": f"candidate-{iteration}.patch", + "implementation_artifact": ( + f"implementation-attempt-{attempt['id']}.json" + ), + } + new_snapshot = json.loads(canonical_json(snapshot)) + new_snapshot["candidate"] = candidate_record + new_snapshot["workspace"] = review_workspace + new_snapshot["check_workspace"] = check_workspace + pending_review = new_snapshot.pop("pending_review_fixture", None) + if pending_review is not None: + fixture_document = { + "schema_version": 1, + "candidate_sha256": review_workspace["candidate_sha256"], + "base_oid": review_workspace["base_oid"], + "target_oid": review_workspace["target_oid"], + "review_mode": review_mode(task), + **pending_review, + } + new_snapshot["internal_review_fixture"] = validate_review_document( + fixture_document, task, review_workspace, + ) + revision_request = new_snapshot.pop("revision_request", None) + iteration_record = { + "iteration": iteration, + "candidate": candidate_record, + "implementation": evidence, + "workspace": review_workspace, + "check_workspace": check_workspace, + } + if revision_request is not None: + iteration_record["revision_request"] = revision_request + iterations.append(iteration_record) + new_snapshot["delivery_iterations"] = iterations + + artifacts = list(stream_artifacts) + documents = { + f"candidate-{iteration}.json": candidate_record, + f"implementation-attempt-{attempt['id']}.json": evidence, + } + for name, document in documents.items(): + content = (canonical_json(document) + "\n").encode() + path, digest, size = self.store.finalize_artifact( + run_id, name, content, + ) + artifacts.append({ + "name": name, "path": path, "sha256": digest, + "byte_size": size, + }) + patch_name = f"candidate-{iteration}.patch" + path, digest, size = self.store.finalize_artifact( + run_id, patch_name, patch, + ) + if digest != candidate["patch_sha256"] or size != candidate["patch_bytes"]: + raise ContractError("saved candidate patch differs from its identity") + artifacts.append({ + "name": patch_name, "path": path, "sha256": digest, + "byte_size": size, + }) + return self.store.commit_delivery_candidate( + run_id, + attempt["attempt_token"], + artifacts, + metadata, + new_snapshot, + review_workspace["path"], + candidate_record, + ) + + def import_durable(self, run_id: str) -> str: + attempt=self.store.attempt(run_id) + if not attempt or attempt["status"] not in {"running","cancelling"}: + return "already_finalized" + identity=inspect_process(attempt["pid"],attempt["pgid"],attempt["process_start_id"]) + if identity=="live": return "live" + receipt_path=Path(attempt["exit_record"] or "") + if identity=="ambiguous" and not receipt_path.is_file(): + self.store.block_recovery(run_id,attempt["attempt_token"],"attempt runner identity is ambiguous") + return "ownership_ambiguous" + if not receipt_path.is_file(): + child_path=Path(attempt["child_record"] or "") + if child_path.exists() or child_path.is_symlink(): + try: + child = _read_child_identity(child_path) + if inspect_process(*child)!="dead": + self.store.block_recovery(run_id,attempt["attempt_token"],"runner died while its child may still be live") + return "ownership_ambiguous" + except (OSError, ValueError, TypeError, json.JSONDecodeError): + self.store.block_recovery(run_id,attempt["attempt_token"],"child identity record is invalid") + return "ownership_ambiguous" + self.store.block_recovery(run_id,attempt["attempt_token"],"runner and child died without an exit receipt",release_writer=True) + return "recovery_required" + try: + _, disposition = self._recover_unstarted_attempt( + run_id, + attempt["attempt_token"], + "runner died before publishing the gated child identity", + ) + return disposition + except ConflictError: + current = self.store.attempt(run_id) + run = self.store.run(run_id) + if (current and current["attempt_token"] == attempt["attempt_token"] + and current["status"] == "recovery_required" + and run["state"] == "queued"): + return "requeued" + if run["state"] in {"succeeded", "failed", "cancelled"}: + return run["state"] + raise + try: + receipt=json.loads(receipt_path.read_text()) + if (type(receipt.get("returncode")) is not int + or type(receipt.get("cancelled")) is not bool + or type(receipt.get("timed_out")) is not bool + or (receipt["cancelled"] and receipt["timed_out"])): + raise ValueError("invalid receipt") + metadata={name:receipt[name] for name in ("stdout","stderr")} + artifacts=[] + captures={} + for stream,column in (("stdout","stdout_spool"),("stderr","stderr_spool")): + data=Path(attempt[column]).read_bytes(); meta=metadata[stream] + if hashlib.sha256(data).hexdigest()!=meta["captured_sha256"] or len(data)!=meta["captured_bytes"]: + raise ValueError("capture hash mismatch") + captures[stream]=data + logical=f"{attempt['id']}.{stream}" + path,digest,size=self.store.finalize_artifact(run_id,logical,data) + artifacts.append({"name":logical,"path":path,"sha256":digest,"byte_size":size}) + snapshot=json.loads(self.store.run(run_id)["mutable_snapshot"]) + workflow = snapshot.get("task", {}).get("workflow") + managed_workflow = "internal_fake_delay" not in snapshot + workflow_review = managed_workflow and workflow == "branch-review" + workflow_delivery = managed_workflow and workflow == "issue-delivery" + workflow_council = managed_workflow and workflow == "council-decision" + role = attempt.get("role", "worker") + semantic_error=None + native_failure = None + if role == "implementer" and workflow_delivery: + # Only normalized, profile-bound diagnostics may reach reports. + # Raw output remains a hash-bound private artifact, not identity. + try: + routed = snapshot["routing"]["roles"][role] + selected = [routed["selected"], *routed["fallbacks"]][attempt["profile_index"]] + adapter = snapshot.get("implementation_adapters", {}).get(selected["profile_id"]) + if isinstance(adapter, dict) and adapter.get("harness") == "claude": + native_failure = validate_failure( + strict_json(captures["stdout"]), selected["profile_sha256"], + ) + except (ContractError, KeyError, TypeError, IndexError): + pass + if (workflow_council and not receipt["cancelled"] and not receipt["timed_out"] and receipt["returncode"] == 0): + try: + from .council_runtime import import_stage + return import_stage(self.store, run_id, attempt, artifacts, metadata, snapshot, captures["stdout"]) + except ContractError as exc: + semantic_error = str(exc) + receipt["error"] = "COUNCIL_OUTPUT_INVALID" + receipt["message"] = semantic_error + if (role == "implementer" and workflow_delivery + and not receipt["cancelled"] + and not receipt["timed_out"] and receipt["returncode"] == 0): + try: + return self._commit_delivery_candidate( + run_id, + attempt, + artifacts, + metadata, + snapshot, + captures["stdout"], + ) + except (ContractError, UnicodeDecodeError, json.JSONDecodeError) as exc: + semantic_error = str(exc) + receipt["error"] = "IMPLEMENTATION_OUTPUT_INVALID" + receipt["message"] = semantic_error + elif (role == "lead" and (workflow_review or workflow_delivery) + and not receipt["cancelled"] + and not receipt["timed_out"] and receipt["returncode"]==0): + try: + handoff = self.store.handoff_snapshot(run_id) + if handoff is None or handoff.status != "open": + raise ContractError("headless lead handoff is missing") + frozen_handoff = { + "handoff_id": handoff.handoff_id, + "packet": handoff.packet, + "packet_sha256": handoff.packet_sha256, + } + evidence = decode_headless_lead_evidence( + captures["stdout"], snapshot, frozen_handoff, + ) + _require_attempt_profile(evidence, snapshot, attempt, "lead") + if evidence["choice"]["disposition"] == "accept": + require_independent_delivery_review(snapshot, handoff.packet) + content = (canonical_json(evidence) + "\n").encode() + name = f"lead-attempt-{attempt['id']}.json" + path,digest,size=self.store.finalize_artifact(run_id,name,content) + artifacts.append({ + "name":name,"path":path,"sha256":digest,"byte_size":size, + }) + return self.store.commit_headless_lead( + run_id, attempt["attempt_token"], artifacts, metadata, + ) + except ContractError as exc: + semantic_error=str(exc) + receipt["error"]="HEADLESS_LEAD_OUTPUT_INVALID" + receipt["message"]=semantic_error + elif (role == "reviewer" and (workflow_review or workflow_delivery) + and not receipt["cancelled"] + and not receipt["timed_out"] and receipt["returncode"]==0): + try: + return self._commit_review_handoff( + run_id, attempt, artifacts, metadata, snapshot, captures["stdout"], + ) + except ContractError as exc: + semantic_error=str(exc) + receipt["error"]="WORKFLOW_OUTPUT_INVALID" + receipt["message"]=semantic_error + terminal="cancelled" if receipt["cancelled"] else ("failed" if receipt["timed_out"] or receipt["returncode"]!=0 or semantic_error else "succeeded") + payload={"returncode":receipt["returncode"],"receipt":"result-receipt.json"} + if receipt["timed_out"]: payload["error"]="TIMEOUT" + if semantic_error: + payload["error"]=( + "HEADLESS_LEAD_OUTPUT_INVALID" + if role == "lead" else "WORKFLOW_OUTPUT_INVALID" + ) + payload["message"]=semantic_error + if workflow_review or workflow_delivery or workflow_council: + from .service import Service + prior_attempts, prior_dispositions = ([], []) if workflow_council else Service._saved_review_progress( + self.store, + run_id, + snapshot, + self.store.handoff_snapshot(run_id), + ) + if receipt["cancelled"]: + report_error = None + elif receipt["timed_out"]: + report_error = { + "error": "TIMEOUT", + "message": f"{workflow} worker exceeded its deadline", + } + elif semantic_error: + report_error = { + "error": ( + "HEADLESS_LEAD_OUTPUT_INVALID" + if role == "lead" else "WORKFLOW_OUTPUT_INVALID" + ), + "message": semantic_error, + } + else: + stderr_text = captures["stderr"].decode("utf-8", "replace") + upper_stderr = stderr_text.upper() + provider_error = next(( + code for code in ("AUTH_ERROR", "RATE_LIMITED") + if f"{code}:" in upper_stderr + ), None) + failure_code = provider_error or ( + "HEADLESS_LEAD_FAILED" + if role == "lead" else "REVIEW_WORKER_FAILED" + if role == "reviewer" else "IMPLEMENTATION_WORKER_FAILED" + ) + report_error = { + "error": failure_code, + "message": ( + "headless lead exited before producing a valid disposition" + if role == "lead" + else f"{workflow} worker exited before producing valid evidence" + ), + "returncode": receipt["returncode"], + } + payload.update(report_error) + if native_failure is not None: + diagnostics = native_failure["native_diagnostics"] + metadata["native_diagnostics"] = diagnostics + if report_error is not None: + report_error["native_diagnostics"] = diagnostics + payload["native_diagnostics"] = diagnostics + candidates = [] + profile_index = None + try: + routed_role = snapshot["routing"]["roles"][role] + candidates = [ + routed_role["selected"], *routed_role["fallbacks"], + ] + profile_index = attempt.get("profile_index") + next_profile = candidates[profile_index + 1] + can_fallback = ( + terminal == "failed" + and type(profile_index) is int + and profile_index >= 0 + and profile_index + 1 < len(candidates) + and self.store.worker_invocations(run_id) + < snapshot["task"]["budget"]["max_worker_invocations"] + and self.store.remaining_wall_seconds(run_id) not in {0} + ) + except (IndexError, KeyError, TypeError): + can_fallback = False + next_profile = None + if can_fallback: + fallback_error = { + **(report_error or {}), + "role": role, + "failed_profile_id": attempt.get("profile_id"), + "failed_profile_index": profile_index, + "next_profile_id": next_profile["profile_id"], + } + return self.store.commit_durable_fallback( + run_id, + attempt["attempt_token"], + artifacts, + metadata, + fallback_error, + ) + if workflow_council: + from .council_runtime import prepared_reports + artifacts.extend(prepared_reports(self.store, run_id, snapshot, terminal, error=report_error)) + return self.store.commit_durable_import(run_id, attempt["attempt_token"], artifacts, metadata, terminal, payload) + reports = build_early_terminal_reports( + run_id=run_id, + state=terminal, + task=snapshot["task"], + snapshot=snapshot, + run_artifacts=( + self.store.artifacts_for_run(run_id) + artifacts + ), + events=self.store.events_for_run(run_id), + completed_at=datetime.now(timezone.utc).isoformat(), + phase=role, + error=report_error, + attempt={ + "id": attempt["id"], + "role": role, + "returncode": receipt["returncode"], + "cancelled": receipt["cancelled"], + "timed_out": receipt["timed_out"], + **({"native_diagnostics": native_failure["native_diagnostics"]} + if native_failure is not None else {}), + "selected_profile": ( + candidates[profile_index] + if type(profile_index) is int + and 0 <= profile_index < len(candidates) + else None + ), + }, + prior_attempts=prior_attempts, + prior_dispositions=prior_dispositions, + ) + for name in sorted(reports): + content = reports[name] + path,digest,size=self.store.finalize_artifact( + run_id, name, content, + ) + artifacts.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + else: + receipt_bytes=canonical_json(receipt).encode() + path,digest,size=self.store.finalize_artifact(run_id,"result-receipt.json",receipt_bytes) + artifacts.append({"name":"result-receipt.json","path":path,"sha256":digest,"byte_size":size}) + return self.store.commit_durable_import( + run_id,attempt["attempt_token"],artifacts,metadata,terminal,payload, + ) + except (OSError,ValueError,KeyError,TypeError,json.JSONDecodeError): + try: + self.store.block_recovery(run_id,attempt["attempt_token"],"durable receipt or capture is invalid") + return "ownership_ambiguous" + except ConflictError: + current=self.store.attempt(run_id); run=self.store.run(run_id) + if current and current["attempt_token"]==attempt["attempt_token"] and current["status"]=="finished" and run["state"] in {"succeeded","failed","cancelled"}: + return run["state"] + raise + + def _terminate_durable(self, handle: DurableAttempt) -> None: + attempt=self.store.active_attempt(handle.reservation.run_id) + if not attempt or inspect_process(attempt["pid"],attempt["pgid"],attempt["process_start_id"])!="live": + raise ConflictError("durable process identity is unsafe to signal") + os.killpg(attempt["pgid"],signal.SIGTERM) + deadline=time.monotonic()+self.grace_seconds + while handle.process.poll() is None and time.monotonic() int: + """Cancel a blocked worker whose durable runner is confirmed dead.""" + attempt = self.store.attempt(run_id) + if (not attempt or attempt["status"] not in { + "ownership_ambiguous", "recovery_required", "cancelling", + }): + raise ConflictError("run has no orphaned attempt to cancel") + runner_identity = ( + attempt["pid"], attempt["pgid"], attempt["process_start_id"], + ) + if all(value is None for value in runner_identity): + runner = "dead" + elif (type(runner_identity[0]) is int and runner_identity[0] > 0 + and type(runner_identity[1]) is int and runner_identity[1] > 0 + and isinstance(runner_identity[2], str) and runner_identity[2]): + runner = inspect_process(*runner_identity) + else: + raise ConflictError("orphan runner identity record is invalid") + if runner != "dead": + raise ConflictError("orphan runner identity is not confirmed dead") + + child = None + classification = "dead" + child_record = attempt["child_record"] + if child_record: + child_path = Path(child_record) + else: + child_path = None + if child_path is not None and (child_path.exists() or child_path.is_symlink()): + try: + child = _read_child_identity(child_path) + except (OSError, ValueError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError("orphan child identity record is invalid") from exc + classification = inspect_process(*child) + if classification == "ambiguous": + raise ConflictError("orphan child identity is ambiguous or reused") + + self.store.request_recovery_cancel(run_id, attempt["attempt_token"]) + if child is not None and classification == "live": + classification = inspect_process(*child) + if classification == "ambiguous": + raise ConflictError("orphan child identity changed before cancellation") + if classification == "live": + try: + os.killpg(child[1], signal.SIGTERM) + except ProcessLookupError: + pass + deadline = time.monotonic() + self.grace_seconds + while _live_group_exists(child[1]) and time.monotonic() < deadline: + time.sleep(0.05) + if _live_group_exists(child[1]): + try: + os.killpg(child[1], signal.SIGKILL) + except ProcessLookupError: + pass + deadline = time.monotonic() + max(self.grace_seconds, 2.0) + while _live_group_exists(child[1]) and time.monotonic() < deadline: + time.sleep(0.05) + if _live_group_exists(child[1]): + raise ConflictError("orphan child survived bounded cancellation") + return self.store.finish_recovery_cancel( + run_id, attempt["attempt_token"], "orphaned worker cleanup confirmed", + ) + + def _persist_output(self, handle: RunningAttempt) -> dict[str, Any]: + stdout_meta, stderr_meta = handle.stdout.finish(), handle.stderr.finish() + run_id, token = handle.reservation.run_id, handle.reservation.attempt_token + stdout_id = self.store.store_artifact(run_id, f"{handle.reservation.attempt_id}.stdout", bytes(handle.stdout.content)) + stderr_id = self.store.store_artifact(run_id, f"{handle.reservation.attempt_id}.stderr", bytes(handle.stderr.content)) + metadata = {"stdout": stdout_meta, "stderr": stderr_meta} + self.store.record_attempt_output(run_id, token, stdout_id, stderr_id, metadata) + return metadata + + def wait(self, handle: RunningAttempt, timeout_seconds: float) -> int: + deadline = time.monotonic() + timeout_seconds + returncode = None + while returncode is None and time.monotonic() < deadline: + returncode = handle.process.poll() + if returncode is None: + self.store.heartbeat_attempt(handle.reservation.run_id, handle.reservation.attempt_token, handle.reservation.supervisor_token) + time.sleep(min(0.2, max(0, deadline - time.monotonic()))) + if returncode is None: + try: + self._terminate(handle) + except ConflictError: + self.store.block_recovery(handle.reservation.run_id, handle.reservation.attempt_token, "timeout raced with process identity change") + raise + metadata = self._persist_output(handle) + self.store.finish_attempt(handle.reservation.run_id, handle.reservation.attempt_token, "failed", {"error": "TIMEOUT", "output": metadata}) + return 124 + self._cleanup_owned_group(handle) + metadata = self._persist_output(handle) + terminal = "succeeded" if returncode == 0 else "failed" + self.store.finish_attempt(handle.reservation.run_id, handle.reservation.attempt_token, terminal, {"returncode": returncode, "output": metadata}) + return returncode + + def _cleanup_owned_group(self, handle: RunningAttempt) -> None: + pgid = handle.process.pid + handle.process.poll() + if not _live_group_exists(pgid): return + os.killpg(pgid, signal.SIGTERM) + deadline = time.monotonic() + self.grace_seconds + while time.monotonic() < deadline: + handle.process.poll() + if not _live_group_exists(pgid): return + time.sleep(0.05) + try: os.killpg(pgid, signal.SIGKILL) + except ProcessLookupError: return + deadline = time.monotonic() + 2 + while time.monotonic() < deadline: + handle.process.poll() + if not _live_group_exists(pgid): return + time.sleep(0.05) + raise RuntimeError("owned process group remains after KILL") + + def _terminate(self, handle: RunningAttempt) -> None: + attempt = self.store.active_attempt(handle.reservation.run_id) + if not attempt or inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]) != "live": + raise ConflictError("process identity is not safe to signal") + os.killpg(attempt["pgid"], signal.SIGTERM) + deadline = time.monotonic() + self.grace_seconds + while time.monotonic() < deadline: + handle.process.poll() + if not _live_group_exists(attempt["pgid"]): return + time.sleep(0.05) + try: os.killpg(attempt["pgid"], signal.SIGKILL) + except ProcessLookupError: return + try: handle.process.wait(timeout=2) + except subprocess.TimeoutExpired: pass + deadline = time.monotonic() + 2 + while time.monotonic() < deadline: + if not _live_group_exists(attempt["pgid"]): return + time.sleep(0.05) + raise RuntimeError("process group cleanup was not confirmed after KILL") + + def cancel(self, run_id: str, handle: RunningAttempt | None = None) -> int: + version, attempt = self.store.request_cancel(run_id) + if attempt is None: + return version + classification = inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]) + if classification != "live": + if classification == "dead": + return self.store.block_recovery(run_id, attempt["attempt_token"], "child exited before cancellation cleanup", release_writer=True) + return self.store.block_recovery(run_id, attempt["attempt_token"], "process identity is ambiguous or reused") + if handle is None or handle.reservation.attempt_token != attempt["attempt_token"]: + return self.store.block_recovery(run_id, attempt["attempt_token"], "live child requires owner reconciliation before cancellation") + try: + self._terminate(handle) + except ConflictError: + return self.store.block_recovery(run_id, attempt["attempt_token"], "cancellation raced with process identity change") + metadata = self._persist_output(handle) + return self.store.finish_attempt(run_id, attempt["attempt_token"], "cancelled", {"output": metadata}) + + def recover(self, run_id: str) -> str: + attempt = self.store.active_attempt(run_id) + if not attempt: + raise ConflictError("run has no active attempt") + classification = inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]) + if classification == "live": + return "live_owned" + reason = "confirmed dead child" if classification == "dead" else "process identity is ambiguous or reused" + self.store.block_recovery(run_id, attempt["attempt_token"], reason, release_writer=classification == "dead") + return "RECOVERY_REQUIRED" diff --git a/plugin/core/src/devsquad/task_entry.py b/plugin/core/src/devsquad/task_entry.py new file mode 100644 index 0000000..f39a4b1 --- /dev/null +++ b/plugin/core/src/devsquad/task_entry.py @@ -0,0 +1,643 @@ +"""Bounded normal-command task preparation over the strict saved-run contract.""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path, PurePosixPath +import shlex +import subprocess +import sys +import time +from typing import Any, Iterable + +from .adapters import AdapterManifest, harness_version +from .catalog import normalize_models +from .codex_protocol import ( + JsonLinePeer, + discover_models, + initialize_request, + initialized_notification, + receive_response, + request, +) +from .contracts import ContractError +from .probe_process import capture_probe_identity, close_probe, subscription_environment +from .store import canonical_json +from .validation import validate_profile, validate_task + + +SOURCE_ROOT = Path(__file__).resolve().parents[2] +CORE_ROOT = ( + SOURCE_ROOT + if (SOURCE_ROOT / "adapters").is_dir() + else Path(sys.prefix) / "share" / "devsquad" +) +MAX_USER_CHECKS = 12 +NORMAL_POLICY = {"id": "managed-normal-entry", "version": 1} +NORMAL_ALIASES = {"implementer": "implement.balanced", "reviewer": "review.deep"} + + +def _git(repo: Path, *arguments: str) -> str: + try: + completed = subprocess.run( + ["git", "-C", str(repo), *arguments], + text=True, + capture_output=True, + timeout=10, + check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise ContractError("cannot inspect the Git project") from exc + if completed.returncode != 0: + raise ContractError(f"Git project check failed: {' '.join(arguments[:2])}") + return completed.stdout.strip() + + +def resolve_repository(project_dir: str | Path) -> Path: + try: + requested = Path(project_dir).expanduser().resolve(strict=True) + except OSError as exc: + raise ContractError("project directory does not exist") from exc + root = Path(_git(requested, "rev-parse", "--show-toplevel")) + try: + return root.resolve(strict=True) + except OSError as exc: + raise ContractError("Git project root does not exist") from exc + + +def _resolve_commit(repo: Path, reference: str) -> str: + if not isinstance(reference, str) or not reference: + raise ContractError("Git reference must be non-empty") + return _git(repo, "rev-parse", "--verify", f"{reference}^{{commit}}") + + +def discover_codex_identity( + repo: Path, + *, + requested_model: str | None = None, + requested_effort: str | None = None, + timeout_seconds: int = 15, + runtime: Path | None = None, + all_models: bool = False, +) -> dict[str, str] | list[dict[str, str]]: + """Discover one currently available exact Codex model without generating.""" + + manifest = AdapterManifest.load(CORE_ROOT / "adapters/codex/adapter.json") + binary = manifest.resolve_binary() + if binary is None: + raise ContractError("Codex is unavailable; run squad doctor") + version = harness_version(binary) + if version not in manifest.verified_versions: + raise ContractError(f"Codex version is not verified: {version or 'unknown'}") + try: + process = subprocess.Popen( + [binary, "app-server", "--listen", "stdio://"], + cwd=repo, + env=subscription_environment(), + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + text=True, + bufsize=1, + start_new_session=True, + ) + except OSError as exc: + raise ContractError("Codex native process could not start") from exc + start_identity = capture_probe_identity(process) + try: + if start_identity is None: + raise ContractError("Codex native process ownership is unavailable") + assert process.stdin is not None and process.stdout is not None + peer = JsonLinePeer(process.stdout, process.stdin) + peer.send(initialize_request(1)) + initialized = receive_response(peer, 1, timeout_seconds=timeout_seconds) + if "error" in initialized: + raise ContractError("Codex native initialization failed") + peer.send(initialized_notification()) + scope = None + if runtime is None: + models = normalize_models( + "codex", version, + discover_models(peer, first_request_id=10, timeout_seconds=timeout_seconds), + ) + else: + from .native_catalog import NativeCatalogCache, native_account_pool, native_scope, normalize_codex_limits + from .service import Service + + deadline = time.monotonic() + timeout_seconds + def read_native(request_id, method, params=None): + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("native discovery deadline expired") + peer.send(request(request_id, method, params)) + reply = receive_response(peer, request_id, timeout_seconds=remaining) + if "error" in reply or not isinstance(reply.get("result"), dict): + raise ContractError(f"native {method} is unavailable") + return reply["result"] + + account = read_native(2, "account/read", {"refreshToken": False}) + configuration = read_native(3, "config/read", {"includeLayers": False}) + scope = native_scope(account, configuration, binary, version) + catalog = NativeCatalogCache(runtime / "catalogs", scope, version).refresh( + lambda: discover_models(peer, first_request_id=10, timeout_seconds=max(0.001, deadline - time.monotonic())), + ) + models = catalog["models"] + pool_id = native_account_pool(account) + try: + limits = read_native(1000, "account/rateLimits/read") + except (ContractError, EOFError, OSError, TimeoutError): + # A failed query is not evidence that a still-fresh exhausted + # window disappeared. Retain prior observations until expiry. + limits = None + service = Service(runtime) + for observation in (normalize_codex_limits(limits, pool_id) if limits is not None else []): + service.capacity_observe(observation) + except (EOFError, OSError, TimeoutError) as exc: + raise ContractError("Codex model discovery did not complete") from exc + finally: + close_probe(process, start_identity=start_identity) + candidates = [ + model for model in models + if model["supported_efforts"] + and (requested_model is None or model["id"] == requested_model) + ] + if not candidates: + qualifier = requested_model or "any model with effort metadata" + raise ContractError(f"Codex model is unavailable: {qualifier}") + if all_models: + identities = [] + for candidate in sorted(candidates, key=lambda item: not item["is_default"]): + efforts = candidate["supported_efforts"] + if requested_effort is not None and requested_effort not in efforts: + continue + family = candidate.get("family") + identities.append({"harness": "codex", "harness_version": version, + "model_id": candidate["id"], "model_family": family if isinstance(family, str) and family else "gpt", + "effort": requested_effort or next((value for value in ("low", "medium") if value in efforts), efforts[0]), + **({"account_pool_id": pool_id, "native_scope": scope, "catalog_fingerprint": candidate["fingerprint"]} if scope is not None else {})}) + return identities + selected = next( + (model for model in candidates if model["is_default"]), + candidates[0], + ) + efforts = selected["supported_efforts"] + if requested_effort is not None: + if requested_effort not in efforts: + raise ContractError( + f"Codex effort {requested_effort!r} is unavailable for " + f"{selected['id']}" + ) + effort = requested_effort + else: + effort = next( + (value for value in ("low", "medium") if value in efforts), + efforts[0], + ) + family = selected.get("family") + return { + "harness": "codex", + "harness_version": version, + "model_id": selected["id"], + "model_family": family if isinstance(family, str) and family else "gpt", + "effort": effort, + **({"account_pool_id": pool_id, "native_scope": scope, "catalog_fingerprint": selected["fingerprint"]} if scope is not None else {}), + } + + +def parse_checks(values: Iterable[str] | None) -> tuple[tuple[str, ...], ...]: + parsed = [] + for value in values or (): + if not isinstance(value, str) or not value.strip(): + raise ContractError("check command must be non-empty") + try: + arguments = tuple(shlex.split(value)) + except ValueError as exc: + raise ContractError("check command has invalid quoting") from exc + if not arguments: + raise ContractError("check command must contain an executable") + parsed.append(arguments) + if len(parsed) > MAX_USER_CHECKS: + raise ContractError(f"at most {MAX_USER_CHECKS} check commands are allowed") + return tuple(parsed) + + +def _relative_path(value: str, label: str) -> str: + if not isinstance(value, str) or not value: + raise ContractError(f"{label} must be non-empty") + path = PurePosixPath(value) + if path.is_absolute() or ".." in path.parts: + raise ContractError(f"{label} must be repository-relative") + normalized = path.as_posix().rstrip("/") + return normalized or "." + + +def _managed_routing( + workflow: str, + codex: dict[str, str], + *, + claude_model: str, + claude_effort: str, + role_bindings: dict[str, Any], + pinned_roles: set[str], +) -> dict[str, Any]: + reviewer = { + "id": "managed-codex-reviewer", + "harness": "codex", + "model_family": codex["model_family"], + "model_id": codex["model_id"], + "effort": {"value": codex["effort"], "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": codex.get("account_pool_id", "codex-subscription"), + "billing_mode": "subscription", + "quality_status": "trial", + "evidence_refs": [ + f"runtime-catalog:{codex['harness_version']}:{codex['model_id']}", + *([f"runtime-catalog-fingerprint:{codex['catalog_fingerprint']}"] if "catalog_fingerprint" in codex else []), + *([f"runtime-native-scope:{codex['native_scope']}"] if "native_scope" in codex else []), + ], + } + trial_profiles = {"reviewer": reviewer} + if workflow == "issue-delivery": + implementer = { + "id": "managed-claude-implementer", + "harness": "claude", + "model_family": next((f"claude-{family}" for family in ("sonnet", "opus", "haiku") if family in claude_model.lower()), "claude"), + "model_id": claude_model, + "effort": {"value": claude_effort, "transport": "native"}, + "required_tools": ["read", "write"], + "permission_policy": "workspace_write", + "account_pool_id": "claude-subscription", + "billing_mode": "subscription", + "quality_status": "trial", + "evidence_refs": ["verified-claude-cli-2.1.220"], + } + trial_profiles = {"implementer": implementer, **trial_profiles} + profiles = [] + bindings = {} + overrides = {} + roles = {} + for role, profile in trial_profiles.items(): + # A catalog/default/pin change must not reuse a concrete profile ID + # with different bytes or overwrite an approved incumbent. + digest = hashlib.sha256(canonical_json(profile).encode()).hexdigest() + profile["id"] += f"-{digest[:16]}" + profiles.append(profile) + alias = NORMAL_ALIASES[role] + approved = role_bindings.get(role) + if approved is not None: + if (not isinstance(approved, dict) or approved.get("alias") != alias + or type(approved.get("version")) is not int or approved["version"] < 1): + raise ContractError("normal role binding identity is invalid") + incumbent = json.loads(canonical_json(approved.get("profile"))) + validate_profile(incumbent) + if (incumbent["harness"] != profile["harness"] + or incumbent["permission_policy"] != profile["permission_policy"] + or incumbent["billing_mode"] != "subscription" + or incumbent["quality_status"] != "proven"): + raise ContractError("normal role binding exceeds the supported role contract") + if role == "reviewer" and "catalog_fingerprint" in codex and ( + incumbent["account_pool_id"] != codex["account_pool_id"] + or f"runtime-catalog-fingerprint:{codex['catalog_fingerprint']}" not in incumbent["evidence_refs"] + or ("native_scope" in codex and f"runtime-native-scope:{codex['native_scope']}" not in incumbent["evidence_refs"]) + ): + raise ContractError("approved reviewer requires requalification for the current native account/config/catalog") + if incumbent["id"] == profile["id"] and incumbent != profile: + raise ContractError("normal role binding conflicts with the trial profile") + if incumbent["id"] != profile["id"]: + profiles.append(incumbent) + bindings[alias] = {"profile_id": incumbent["id"], "version": approved["version"]} + else: + bindings[alias] = {"profile_id": profile["id"], "version": 1} + roles[role] = [{"kind": "alias", "id": alias}] + if role in pinned_roles: + overrides[role] = {"profile_id": profile["id"], "fallback": "none"} + account_pools = { + profile["account_pool_id"]: { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + } + for profile in profiles + } + return { + "profiles": { + "schema_version": 1, + "profiles": profiles, + "bindings": bindings, + }, + "policy": { + "schema_version": 1, + **NORMAL_POLICY, + "roles": roles, + "task_classes": { + "managed-review" if workflow == "branch-review" + else "managed-fix": "trial", + }, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": account_pools, + "experiment_budget": {}, + "decision_helper": {"schema_version": 1, "mode": "off"}, + }, + **({"overrides": overrides} if overrides else {}), + } + + +def _tree_entry(repo: Path, oid: str, relative: str) -> tuple[str, str] | None: + """Return (mode, type) for a path in the exact oid tree, or None if absent.""" + output = _git(repo, "ls-tree", oid, "--", relative) + if not output: + return None + meta, _, _ = output.splitlines()[0].partition("\t") + parts = meta.split() + if len(parts) < 2: + return None + return parts[0], parts[1] + + +def _is_regular_blob(repo: Path, oid: str, relative: str) -> bool: + entry = _tree_entry(repo, oid, relative) + return entry is not None and entry[1] == "blob" and entry[0] != "120000" + + +def _is_tree(repo: Path, oid: str, relative: str) -> bool: + entry = _tree_entry(repo, oid, relative) + return entry is not None and entry[1] == "tree" + + +def _git_entries_z(repo: Path, *arguments: str) -> list[str]: + """Run git and split NUL-delimited output, preserving embedded tabs/newlines.""" + try: + completed = subprocess.run( + ["git", "-C", str(repo), *arguments], + capture_output=True, + timeout=10, + check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise ContractError("cannot inspect the Git project") from exc + if completed.returncode != 0: + raise ContractError(f"Git project check failed: {' '.join(arguments[:2])}") + return [ + entry.decode("utf-8", errors="surrogateescape") + for entry in completed.stdout.split(b"\0") + if entry + ] + + +def _has_python_tests(repo: Path, oid: str, relative: str) -> bool: + """Return True if a real tracked tree at `relative` holds a regular test*.py file.""" + if not _is_tree(repo, oid, relative): + return False + for entry in _git_entries_z(repo, "ls-tree", "-r", "-z", oid, "--", relative): + meta, _, path = entry.partition("\t") + parts = meta.split() + if len(parts) < 2: + continue + mode, object_type = parts[0], parts[1] + if object_type != "blob" or mode == "120000": + continue + name = PurePosixPath(path).name + if name.startswith("test") and name.endswith(".py"): + return True + return False + + +def _detect_tests(repo: Path, oid: str) -> list[tuple[str, ...]]: + detected: list[tuple[str, ...]] = [] + if _is_regular_blob(repo, oid, "test/run.sh"): + detected.append(("bash", "test/run.sh")) + elif _has_python_tests(repo, oid, "tests"): + detected.append(("python3", "-m", "unittest", "discover", "-s", "tests")) + elif _has_python_tests(repo, oid, "test"): + detected.append(("python3", "-m", "unittest", "discover", "-s", "test")) + if ( + _is_tree(repo, oid, "plugin/core/src") + and _is_tree(repo, oid, "test/core") + and _is_regular_blob(repo, oid, "scripts/run-core-tests.py") + ): + detected.append(( + "env", + "PYTHONPATH=plugin/core/src:test/core", + "PYTHONWARNINGS=error::ResourceWarning", + "python3", + "scripts/run-core-tests.py", + )) + if _is_regular_blob(repo, oid, "scripts/generate-core-reference.py"): + detected.append(("python3", "scripts/generate-core-reference.py", "--check")) + return detected + + +def _checks( + repo: Path, + *, + workflow: str, + base_oid: str, + target_oid: str, + supplied: tuple[tuple[str, ...], ...], + timeout_seconds: int, +) -> list[dict[str, Any]]: + required = workflow == "issue-delivery" + checks: list[dict[str, Any]] = [{ + "id": "candidate-diff-check", + "argv": ( + ["git", "diff", "--check", base_oid, target_oid, "--"] + if workflow == "branch-review" + else ["git", "diff", "--check", base_oid, "HEAD", "--"] + ), + "cwd": ".", + "timeout_seconds": min(timeout_seconds, 120), + "required_to_pass": required, + }] + seen: set[tuple[str, ...]] = {tuple(checks[0]["argv"])} + for argv in _detect_tests(repo, target_oid): + check_id = ( + "generated-core-reference" if argv == ("python3", "scripts/generate-core-reference.py", "--check") + else "detected-core-tests" if argv[-1] == "scripts/run-core-tests.py" + else "detected-tests" + ) + checks.append({ + "id": check_id, + "argv": list(argv), + "cwd": ".", + "timeout_seconds": timeout_seconds, + "required_to_pass": required, + }) + seen.add(argv) + index = 0 + for arguments in supplied: + if arguments in seen: + continue + index += 1 + checks.append({ + "id": f"user-check-{index}", + "argv": list(arguments), + "cwd": ".", + "timeout_seconds": timeout_seconds, + "required_to_pass": required, + }) + seen.add(arguments) + if len(checks) > 16: + raise ContractError("normal entry produced too many checks") + return checks + + +def build_managed_task( + *, + workflow: str, + project_dir: str | Path, + base_ref: str, + target_ref: str, + goal: str, + codex_identity: dict[str, str], + write_paths: Iterable[str] = (), + checks: tuple[tuple[str, ...], ...] = (), + check_timeout: int = 600, + review_mode: str = "standard", + review_focus: str | None = None, + claude_model: str = "sonnet", + claude_effort: str = "high", + role_bindings: dict[str, Any] | None = None, + pinned_roles: Iterable[str] = (), +) -> tuple[dict[str, Any], dict[str, Any]]: + if workflow not in {"branch-review", "issue-delivery"}: + raise ContractError("normal entry workflow is unsupported") + if role_bindings is not None and not isinstance(role_bindings, dict): + raise ContractError("normal role bindings must be an object") + pins = set(pinned_roles) + if not isinstance(goal, str) or not goal.strip(): + raise ContractError("goal must be non-empty") + if type(check_timeout) is not int or not 1 <= check_timeout <= 3600: + raise ContractError("check timeout must be between 1 and 3600 seconds") + if review_mode not in {"standard", "adversarial"}: + raise ContractError("review mode must be standard or adversarial") + if review_focus is not None: + if review_mode != "adversarial": + raise ContractError("review focus requires adversarial mode") + if not isinstance(review_focus, str) or not review_focus.strip(): + raise ContractError("review focus must be non-empty") + required_identity = { + "harness", "harness_version", "model_id", "model_family", "effort", + } + if (not isinstance(codex_identity, dict) + or set(codex_identity) - (required_identity | {"account_pool_id", "native_scope", "catalog_fingerprint"}) + or required_identity - set(codex_identity) + or codex_identity.get("harness") != "codex" + or not all( + isinstance(codex_identity[field], str) + and codex_identity[field] + for field in required_identity - {"harness"} + )): + raise ContractError("Codex reviewer identity is invalid") + if workflow == "issue-delivery" and ( + not isinstance(claude_model, str) or not claude_model + or not isinstance(claude_effort, str) or not claude_effort + ): + raise ContractError("Claude model and effort must be non-empty") + repo = resolve_repository(project_dir) + base_oid = _resolve_commit(repo, base_ref) + target_oid = _resolve_commit(repo, target_ref) + normalized_writes = tuple( + dict.fromkeys(_relative_path(path, "write path") for path in write_paths) + ) + if workflow == "branch-review" and normalized_writes: + raise ContractError("review entry cannot declare write paths") + if workflow == "issue-delivery" and not normalized_writes: + normalized_writes = (".",) + routing = _managed_routing( + workflow, + codex_identity, + claude_model=claude_model, + claude_effort=claude_effort, + role_bindings={} if role_bindings is None else role_bindings, + pinned_roles=pins, + ) + if pins - set(routing["policy"]["roles"]): + raise ContractError("normal pin names an unsupported role") + task_class = "managed-review" if workflow == "branch-review" else "managed-fix" + task: dict[str, Any] = { + "schema_version": 1, + "project": { + "repo_path": str(repo), + "base_ref": base_oid, + "target_ref": target_oid, + }, + "workflow": workflow, + "goal": goal.strip(), + "task_class": task_class, + "acceptance": [ + { + "id": "bounded-goal", + "description": ( + "Review findings are bound to the exact base and target commits." + if workflow == "branch-review" + else f"The candidate addresses this bounded issue: {goal.strip()}" + ), + "evidence_kind": "review", + }, + { + "id": "declared-checks", + "description": "Every declared check result is retained in the receipt.", + "evidence_kind": "check", + }, + ], + "checks": _checks( + repo, + workflow=workflow, + base_oid=base_oid, + target_oid=target_oid, + supplied=checks, + timeout_seconds=check_timeout, + ), + "scope": { + "read_paths": ["."], + "write_paths": list(normalized_writes), + }, + "lead": {"mode": "host"}, + "routing": routing, + "budget": { + "wall_seconds": 900 if workflow == "branch-review" else 1800, + "max_worker_invocations": 1 if workflow == "branch-review" else 7, + "max_revisions": 0 if workflow == "branch-review" else 2, + "max_fallbacks_per_step": 0, + }, + "origin": {"surface": "cli-normal-entry"}, + "review": {"mode": review_mode}, + } + if review_focus is not None: + task["review"]["focus"] = review_focus.strip() + validate_task(task, require_existing_repo=True) + task_sha256 = hashlib.sha256(canonical_json(task).encode()).hexdigest() + profiles_by_id = {p["id"]: p for p in routing["profiles"]["profiles"]} + roles = {} + for role in routing["policy"]["roles"]: + alias = NORMAL_ALIASES[role] + override = routing.get("overrides", {}).get(role) + profile = profiles_by_id[(override or routing["profiles"]["bindings"][alias])["profile_id"]] + roles[role] = { + "profile_id": profile["id"], + "harness": profile["harness"], + "model_id": profile["model_id"], + "effort": profile["effort"]["value"], + "permission": profile["permission_policy"], + "alias": alias, + "selection_mode": "pinned" if override else "approved_alias" if profile["quality_status"] == "proven" else "bounded_trial", + } + return task, { + "workflow": workflow, + "task_sha256": task_sha256, + "project": str(repo), + "base_oid": base_oid, + "target_oid": target_oid, + "planned_roles": roles, + "selection_reason": ( + "stable policy aliases; explicit pins are fixed, approved incumbents " + "are retained, otherwise a bounded trial is used; " + "different-harness review is mandatory for delivery" + ), + "scope": task["scope"], + "checks": [check["id"] for check in task["checks"]], + "check_plan": task["checks"], + } diff --git a/plugin/core/src/devsquad/validation.py b/plugin/core/src/devsquad/validation.py new file mode 100644 index 0000000..0b3554f --- /dev/null +++ b/plugin/core/src/devsquad/validation.py @@ -0,0 +1,236 @@ +"""Dependency-free strict validation for M1 public fixtures.""" + +from __future__ import annotations +from pathlib import Path +from typing import Any +from .contracts import ContractError + +TASK_FIELDS = {"schema_version", "project", "workflow", "goal", "task_class", "acceptance", "checks", "scope", "lead", "routing", "budget", "origin", "review", "council"} +MAX_ACCEPTANCE_CRITERIA = 100 +MAX_CHECKS = 16 +MAX_SCOPE_PATHS = 256 + + +def _exact(value: dict[str, Any], allowed: set[str], required: set[str], label: str) -> None: + if not isinstance(value, dict): + raise ContractError(f"{label} must be an object") + unknown, missing = set(value) - allowed, required - set(value) + if unknown or missing: + raise ContractError(f"{label} fields invalid: unknown={sorted(unknown)} missing={sorted(missing)}") + + +def _relative(path: str, label: str) -> None: + if not isinstance(path, str) or not path: + raise ContractError(f"{label} must be a non-empty string") + p = Path(path) + if p.is_absolute() or ".." in p.parts: + raise ContractError(f"{label} must be repository-relative without traversal") + + +def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) -> None: + _exact(value, TASK_FIELDS, TASK_FIELDS - {"review", "council"}, "task") + if type(value["schema_version"]) is not int or value["schema_version"] != 1 or not isinstance(value["workflow"], str) or value["workflow"] not in {"branch-review", "issue-delivery", "council-decision"}: + raise ContractError("unsupported task schema or workflow") + if value["workflow"] == "council-decision": + from .council import validate_spec + validate_spec(value.get("council")) + elif "council" in value: + raise ContractError("CouncilSpec requires the council-decision workflow") + project = value["project"] + _exact(project, {"repo_path", "base_ref", "target_ref"}, {"repo_path", "base_ref", "target_ref"}, "project") + if not all(isinstance(project[k], str) and project[k] for k in ("repo_path", "base_ref", "target_ref")) or not Path(project["repo_path"]).is_absolute() or (require_existing_repo and not (Path(project["repo_path"]) / ".git").exists()): + raise ContractError("project.repo_path must be an existing absolute Git repository") + if not isinstance(value["goal"], str) or not value["goal"].strip() or not isinstance(value["task_class"], str) or not value["task_class"].strip(): + raise ContractError("goal and task_class must be non-empty strings") + if (not isinstance(value["acceptance"], list) or not value["acceptance"] + or len(value["acceptance"]) > MAX_ACCEPTANCE_CRITERIA): + raise ContractError("acceptance must be a non-empty bounded array") + acceptance_ids = set() + for item in value["acceptance"]: + _exact(item, {"id", "description", "evidence_kind"}, {"id", "description", "evidence_kind"}, "acceptance item") + if not all(isinstance(item[k], str) and item[k].strip() for k in ("id", "description")): + raise ContractError("acceptance id and description must be non-empty strings") + if not isinstance(item["evidence_kind"], str) or item["evidence_kind"] not in {"review", "check", "artifact", "host"}: + raise ContractError("invalid evidence_kind") + if item["id"] in acceptance_ids: + raise ContractError("acceptance ids must be unique") + acceptance_ids.add(item["id"]) + if not isinstance(value["checks"], list) or len(value["checks"]) > MAX_CHECKS: + raise ContractError("checks must be a bounded array") + check_ids = set() + for check in value["checks"]: + _exact(check, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass", "output_paths"}, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, "check") + if not isinstance(check["id"], str) or not check["id"]: raise ContractError("check id must be non-empty") + if check["id"] in check_ids: raise ContractError("check ids must be unique") + check_ids.add(check["id"]) + if not isinstance(check["argv"], list) or not check["argv"] or not all(isinstance(v, str) and v for v in check["argv"]): raise ContractError("check argv must be a non-empty string array") + _relative(check["cwd"], "check cwd") + if not isinstance(check["timeout_seconds"], int) or isinstance(check["timeout_seconds"], bool) or check["timeout_seconds"] <= 0: raise ContractError("check timeout must be positive") + if type(check["required_to_pass"]) is not bool: raise ContractError("required_to_pass must be boolean") + outputs = check.get("output_paths", []) + if not isinstance(outputs, list) or len(outputs) > 32: + raise ContractError("check output_paths must be a bounded array") + for output in outputs: + _relative(output, "check output path") + if Path(output).as_posix() != output or output == "." or ".git" in Path(output).parts: + raise ContractError("check output path must be canonical and exclude Git metadata/root") + if len(set(outputs)) != len(outputs): + raise ContractError("check output paths must be unique") + scope = value["scope"]; _exact(scope, {"read_paths", "write_paths"}, {"read_paths", "write_paths"}, "scope") + if not isinstance(scope["read_paths"], list) or not isinstance(scope["write_paths"], list) or not all(isinstance(p, str) and p for p in scope["read_paths"] + scope["write_paths"]): raise ContractError("scope paths must be non-empty string arrays") + if len(scope["read_paths"]) + len(scope["write_paths"]) > MAX_SCOPE_PATHS: raise ContractError("scope paths exceed their bound") + if len(set(scope["read_paths"])) != len(scope["read_paths"]) or len(set(scope["write_paths"])) != len(scope["write_paths"]): raise ContractError("scope paths must be unique") + for p in scope["read_paths"] + scope["write_paths"]: _relative(p, "scope path") + if value["workflow"] in {"branch-review", "council-decision"} and scope["write_paths"]: raise ContractError("read-only workflow cannot write") + lead = value["lead"]; _exact(lead, {"mode"}, {"mode"}, "lead") + if not isinstance(lead["mode"], str) or lead["mode"] not in {"host", "headless"}: raise ContractError("invalid lead mode") + routing = value["routing"] + _exact( + routing, + {"profiles_file", "policy_file", "profiles", "policy", "overrides"}, + set(), + "routing", + ) + file_fields = {"profiles_file", "policy_file"} + embedded_fields = {"profiles", "policy"} + if set(routing) & file_fields == file_fields and not set(routing) & embedded_fields: + for key in file_fields: + if not isinstance(routing[key], str) or not routing[key]: + raise ContractError(f"routing {key} must be a path") + elif set(routing) & embedded_fields == embedded_fields and not set(routing) & file_fields: + validate_profile_registry(routing["profiles"]) + validate_policy(routing["policy"]) + else: + raise ContractError( + "routing requires exactly one complete file or embedded configuration" + ) + overrides = routing.get("overrides", {}) + if not isinstance(overrides, dict): raise ContractError("routing overrides must be an object") + for role, override in overrides.items(): + if role not in {"implementer", "reviewer", "lead", "researcher", "proposer_a", "proposer_b", "critic"}: raise ContractError("invalid override role") + _exact(override, {"profile_id", "fallback"}, {"profile_id"}, "routing override") + if not isinstance(override["profile_id"], str) or not override["profile_id"]: raise ContractError("override profile_id must be non-empty") + if not isinstance(override.get("fallback", "none"), str) or override.get("fallback", "none") not in {"none", "policy"}: raise ContractError("override fallback must be none or policy") + origin = value["origin"]; _exact(origin, {"surface", "session_ref"}, {"surface"}, "origin") + if not isinstance(origin["surface"], str) or not origin["surface"]: raise ContractError("origin surface must be non-empty") + if "session_ref" in origin and (not isinstance(origin["session_ref"], str) or not origin["session_ref"]): raise ContractError("origin session_ref must be non-empty") + if "review" in value: + review = value["review"]; _exact(review, {"mode", "focus"}, {"mode"}, "review") + if not isinstance(review["mode"], str) or review["mode"] not in {"standard", "adversarial"}: raise ContractError("invalid review mode") + if "focus" in review and review["mode"] != "adversarial": raise ContractError("review focus requires adversarial mode") + if "focus" in review and (not isinstance(review["focus"], str) or not review["focus"]): raise ContractError("review focus must be non-empty") + budget = value["budget"]; required = {"wall_seconds", "max_worker_invocations", "max_revisions", "max_fallbacks_per_step"}; _exact(budget, required, required, "budget") + for key, number in budget.items(): + if not isinstance(number, int) or isinstance(number, bool) or number < 0: raise ContractError(f"budget {key} must be a finite non-negative integer") + if budget["wall_seconds"] == 0 or budget["max_worker_invocations"] == 0: raise ContractError("wall_seconds and max_worker_invocations must be positive") + if value["workflow"] == "council-decision": + minimum = 4 if lead["mode"] == "headless" else 3 + if budget["max_worker_invocations"] < minimum or budget["max_worker_invocations"] > value["council"]["max_invocations"]: + raise ContractError("Council worker budget must fit the explicit Council invocation cap") + if budget["max_revisions"] != 0: + raise ContractError("bounded Council supports one round; additional rounds need a new capped run") + if "review" in value: + raise ContractError("Council uses its frozen rubric, not branch-review options") + + +def validate_profile(value: dict[str, Any]) -> None: + fields = {"id", "harness", "model_family", "model_id", "effort", "required_tools", "permission_policy", "account_pool_id", "billing_mode", "quality_status", "evidence_refs"} + _exact(value, fields, fields, "profile") + for key in ("id", "harness", "model_family", "model_id", "account_pool_id"): + if not isinstance(value[key], str) or not value[key]: raise ContractError(f"profile {key} must be non-empty") + effort = value["effort"]; _exact(effort, {"value", "transport"}, {"value", "transport"}, "profile effort") + if effort["value"] is not None and not isinstance(effort["value"], str): raise ContractError("effort value must be string or null") + if not isinstance(effort["transport"], str) or effort["transport"] not in {"native", "model_variant", "provider_default"}: raise ContractError("invalid effort transport") + for key in ("required_tools", "evidence_refs"): + if not isinstance(value[key], list) or not all(isinstance(v, str) and v for v in value[key]) or len(set(value[key])) != len(value[key]): raise ContractError(f"profile {key} must contain unique non-empty strings") + if not isinstance(value["permission_policy"], str) or value["permission_policy"] not in {"read_only", "workspace_write"}: raise ContractError("invalid permission policy") + if not isinstance(value["billing_mode"], str) or value["billing_mode"] not in {"subscription", "paid_api"}: raise ContractError("invalid billing mode") + if not isinstance(value["quality_status"], str) or value["quality_status"] not in {"unvalidated", "trial", "proven", "suspended"}: raise ContractError("invalid quality status") + + +def validate_profile_registry(value: dict[str, Any]) -> None: + fields = {"schema_version", "profiles", "bindings"} + _exact(value, fields, fields, "profile registry") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("invalid profile registry schema version") + profiles = value["profiles"] + if not isinstance(profiles, list) or not profiles: + raise ContractError("profile registry profiles must be a non-empty array") + identifiers = [] + for profile in profiles: + validate_profile(profile) + identifiers.append(profile["id"]) + if len(set(identifiers)) != len(identifiers): + raise ContractError("profile registry profile ids must be unique") + bindings = value["bindings"] + if not isinstance(bindings, dict): + raise ContractError("profile registry bindings must be an object") + for alias, binding in bindings.items(): + if not isinstance(alias, str) or not alias: + raise ContractError("profile binding aliases must be non-empty strings") + _exact(binding, {"profile_id", "version"}, {"profile_id", "version"}, "profile binding") + if not isinstance(binding["profile_id"], str) or not binding["profile_id"]: + raise ContractError("profile binding profile_id must be non-empty") + if type(binding["version"]) is not int or binding["version"] < 1: + raise ContractError("profile binding version must be positive") + if binding["profile_id"] not in identifiers: + raise ContractError(f"profile binding target does not exist: {binding['profile_id']}") + + +def validate_policy(value: dict[str, Any]) -> None: + fields = {"schema_version", "id", "version", "roles", "task_classes", "require_different_model_for_review", "prefer_different_harness_for_review", "account_pools", "experiment_budget", "decision_helper"} + required = fields - {"prefer_different_harness_for_review", "decision_helper"} + _exact(value, fields, required, "policy") + if type(value["schema_version"]) is not int or value["schema_version"] != 1 or type(value["version"]) is not int or value["version"] < 1: raise ContractError("invalid policy version") + if not isinstance(value["id"], str) or not value["id"]: raise ContractError("policy id must be non-empty") + if type(value["require_different_model_for_review"]) is not bool or ("prefer_different_harness_for_review" in value and type(value["prefer_different_harness_for_review"]) is not bool): raise ContractError("policy review flags must be boolean") + if not isinstance(value["roles"], dict) or set(value["roles"]) - {"implementer", "reviewer", "lead", "researcher", "proposer_a", "proposer_b", "critic"}: raise ContractError("invalid policy roles") + for candidates in value["roles"].values(): + if not isinstance(candidates, list) or not candidates: raise ContractError("role candidates must be non-empty arrays") + for ref in candidates: + _exact(ref, {"kind", "id"}, {"kind", "id"}, "candidate reference") + if not isinstance(ref["kind"], str) or ref["kind"] not in {"profile", "alias"} or not isinstance(ref["id"], str) or not ref["id"]: raise ContractError("invalid candidate reference") + for key in ("task_classes", "account_pools", "experiment_budget"): + if not isinstance(value[key], dict): raise ContractError(f"policy {key} must be an object") + _validate_json_tree(value[key], f"policy {key}") + if not all(isinstance(k, str) and k and isinstance(v, str) and v in {"unvalidated", "trial", "proven", "suspended"} for k, v in value["task_classes"].items()): + raise ContractError("task_classes must map names to quality status") + if not all(isinstance(k, str) and k and isinstance(v, dict) for k, v in value["account_pools"].items()): + raise ContractError("account_pools must map names to objects") + for pool in value["account_pools"].values(): + _exact( + pool, + {"allowed_billing_modes", "max_concurrency", "unknown_capacity_policy"}, + {"allowed_billing_modes", "max_concurrency"}, + "account pool policy", + ) + modes = pool["allowed_billing_modes"] + if (not isinstance(modes, list) or not modes + or len(set(modes)) != len(modes) + or not all(isinstance(mode, str) and mode in {"subscription", "paid_api"} + for mode in modes)): + raise ContractError("account pool billing modes are invalid") + if type(pool["max_concurrency"]) is not int or pool["max_concurrency"] < 1: + raise ContractError("account pool max_concurrency must be positive") + if pool.get("unknown_capacity_policy", "allow_bounded") not in {"allow_bounded", "block"}: + raise ContractError("account pool unknown_capacity_policy is invalid") + if not all(isinstance(k, str) and k and type(v) is int and v >= 0 for k, v in value["experiment_budget"].items()): + raise ContractError("experiment_budget must contain non-negative integers") + from .decision import validate_decision_policy + validate_decision_policy(value.get("decision_helper")) + + +def _validate_json_tree(value: Any, label: str) -> None: + """Reject non-JSON and numerically ambiguous values in extension maps.""" + if value is None or isinstance(value, str) or type(value) is bool: + return + if type(value) is int: + return + if isinstance(value, list): + for item in value: _validate_json_tree(item, label) + return + if isinstance(value, dict): + if not all(isinstance(k, str) and k for k in value): raise ContractError(f"{label} keys must be non-empty strings") + for item in value.values(): _validate_json_tree(item, label) + return + raise ContractError(f"{label} contains a non-JSON or non-finite value") diff --git a/plugin/core/src/devsquad/worker_gate.py b/plugin/core/src/devsquad/worker_gate.py new file mode 100644 index 0000000..7b24896 --- /dev/null +++ b/plugin/core/src/devsquad/worker_gate.py @@ -0,0 +1,11 @@ +"""Execute an internal worker only after its supervisor persists ownership.""" +import argparse, os + +def main(argv=None): + parser=argparse.ArgumentParser(); parser.add_argument("--gate-fd",type=int,required=True); parser.add_argument("command",nargs=argparse.REMAINDER) + args=parser.parse_args(argv); command=args.command[1:] if args.command[:1]==["--"] else args.command + with os.fdopen(args.gate_fd,"rb",closefd=True) as gate: + if gate.read(1)!=b"1": return 125 + os.execvpe(command[0],command,os.environ) + +if __name__=="__main__": raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py new file mode 100644 index 0000000..57f395e --- /dev/null +++ b/plugin/core/src/devsquad/workflows.py @@ -0,0 +1,1215 @@ +"""Strict, side-effect-free contracts for the two fixed engineering workflows.""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import PurePosixPath +import re +from typing import Any + +from .contracts import ContractError +from .claude_identity import validate_observation +from .store import canonical_json +from .validation import validate_task + + +MAX_REVIEW_BYTES = 512 * 1024 +MAX_LEAD_BYTES = 128 * 1024 +MAX_EVIDENCE_BYTES = 1024 * 1024 +MAX_FINDINGS = 100 +MAX_TEXT_CHARS = 20_000 +MAX_PREVIEW_CHARS = 1024 +FINDING_SEVERITIES = {"critical", "high", "medium", "low"} +CHECK_STATUSES = {"passed", "failed", "timed_out", "launch_failed", "invalidated", "not_run"} +REVIEW_MODES = {"standard", "adversarial"} +_SHA256 = re.compile(r"[0-9a-f]{64}\Z") +_COMMIT_OID = re.compile(r"[0-9a-f]{40}\Z") + + +def review_output_schema() -> dict[str, Any]: + """Return the strict native structured-output schema for one review.""" + return { + "type": "object", + "additionalProperties": False, + "required": [ + "schema_version", "candidate_sha256", "base_oid", "target_oid", + "review_mode", "verdict", "summary", "findings", + ], + "properties": { + "schema_version": {"type": "integer", "enum": [1]}, + "candidate_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "base_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, + "target_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, + "review_mode": { + "type": "string", "enum": ["standard", "adversarial"], + }, + "verdict": {"type": "string", "enum": ["clean", "findings"]}, + "summary": {"type": "string"}, + "findings": { + "type": "array", + "maxItems": MAX_FINDINGS, + "items": { + "type": "object", + "additionalProperties": False, + "required": [ + "id", "severity", "title", "description", "path", + "start_line", "end_line", "evidence", + ], + "properties": { + "id": {"type": "string"}, + "severity": { + "type": "string", "enum": sorted(FINDING_SEVERITIES), + }, + "title": {"type": "string"}, + "description": {"type": "string"}, + "path": {"type": "string"}, + "start_line": {"type": "integer", "minimum": 1}, + "end_line": {"type": "integer", "minimum": 1}, + "evidence": {"type": "string"}, + }, + }, + }, + }, + } + + +def lead_output_schema() -> dict[str, Any]: + """Return the strict native structured-output schema for one lead choice.""" + return { + "type": "object", + "additionalProperties": False, + "required": [ + "schema_version", "candidate_sha256", "disposition", "reason", + ], + "properties": { + "schema_version": {"type": "integer", "enum": [1]}, + "candidate_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "disposition": { + "type": "string", "enum": ["accept", "revise", "reject"], + }, + "reason": {"type": "string"}, + }, + } + + +def _exact( + value: Any, + fields: set[str], + label: str, +) -> dict[str, Any]: + if not isinstance(value, dict): + raise ContractError(f"{label} must be an object") + unknown, missing = set(value) - fields, fields - set(value) + if unknown or missing: + raise ContractError( + f"{label} fields invalid: unknown={sorted(unknown)} " + f"missing={sorted(missing)}" + ) + return value + + +def _text(value: Any, label: str, *, maximum: int = MAX_TEXT_CHARS) -> str: + if not isinstance(value, str) or not value.strip(): + raise ContractError(f"{label} must be a non-empty string") + if len(value) > maximum: + raise ContractError(f"{label} exceeds its size limit") + return value + + +def _sha256(value: Any, label: str) -> str: + if not isinstance(value, str) or _SHA256.fullmatch(value) is None: + raise ContractError(f"{label} must be a lowercase SHA-256") + return value + + +def _commit_oid(value: Any, label: str) -> str: + if not isinstance(value, str) or _COMMIT_OID.fullmatch(value) is None: + raise ContractError(f"{label} must be a full lowercase commit OID") + return value + + +def _relative_path(value: Any, label: str) -> str: + if (not isinstance(value, str) or not value or "\\" in value + or "\0" in value): + raise ContractError(f"{label} must be a canonical repository-relative path") + path = PurePosixPath(value) + normalized = path.as_posix() + if (path.is_absolute() or ".." in path.parts or normalized in {"", "."} + or normalized != value): + raise ContractError(f"{label} must be a canonical repository-relative path") + return normalized + + +def _inside_scope(path: str, scopes: list[str]) -> bool: + for raw_scope in scopes: + scope = PurePosixPath(raw_scope).as_posix().rstrip("/") or "." + if scope == "." or path == scope or path.startswith(f"{scope}/"): + return True + return False + + +def _strict_json_object( + payload: bytes | str, + label: str, + *, + maximum: int = MAX_REVIEW_BYTES, +) -> dict[str, Any]: + if isinstance(payload, str): + encoded = payload.encode() + elif isinstance(payload, bytes): + encoded = payload + else: + raise ContractError(f"{label} must be UTF-8 JSON bytes or text") + if len(encoded) > maximum: + raise ContractError(f"{label} exceeds its byte limit") + + def object_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result = {} + for key, value in pairs: + if key in result: + raise ContractError(f"{label} contains duplicate key: {key}") + result[key] = value + return result + + def reject_constant(value: str) -> None: + raise ContractError(f"{label} contains non-finite number: {value}") + + try: + value = json.loads( + encoded.decode("utf-8"), + object_pairs_hook=object_pairs, + parse_constant=reject_constant, + ) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError(f"{label} is not valid UTF-8 JSON") from exc + if not isinstance(value, dict): + raise ContractError(f"{label} must contain one JSON object") + return value + + +def _workspace_identity(workspace: dict[str, Any]) -> tuple[str, str, str]: + if not isinstance(workspace, dict): + raise ContractError("workspace snapshot must be an object") + return ( + _sha256(workspace.get("candidate_sha256"), "workspace candidate_sha256"), + _commit_oid(workspace.get("base_oid"), "workspace base_oid"), + _commit_oid(workspace.get("target_oid"), "workspace target_oid"), + ) + + +def review_mode(task: dict[str, Any]) -> str: + mode = task.get("review", {}).get("mode", "standard") + if mode not in REVIEW_MODES: + raise ContractError("review mode is invalid") + return mode + + +def validate_review_document( + value: dict[str, Any], + task: dict[str, Any], + workspace: dict[str, Any], +) -> dict[str, Any]: + """Validate model output and bind it to the exact frozen candidate.""" + validate_task(task) + if task["workflow"] not in {"branch-review", "issue-delivery"}: + raise ContractError("review document requires a reviewable workflow") + document = _exact(value, { + "schema_version", "candidate_sha256", "base_oid", "target_oid", + "review_mode", "verdict", "summary", "findings", + }, "review document") + if document["schema_version"] != 1 or type(document["schema_version"]) is not int: + raise ContractError("review document schema_version is invalid") + candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) + if _sha256(document["candidate_sha256"], "review candidate_sha256") != candidate_sha256: + raise ContractError("review targets a different candidate hash") + if _commit_oid(document["base_oid"], "review base_oid") != base_oid: + raise ContractError("review targets a different base commit") + if _commit_oid(document["target_oid"], "review target_oid") != target_oid: + raise ContractError("review targets a different target commit") + if document["review_mode"] != review_mode(task): + raise ContractError("review mode does not match the frozen task") + if document["verdict"] not in {"clean", "findings"}: + raise ContractError("review verdict must be clean or findings") + _text(document["summary"], "review summary") + findings = document["findings"] + if not isinstance(findings, list) or len(findings) > MAX_FINDINGS: + raise ContractError("review findings must be a bounded array") + seen = set() + for finding in findings: + item = _exact(finding, { + "id", "severity", "title", "description", "path", + "start_line", "end_line", "evidence", + }, "review finding") + finding_id = _text(item["id"], "finding id", maximum=200) + if finding_id in seen: + raise ContractError(f"review finding id is duplicated: {finding_id}") + seen.add(finding_id) + if item["severity"] not in FINDING_SEVERITIES: + raise ContractError("review finding severity is invalid") + _text(item["title"], "finding title", maximum=500) + _text(item["description"], "finding description") + finding_path = _relative_path(item["path"], "finding path") + if not _inside_scope(finding_path, task["scope"]["read_paths"]): + raise ContractError("review finding path is outside the declared read scope") + if type(item["start_line"]) is not int or item["start_line"] < 1: + raise ContractError("finding start_line must be a positive integer") + if type(item["end_line"]) is not int or item["end_line"] < item["start_line"]: + raise ContractError("finding end_line must not precede start_line") + _text(item["evidence"], "finding evidence") + if (document["verdict"] == "clean") != (not findings): + raise ContractError("review verdict and findings disagree") + return json.loads(canonical_json(document)) + + +def decode_review_document( + payload: bytes | str, + task: dict[str, Any], + workspace: dict[str, Any], +) -> dict[str, Any]: + return validate_review_document( + _strict_json_object(payload, "review output"), task, workspace, + ) + + +def _validate_stream(value: Any, label: str) -> None: + stream = _exact(value, { + "preview", "captured_bytes", "total_bytes", "truncated", "full_sha256", + }, label) + if not isinstance(stream["preview"], str) or len(stream["preview"]) > MAX_PREVIEW_CHARS: + raise ContractError(f"{label} preview exceeds its size limit") + for field in ("captured_bytes", "total_bytes"): + if type(stream[field]) is not int or stream[field] < 0: + raise ContractError(f"{label} {field} must be a non-negative integer") + if stream["captured_bytes"] > stream["total_bytes"]: + raise ContractError(f"{label} captured bytes exceed total bytes") + if type(stream["truncated"]) is not bool: + raise ContractError(f"{label} truncated must be boolean") + if stream["truncated"] != (stream["captured_bytes"] < stream["total_bytes"]): + raise ContractError(f"{label} truncation metadata is inconsistent") + _sha256(stream["full_sha256"], f"{label} full_sha256") + + +def validate_check_results( + values: list[dict[str, Any]], + task: dict[str, Any], + workspace: dict[str, Any], +) -> list[dict[str, Any]]: + """Validate check evidence against the host-supplied immutable check plan.""" + validate_task(task) + if not isinstance(values, list) or len(values) != len(task["checks"]): + raise ContractError("check results do not match the declared check count") + candidate_sha256, _, target_oid = _workspace_identity(workspace) + normalized = [] + for configured, supplied in zip(task["checks"], values): + version = supplied.get("schema_version") if isinstance(supplied, dict) else None + fields = { + "schema_version", "candidate_sha256", "target_oid", "id", "argv", + "cwd", "required_to_pass", "status", "returncode", "error_code", + "duration_ms", "stdout", "stderr", + } + if version == 2: + fields.update({"integrity", "output_paths"}) + result = _exact(supplied, fields, "check result") + if type(version) is not int or version not in {1, 2}: + raise ContractError("check result schema_version is invalid") + if _sha256(result["candidate_sha256"], "check candidate_sha256") != candidate_sha256: + raise ContractError("check result targets a different candidate hash") + if _commit_oid(result["target_oid"], "check target_oid") != target_oid: + raise ContractError("check result targets a different target commit") + for field in ("id", "argv", "cwd", "required_to_pass"): + if result[field] != configured[field]: + raise ContractError(f"check result changes declared field: {field}") + if not isinstance(result["status"], str) or result["status"] not in CHECK_STATUSES: + raise ContractError("check result status is invalid") + if type(result["duration_ms"]) is not int or result["duration_ms"] < 0: + raise ContractError("check duration_ms must be a non-negative integer") + status, returncode, error_code = ( + result["status"], result["returncode"], result["error_code"] + ) + if version == 2: + if result["output_paths"] != configured.get("output_paths", []): + raise ContractError("check result changes declared field: output_paths") + integrity = _exact(result["integrity"], { + "status", "reasons", "before_state_sha256", "after_state_sha256", + "changes", "changes_truncated", + }, "check integrity") + expected = {"invalidated": "violated", "not_run": "not_run"}.get(status, "verified") + if integrity["status"] != expected: + raise ContractError("check integrity contradicts its status") + reasons = integrity["reasons"] + if (not isinstance(reasons, list) or len(reasons) > 20 + or any(not isinstance(reason, str) or not reason or len(reason) > 200 + for reason in reasons) + or (expected == "verified") != (reasons == [])): + raise ContractError("check integrity reasons are inconsistent") + for field in ("before_state_sha256", "after_state_sha256"): + if status == "not_run": + if integrity[field] is not None: + raise ContractError("skipped check cannot claim an integrity observation") + else: + _sha256(integrity[field], f"check integrity {field}") + changes = integrity["changes"] + if (not isinstance(changes, list) or len(changes) > 100 + or type(integrity["changes_truncated"]) is not bool): + raise ContractError("check integrity changes must be bounded") + if status != "invalidated" and (changes or integrity["changes_truncated"]): + raise ContractError("clean/skipped check cannot report input changes") + if expected == "verified" and integrity["before_state_sha256"] != integrity["after_state_sha256"]: + raise ContractError("verified check has differing input fingerprints") + for change in changes: + _exact(change, {"workspace", "path", "before", "after"}, "check integrity change") + if not isinstance(change["workspace"], str) or change["workspace"] not in {"check", "review"}: + raise ContractError("check integrity change workspace is invalid") + path = _text(change["path"], "check integrity path", maximum=4096) + if PurePosixPath(path).is_absolute() or ".." in PurePosixPath(path).parts: + raise ContractError("check integrity path escapes workspace") + for field in ("before", "after"): + if change[field] is not None and change[field] != "undeclared": + _sha256(change[field], f"check change {field}") + if change["before"] == change["after"]: + raise ContractError("check integrity change has identical evidence") + elif status in {"invalidated", "not_run"}: + raise ContractError("check integrity requires schema version 2") + if status in {"invalidated", "not_run"} and ( + error_code != "CLI_ERROR" + or (returncode is not None and type(returncode) is not int) + or (status == "not_run" and returncode is not None)): + raise ContractError("invalidated/skipped check result is inconsistent") + if status == "passed" and (type(returncode) is not int or returncode != 0 or error_code is not None): + raise ContractError("passing check result is inconsistent") + if status == "failed" and ( + type(returncode) is not int or returncode == 0 or error_code is not None + ): + raise ContractError("failed check result is inconsistent") + if status == "timed_out" and error_code != "TIMEOUT": + raise ContractError("timed-out check result is inconsistent") + if status == "launch_failed" and ( + returncode is not None or error_code != "CLI_ERROR" + ): + raise ContractError("launch-failed check result is inconsistent") + if status in {"timed_out", "launch_failed"} and returncode is not None: + raise ContractError("incomplete check cannot report a return code") + _validate_stream(result["stdout"], "check stdout") + _validate_stream(result["stderr"], "check stderr") + normalized.append(json.loads(canonical_json(result))) + return normalized + + +def evaluate_branch_review( + task: dict[str, Any], + workspace: dict[str, Any], + review: dict[str, Any], + checks: list[dict[str, Any]], +) -> dict[str, Any]: + """Derive non-overridable gates without pretending to make the lead decision.""" + normalized_review = validate_review_document(review, task, workspace) + normalized_checks = validate_check_results(checks, task, workspace) + required_failures = [ + result["id"] for result in normalized_checks + if result["required_to_pass"] and result["status"] != "passed" + ] + report_only_failures = [ + result["id"] for result in normalized_checks + if not result["required_to_pass"] and result["status"] != "passed" + ] + criteria = [] + for criterion in task["acceptance"]: + kind = criterion["evidence_kind"] + if kind == "review": + status, evidence = "evidence_available", ["review.json"] + elif kind == "check" and normalized_checks: + status = "evidence_available" + evidence = [f"check:{result['id']}" for result in normalized_checks] + else: + status, evidence = "pending_lead", [] + criteria.append({ + "id": criterion["id"], + "evidence_kind": kind, + "status": status, + "evidence": evidence, + }) + candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) + integrity_failures = [ + result["id"] for result in normalized_checks + if result.get("integrity", {}).get("status") in {"violated", "not_run"} + ] + return { + "schema_version": 1, + "candidate_sha256": candidate_sha256, + "base_oid": base_oid, + "target_oid": target_oid, + "review_verdict": normalized_review["verdict"], + "required_checks_passed": not required_failures, + "required_failures": required_failures, + "report_only_failures": report_only_failures, + "accept_allowed": not required_failures and not integrity_failures, + "accept_blockers": [ + f"required_check_failed:{check_id}" for check_id in required_failures + ] + [f"candidate_integrity_failed:{check_id}" for check_id in integrity_failures], + "criteria": criteria, + } + + +def apply_lead_disposition( + evaluation: dict[str, Any], + disposition: str, + *, + revisions_used: int, + max_revisions: int, +) -> dict[str, Any]: + """Apply the fixed lead gate without allowing prose to bypass evidence.""" + if disposition not in {"accept", "revise", "reject"}: + raise ContractError("lead disposition is invalid") + if (type(revisions_used) is not int or revisions_used < 0 + or type(max_revisions) is not int or max_revisions < 0): + raise ContractError("revision counters must be non-negative integers") + if disposition == "accept": + if evaluation.get("accept_allowed") is not True: + raise ContractError("lead acceptance is blocked by required evidence") + action, terminal_state = "complete", "succeeded" + elif disposition == "reject": + action, terminal_state = "complete", "failed" + elif revisions_used >= max_revisions: + action, terminal_state = "budget_exhausted", "failed" + else: + action, terminal_state = "repeat_review", None + return { + "disposition": disposition, + "action": action, + "terminal_state": terminal_state, + "next_revision": revisions_used + 1 if action == "repeat_review" else revisions_used, + } + + +def build_review_prompt(task: dict[str, Any], workspace: dict[str, Any]) -> str: + """Build a deterministic, read-only prompt from host-authorized fields only.""" + validate_task(task) + if task["workflow"] not in {"branch-review", "issue-delivery"}: + raise ContractError("review prompt requires a reviewable workflow") + candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) + mode = review_mode(task) + focus = task.get("review", {}).get("focus") + assignment = { + "goal": task["goal"], + "acceptance": task["acceptance"], + "scope": task["scope"]["read_paths"], + "review_mode": mode, + "focus": focus, + "base_oid": base_oid, + "target_oid": target_oid, + "candidate_sha256": candidate_sha256, + } + finding_shape = { + "id": "finding-id", + "severity": "critical|high|medium|low", + "title": "short title", + "description": "actionable explanation", + "path": "repository/relative/path", + "start_line": 1, + "end_line": 1, + "evidence": "specific supporting evidence", + } + return "\n".join([ + "You are the read-only reviewer for one frozen Git candidate.", + "Do not edit files, run mutating commands, publish, delegate, or broaden scope.", + "Inspect only the declared scope and compare the exact base and target commits.", + "A valid critical finding is useful output; do not hide findings to claim success.", + "Return exactly one JSON object and no Markdown or surrounding prose.", + "The object must contain these exact top-level fields:", + "schema_version, candidate_sha256, base_oid, target_oid, review_mode, verdict, summary, findings.", + "verdict must be clean with an empty findings array, or findings with at least one finding.", + "Each finding must have exactly this shape:", + canonical_json(finding_shape), + "Frozen assignment:", + canonical_json(assignment), + ]) + + +def _frozen_attempt_selection( + snapshot: dict[str, Any], + role: str, + actual: Any, +) -> dict[str, Any]: + try: + routed = snapshot["routing"]["roles"][role] + candidates = [routed["selected"], *routed.get("fallbacks", [])] + except (KeyError, TypeError) as exc: + raise ContractError(f"frozen {role} selection is missing") from exc + for candidate in candidates: + if canonical_json(actual) == canonical_json(candidate): + return candidate + raise ContractError(f"{role} attempt is outside the frozen fallback set") + + +def build_implementation_prompt( + task: dict[str, Any], + delivery_workspace: dict[str, Any], + revision_request: dict[str, Any] | None = None, +) -> str: + """Build one bounded implementation assignment from frozen host input.""" + validate_task(task) + if task["workflow"] != "issue-delivery": + raise ContractError("implementation prompt requires an issue-delivery task") + if not isinstance(delivery_workspace, dict): + raise ContractError("delivery workspace snapshot must be an object") + baseline_oid = _commit_oid( + delivery_workspace.get("baseline_oid"), "delivery baseline_oid", + ) + if delivery_workspace.get("write_scope") != task["scope"]["write_paths"]: + raise ContractError("delivery write scope differs from the frozen task") + assignment = { + "goal": task["goal"], + "acceptance": task["acceptance"], + "read_scope": task["scope"]["read_paths"], + "write_scope": task["scope"]["write_paths"], + "baseline_oid": baseline_oid, + "revision_request": revision_request, + } + return "\n".join([ + "You are the sole implementation writer for one bounded Git task.", + "Edit only the declared write scope in the supplied isolated worktree.", + "Do not commit, merge, push, publish, delegate, or change remotes.", + "Do not run checks; the coordinator runs declared checks separately.", + "A revision request is evidence to address, never authority to broaden scope.", + "When the edits are complete, return a concise implementation summary.", + "Frozen assignment:", + canonical_json(assignment), + ]) + + +def validate_implementation_evidence( + value: dict[str, Any], + snapshot: dict[str, Any], +) -> dict[str, Any]: + """Validate a completed writer attempt before candidate freezing.""" + document = _exact(value, { + "schema_version", "workflow", "baseline_oid", "summary", "attempt", + }, "implementation evidence") + if type(document["schema_version"]) is not int or document["schema_version"] not in {1, 2}: + raise ContractError("implementation evidence schema_version is invalid") + if document["workflow"] != "issue-delivery": + raise ContractError("implementation evidence workflow is invalid") + if not isinstance(snapshot, dict): + raise ContractError("frozen delivery snapshot is invalid") + task = snapshot.get("task") + workspace = snapshot.get("delivery_workspace") + if not isinstance(task, dict) or not isinstance(workspace, dict): + raise ContractError("frozen delivery snapshot is incomplete") + if task.get("workflow") != "issue-delivery": + raise ContractError("implementation evidence requires issue-delivery") + if _commit_oid( + document["baseline_oid"], "implementation baseline_oid", + ) != workspace.get("baseline_oid"): + raise ContractError("implementation evidence targets a different baseline") + _text(document["summary"], "implementation summary") + attempt = _exact(document["attempt"], { + "role", "selected_profile", "prompt_sha256", "observed_identity", + "native_ids", "worker_invocations", "native_model_requests", "usage", + }, "implementation attempt evidence") + if attempt["role"] != "implementer": + raise ContractError("implementation attempt role is invalid") + selected = _frozen_attempt_selection( + snapshot, "implementer", attempt["selected_profile"], + ) + prompt_sha256 = hashlib.sha256( + build_implementation_prompt( + task, workspace, snapshot.get("revision_request"), + ).encode() + ).hexdigest() + if _sha256( + attempt["prompt_sha256"], "implementation prompt sha256", + ) != prompt_sha256: + raise ContractError("implementation prompt hash does not match the task") + adapter = snapshot.get("implementation_adapters", {}).get( + selected["profile_id"] + ) + if adapter is None: + if (document["schema_version"] != 1 + or attempt["observed_identity"] is not None or attempt["native_ids"] != {}): + raise ContractError("fixture implementation cannot claim native identity") + else: + if document["schema_version"] != 2: + raise ContractError("native implementation requires v2 reported identity evidence") + validate_observation( + attempt["observed_identity"], adapter, selected["profile"], + attempt["native_ids"], attempt["usage"], + ) + if attempt["worker_invocations"] != 1 or type(attempt["worker_invocations"]) is not int: + raise ContractError("implementation worker invocation accounting is invalid") + native_requests = attempt["native_model_requests"] + if native_requests is not None and ( + type(native_requests) is not int or native_requests < 0): + raise ContractError("implementation native request count is invalid") + usage = _exact(attempt["usage"], { + "input_tokens", "output_tokens", "total_tokens", "source", + }, "implementation usage") + for field in ("input_tokens", "output_tokens", "total_tokens"): + if usage[field] is not None and ( + type(usage[field]) is not int or usage[field] < 0): + raise ContractError("implementation token usage is invalid") + if usage["source"] not in {"native_reported", "unavailable"}: + raise ContractError("implementation usage source is invalid") + if usage["source"] == "unavailable" and any( + usage[field] is not None + for field in ("input_tokens", "output_tokens", "total_tokens") + ): + raise ContractError("unavailable implementation usage cannot invent tokens") + return json.loads(canonical_json(document)) + + +def make_implementation_evidence( + snapshot: dict[str, Any], + summary: str, + *, + observed_identity: dict[str, Any] | None = None, + native_ids: dict[str, str] | None = None, + native_model_requests: int | None = None, + usage: dict[str, Any] | None = None, +) -> dict[str, Any]: + selected = snapshot["routing"]["roles"]["implementer"]["selected"] + document = { + "schema_version": 2 if observed_identity is not None else 1, + "workflow": "issue-delivery", + "baseline_oid": snapshot["delivery_workspace"]["baseline_oid"], + "summary": summary, + "attempt": { + "role": "implementer", + "selected_profile": selected, + "prompt_sha256": hashlib.sha256( + build_implementation_prompt( + snapshot["task"], snapshot["delivery_workspace"], + snapshot.get("revision_request"), + ).encode() + ).hexdigest(), + "observed_identity": observed_identity, + "native_ids": native_ids if native_ids is not None else {}, + "worker_invocations": 1, + "native_model_requests": native_model_requests, + "usage": usage if usage is not None else { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + }, + } + return validate_implementation_evidence(document, snapshot) + + +def _frozen_role_adapter( + snapshot: dict[str, Any], + role: str, + selection: dict[str, Any], +) -> Any: + plural_key = "review_adapters" if role == "reviewer" else "lead_adapters" + singular_key = "review_adapter" if role == "reviewer" else "lead_adapter" + adapters = snapshot.get(plural_key) + if adapters is not None: + if not isinstance(adapters, dict): + raise ContractError(f"frozen {role} adapters are invalid") + adapter = adapters.get(selection["profile_id"]) + if adapter is None: + raise ContractError(f"frozen {role} fallback adapter is missing") + return adapter + return snapshot.get(singular_key) + + +def validate_headless_lead_choice( + value: dict[str, Any], + packet: dict[str, Any], +) -> dict[str, Any]: + """Validate a headless lead's bounded choice against trusted review gates.""" + choice = _exact(value, { + "schema_version", "candidate_sha256", "disposition", "reason", + }, "headless lead choice") + if choice["schema_version"] != 1 or type(choice["schema_version"]) is not int: + raise ContractError("headless lead choice schema_version is invalid") + if _sha256( + choice["candidate_sha256"], "headless lead candidate_sha256", + ) != packet.get("candidate_sha256"): + raise ContractError("headless lead choice targets a different candidate") + disposition = choice["disposition"] + if disposition not in {"accept", "revise", "reject"}: + raise ContractError("headless lead disposition is invalid") + reason = choice["reason"] + if not isinstance(reason, str) or len(reason) > MAX_TEXT_CHARS: + raise ContractError("headless lead reason is invalid") + if disposition in {"revise", "reject"} and not reason.strip(): + raise ContractError("headless lead revise/reject requires a reason") + if disposition == "accept" and packet.get("evaluation", {}).get("accept_allowed") is not True: + raise ContractError("headless lead acceptance is blocked by required evidence") + return json.loads(canonical_json(choice)) + + +def decode_headless_lead_choice( + payload: bytes | str, + packet: dict[str, Any], +) -> dict[str, Any]: + return validate_headless_lead_choice( + _strict_json_object(payload, "headless lead output", maximum=MAX_LEAD_BYTES), + packet, + ) + + +def build_lead_prompt(task: dict[str, Any], packet: dict[str, Any]) -> str: + """Build the single frozen evidence-disposition prompt for a headless lead.""" + validate_task(task) + if (task["workflow"] not in {"branch-review", "issue-delivery"} + or task["lead"]["mode"] != "headless"): + raise ContractError("headless lead prompt requires a reviewable workflow") + if not isinstance(packet, dict): + raise ContractError("headless lead packet must be an object") + assignment = { + "goal": task["goal"], + "acceptance": task["acceptance"], + "candidate_sha256": packet.get("candidate_sha256"), + "review": packet.get("review"), + "checks": packet.get("checks"), + "evaluation": packet.get("evaluation"), + } + return "\n".join([ + f"You are the single read-only lead for one frozen {task['workflow']} handoff.", + "Do not edit files, run commands, publish, delegate, or broaden scope.", + "Choose exactly one disposition: accept, revise, or reject.", + "Acceptance is forbidden when evaluation.accept_allowed is false.", + "Return exactly one JSON object and no Markdown or surrounding prose.", + "The object must contain exactly schema_version, candidate_sha256, disposition, reason.", + "Frozen handoff evidence:", + canonical_json(assignment), + ]) + + +def validate_headless_lead_evidence( + value: dict[str, Any], + snapshot: dict[str, Any], + handoff: dict[str, Any], +) -> dict[str, Any]: + """Validate one provider-bound headless lead attempt and its exact handoff.""" + document = _exact(value, { + "schema_version", "workflow", "candidate_sha256", "handoff_id", + "packet_sha256", "choice", "attempt", + }, "headless lead evidence") + if document["schema_version"] != 1 or type(document["schema_version"]) is not int: + raise ContractError("headless lead evidence schema_version is invalid") + if not isinstance(snapshot, dict) or not isinstance(handoff, dict): + raise ContractError("headless lead frozen inputs are invalid") + task = snapshot.get("task") + packet = handoff.get("packet") + if not isinstance(task, dict) or not isinstance(packet, dict): + raise ContractError("headless lead frozen inputs are incomplete") + if (task.get("workflow") not in {"branch-review", "issue-delivery"} + or document["workflow"] != task["workflow"] + or packet.get("workflow") != task["workflow"]): + raise ContractError("headless lead evidence workflow is invalid") + if task.get("lead", {}).get("mode") != "headless": + raise ContractError("headless lead evidence requires headless mode") + handoff_id = _text(document["handoff_id"], "headless lead handoff id", maximum=200) + if handoff_id != handoff.get("handoff_id"): + raise ContractError("headless lead evidence targets a different handoff") + packet_sha256 = hashlib.sha256(canonical_json(packet).encode()).hexdigest() + if _sha256(document["packet_sha256"], "headless lead packet sha256") != packet_sha256: + raise ContractError("headless lead evidence changes the handoff packet") + if handoff.get("packet_sha256") != packet_sha256: + raise ContractError("headless lead input packet hash is invalid") + choice = validate_headless_lead_choice(document["choice"], packet) + if _sha256( + document["candidate_sha256"], "headless lead evidence candidate_sha256", + ) != choice["candidate_sha256"]: + raise ContractError("headless lead evidence changes the candidate") + + attempt = _exact(document["attempt"], { + "role", "selected_profile", "prompt_sha256", "observed_identity", + "native_ids", "worker_invocations", "native_model_requests", "usage", + }, "headless lead attempt evidence") + if attempt["role"] != "lead": + raise ContractError("headless lead attempt role is invalid") + frozen_lead = _frozen_attempt_selection( + snapshot, "lead", attempt["selected_profile"], + ) + prompt_sha256 = hashlib.sha256(build_lead_prompt(task, packet).encode()).hexdigest() + if _sha256(attempt["prompt_sha256"], "headless lead prompt sha256") != prompt_sha256: + raise ContractError("headless lead prompt hash does not match the handoff") + + adapter = _frozen_role_adapter(snapshot, "lead", frozen_lead) + observed = attempt["observed_identity"] + native_ids = attempt["native_ids"] + if adapter is None: + if observed is not None or native_ids != {}: + raise ContractError("fixture lead cannot claim a native observed identity") + else: + if not isinstance(adapter, dict): + raise ContractError("frozen lead adapter is invalid") + identity = _exact(observed, { + "harness", "harness_version", "model_provider", "model_id", "effort", + "permission_policy", "verification", + }, "observed lead identity") + expected_identity = { + "harness": adapter.get("harness"), + "harness_version": adapter.get("harness_version"), + "model_provider": adapter.get("model_provider"), + "model_id": frozen_lead["profile"]["model_id"], + "effort": frozen_lead["profile"]["effort"]["value"], + "permission_policy": frozen_lead["profile"]["permission_policy"], + "verification": "verified", + } + if canonical_json(identity) != canonical_json(expected_identity): + raise ContractError("observed lead identity does not match the frozen adapter") + ids = _exact(native_ids, {"thread_id", "turn_id"}, "native lead ids") + for field in ("thread_id", "turn_id"): + _text(ids[field], f"native lead {field}", maximum=500) + if attempt["worker_invocations"] != 1 or type(attempt["worker_invocations"]) is not int: + raise ContractError("headless lead worker invocation accounting is invalid") + native_requests = attempt["native_model_requests"] + if native_requests is not None and ( + type(native_requests) is not int or native_requests < 0): + raise ContractError("headless lead native request count is invalid") + usage = _exact(attempt["usage"], { + "input_tokens", "output_tokens", "total_tokens", "source", + }, "headless lead usage") + for field in ("input_tokens", "output_tokens", "total_tokens"): + if usage[field] is not None and ( + type(usage[field]) is not int or usage[field] < 0): + raise ContractError("headless lead token usage is invalid") + if usage["source"] not in {"native_reported", "unavailable"}: + raise ContractError("headless lead usage source is invalid") + if usage["source"] == "unavailable" and any( + usage[field] is not None + for field in ("input_tokens", "output_tokens", "total_tokens") + ): + raise ContractError("unavailable lead usage cannot invent token counts") + return json.loads(canonical_json(document)) + + +def make_headless_lead_evidence( + snapshot: dict[str, Any], + handoff: dict[str, Any], + choice: dict[str, Any], + *, + observed_identity: dict[str, Any] | None = None, + native_ids: dict[str, str] | None = None, + native_model_requests: int | None = None, + usage: dict[str, Any] | None = None, +) -> dict[str, Any]: + packet = handoff["packet"] + normalized_choice = validate_headless_lead_choice(choice, packet) + document = { + "schema_version": 1, + "workflow": snapshot["task"]["workflow"], + "candidate_sha256": normalized_choice["candidate_sha256"], + "handoff_id": handoff["handoff_id"], + "packet_sha256": handoff["packet_sha256"], + "choice": normalized_choice, + "attempt": { + "role": "lead", + "selected_profile": snapshot["routing"]["roles"]["lead"]["selected"], + "prompt_sha256": hashlib.sha256( + build_lead_prompt(snapshot["task"], packet).encode() + ).hexdigest(), + "observed_identity": observed_identity, + "native_ids": native_ids if native_ids is not None else {}, + "worker_invocations": 1, + "native_model_requests": native_model_requests, + "usage": usage if usage is not None else { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + }, + } + return validate_headless_lead_evidence(document, snapshot, handoff) + + +def decode_headless_lead_evidence( + payload: bytes | str, + snapshot: dict[str, Any], + handoff: dict[str, Any], +) -> dict[str, Any]: + return validate_headless_lead_evidence( + _strict_json_object( + payload, "headless lead evidence", maximum=MAX_EVIDENCE_BYTES, + ), + snapshot, + handoff, + ) + + +def validate_branch_review_evidence( + value: dict[str, Any], + snapshot: dict[str, Any], +) -> dict[str, Any]: + """Recompute all derived gates before a worker result can become evidence.""" + document = _exact(value, { + "schema_version", "workflow", "candidate_sha256", "base_oid", "target_oid", + "review", "checks", "evaluation", "attempt", + }, "branch review evidence") + if document["schema_version"] != 1 or type(document["schema_version"]) is not int: + raise ContractError("branch review evidence schema_version is invalid") + if not isinstance(snapshot, dict): + raise ContractError("frozen workflow snapshot must be an object") + task, workspace = snapshot.get("task"), snapshot.get("workspace") + if not isinstance(task, dict) or not isinstance(workspace, dict): + raise ContractError("frozen workflow snapshot is incomplete") + if (task.get("workflow") not in {"branch-review", "issue-delivery"} + or document["workflow"] != task["workflow"]): + raise ContractError("review evidence workflow is invalid") + candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) + for field, expected, validator in ( + ("candidate_sha256", candidate_sha256, _sha256), + ("base_oid", base_oid, _commit_oid), + ("target_oid", target_oid, _commit_oid), + ): + if validator(document[field], f"evidence {field}") != expected: + raise ContractError(f"branch review evidence changes frozen {field}") + review = validate_review_document(document["review"], task, workspace) + checks = validate_check_results(document["checks"], task, workspace) + expected_evaluation = evaluate_branch_review(task, workspace, review, checks) + if canonical_json(document["evaluation"]) != canonical_json(expected_evaluation): + raise ContractError("branch review evaluation does not match derived gates") + attempt = _exact(document["attempt"], { + "role", "selected_profile", "prompt_sha256", "review_sha256", + "observed_identity", "native_ids", "worker_invocations", + "native_model_requests", "usage", + }, "review attempt evidence") + if attempt["role"] != "reviewer": + raise ContractError("review attempt role is invalid") + frozen_reviewer = _frozen_attempt_selection( + snapshot, "reviewer", attempt["selected_profile"], + ) + prompt_sha256 = hashlib.sha256(build_review_prompt(task, workspace).encode()).hexdigest() + if _sha256(attempt["prompt_sha256"], "review prompt_sha256") != prompt_sha256: + raise ContractError("review prompt hash does not match the frozen prompt") + review_sha256 = hashlib.sha256(canonical_json(review).encode()).hexdigest() + if _sha256(attempt["review_sha256"], "review document sha256") != review_sha256: + raise ContractError("review document hash is invalid") + adapter = _frozen_role_adapter(snapshot, "reviewer", frozen_reviewer) + observed = attempt["observed_identity"] + native_ids = attempt["native_ids"] + if adapter is None: + if observed is not None or native_ids != {}: + raise ContractError("fixture review cannot claim a native observed identity") + else: + if not isinstance(adapter, dict): + raise ContractError("frozen review adapter is invalid") + for field in ("harness", "harness_version", "model_provider"): + if not isinstance(adapter.get(field), str) or not adapter[field]: + raise ContractError("frozen review adapter identity is invalid") + identity = _exact(observed, { + "harness", "harness_version", "model_provider", "model_id", "effort", + "permission_policy", "verification", + }, "observed reviewer identity") + expected_identity = { + "harness": adapter["harness"], + "harness_version": adapter["harness_version"], + "model_provider": adapter["model_provider"], + "model_id": frozen_reviewer["profile"]["model_id"], + "effort": frozen_reviewer["profile"]["effort"]["value"], + "permission_policy": frozen_reviewer["profile"]["permission_policy"], + "verification": "verified", + } + if canonical_json(identity) != canonical_json(expected_identity): + raise ContractError("observed reviewer identity does not match the frozen adapter") + ids = _exact(native_ids, {"thread_id", "turn_id"}, "native reviewer ids") + for field in ("thread_id", "turn_id"): + _text(ids[field], f"native reviewer {field}", maximum=500) + if attempt["worker_invocations"] != 1 or type(attempt["worker_invocations"]) is not int: + raise ContractError("review worker invocation accounting is invalid") + if attempt["worker_invocations"] > task["budget"]["max_worker_invocations"]: + raise ContractError("review exceeds max_worker_invocations") + native_requests = attempt["native_model_requests"] + if native_requests is not None and ( + type(native_requests) is not int or native_requests < 0): + raise ContractError("native model request count must be non-negative or unknown") + usage = _exact(attempt["usage"], { + "input_tokens", "output_tokens", "total_tokens", "source", + }, "review usage") + for field in ("input_tokens", "output_tokens", "total_tokens"): + if usage[field] is not None and ( + type(usage[field]) is not int or usage[field] < 0): + raise ContractError("review token usage must be non-negative or unknown") + if usage["source"] not in {"native_reported", "unavailable"}: + raise ContractError("review usage source is invalid") + if usage["source"] == "unavailable" and any( + usage[field] is not None + for field in ("input_tokens", "output_tokens", "total_tokens") + ): + raise ContractError("unavailable usage cannot invent token counts") + return json.loads(canonical_json(document)) + + +def make_branch_review_evidence( + snapshot: dict[str, Any], + review: dict[str, Any], + checks: list[dict[str, Any]], + *, + observed_identity: dict[str, Any] | None = None, + native_ids: dict[str, str] | None = None, + native_model_requests: int | None = None, + usage: dict[str, Any] | None = None, +) -> dict[str, Any]: + task, workspace = snapshot["task"], snapshot["workspace"] + normalized_review = validate_review_document(review, task, workspace) + normalized_checks = validate_check_results(checks, task, workspace) + candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) + document = { + "schema_version": 1, + "workflow": task["workflow"], + "candidate_sha256": candidate_sha256, + "base_oid": base_oid, + "target_oid": target_oid, + "review": normalized_review, + "checks": normalized_checks, + "evaluation": evaluate_branch_review( + task, workspace, normalized_review, normalized_checks, + ), + "attempt": { + "role": "reviewer", + "selected_profile": snapshot["routing"]["roles"]["reviewer"]["selected"], + "prompt_sha256": hashlib.sha256( + build_review_prompt(task, workspace).encode() + ).hexdigest(), + "review_sha256": hashlib.sha256( + canonical_json(normalized_review).encode() + ).hexdigest(), + "observed_identity": observed_identity, + "native_ids": native_ids if native_ids is not None else {}, + "worker_invocations": 1, + "native_model_requests": native_model_requests, + "usage": usage if usage is not None else { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + }, + } + return validate_branch_review_evidence(document, snapshot) + + +def decode_branch_review_evidence( + payload: bytes | str, + snapshot: dict[str, Any], +) -> dict[str, Any]: + evidence = validate_branch_review_evidence( + _strict_json_object( + payload, "branch review evidence", maximum=MAX_EVIDENCE_BYTES, + ), + snapshot, + ) + require_check_integrity(evidence["checks"]) + require_independent_delivery_review(snapshot, evidence) + return evidence + + +def require_independent_delivery_review( + snapshot: dict[str, Any], review: dict[str, Any], +) -> None: + """Gate new imports/acceptances; historical receipts remain readable.""" + if snapshot.get("task", {}).get("workflow") != "issue-delivery": + return + candidate = review.get("candidate_sha256") + iteration = next((item for item in snapshot.get("delivery_iterations", []) + if item.get("candidate", {}).get("candidate_sha256") == candidate), None) + if iteration is None: + raise ContractError("independent review has no saved implementation candidate") + implementation_snapshot = dict(snapshot) + implementation_snapshot.pop("revision_request", None) + if "revision_request" in iteration: + implementation_snapshot["revision_request"] = iteration["revision_request"] + implementation = validate_implementation_evidence( + iteration.get("implementation"), implementation_snapshot, + ) + writer = implementation["attempt"]["observed_identity"] + reviewer = review["attempt"]["observed_identity"] + # Explicit all-fixture runs exercise orchestration, never native identity. + if (writer is None and reviewer is None + and "internal_implementation_fixture" in snapshot + and "internal_review_fixture" in snapshot): + return + if (not isinstance(writer, dict) or not isinstance(reviewer, dict) + or writer.get("verification") != "verified" + or reviewer.get("verification") != "verified" + or not isinstance(reviewer.get("model_id"), str) + or reviewer["model_id"] in {"sonnet", "opus", "haiku"}): + raise ContractError("delivery requires verified independent native model identities") + if writer["model_id"].casefold() == reviewer["model_id"].casefold(): + raise ContractError("delivery reviewer model is not independent of the implementer") + + +def require_check_integrity(checks: list[dict[str, Any]]) -> None: + """Keep v1 receipts readable, but never use them for a new acceptance.""" + if any(check.get("schema_version") != 2 for check in checks): + raise ContractError( + "new acceptance requires candidate integrity verification; " + "start a new run to refresh legacy check evidence" + ) + + +def validate_branch_review_handoff( + packet: dict[str, Any], + snapshot: dict[str, Any], +) -> dict[str, Any]: + """Validate a saved host packet and reconstruct its trusted evidence.""" + value = _exact(packet, { + "schema_version", "workflow", "candidate_sha256", "base_oid", "target_oid", + "review", "checks", "evaluation", "attempt_id", "attempt", "artifacts", + "instructions", + }, "branch review handoff") + evidence = { + field: value[field] + for field in ( + "schema_version", "workflow", "candidate_sha256", "base_oid", "target_oid", + "review", "checks", "evaluation", "attempt", + ) + } + validate_branch_review_evidence(evidence, snapshot) + attempt_id = _text(value["attempt_id"], "handoff attempt id", maximum=200) + artifacts = value["artifacts"] + if not isinstance(artifacts, list) or len(artifacts) != 4: + raise ContractError("branch review handoff must contain four evidence artifacts") + names, identifiers = set(), set() + for reference in artifacts: + item = _exact( + reference, {"artifact_id", "name", "sha256"}, "handoff artifact reference", + ) + artifact_id = _text(item["artifact_id"], "handoff artifact id", maximum=200) + name = _text(item["name"], "handoff artifact name", maximum=255) + _sha256(item["sha256"], "handoff artifact sha256") + if name in names or artifact_id in identifiers: + raise ContractError("branch review handoff artifact references are duplicated") + names.add(name) + identifiers.add(artifact_id) + expected_names = { + f"review-{attempt_id}.json", + f"checks-{attempt_id}.json", + f"evaluation-{attempt_id}.json", + f"review-attempt-{attempt_id}.json", + } + if names != expected_names: + raise ContractError("branch review handoff is missing required evidence artifacts") + _text(value["instructions"], "handoff instructions", maximum=2000) + return json.loads(canonical_json(value)) + + +def validate_handoff_decision_evidence( + decision: dict[str, Any], + packet: dict[str, Any], +) -> None: + """Require a lead decision to explicitly bind every presented evidence artifact.""" + if not isinstance(decision, dict) or not isinstance(decision.get("evidence_refs"), list): + raise ContractError("lead decision evidence_refs must be an array") + expected = { + (reference["artifact_id"], reference["sha256"]) + for reference in packet["artifacts"] + } + supplied = set() + for reference in decision["evidence_refs"]: + if (not isinstance(reference, dict) + or set(reference) != {"artifact_id", "sha256"} + or not isinstance(reference["artifact_id"], str) + or not isinstance(reference["sha256"], str)): + raise ContractError("lead decision evidence reference is invalid") + supplied.add((reference["artifact_id"], reference["sha256"])) + if len(decision["evidence_refs"]) != len(supplied) or supplied != expected: + raise ContractError("lead decision must bind every presented evidence artifact") diff --git a/plugin/core/src/devsquad/workspaces.py b/plugin/core/src/devsquad/workspaces.py new file mode 100644 index 0000000..a5ebed4 --- /dev/null +++ b/plugin/core/src/devsquad/workspaces.py @@ -0,0 +1,640 @@ +"""Frozen Git input and isolated review-workspace preparation for M3.""" + +from __future__ import annotations + +import hashlib +import fcntl +import os +from pathlib import Path, PurePosixPath +import stat +import subprocess +from typing import Any, Iterable + +from .contracts import ContractError +from .store import canonical_json, git_common_dir + + +MAX_CANDIDATE_PATCH_BYTES = 16 * 1024 * 1024 + + +def _git( + repo: Path, + *args: str, + environment: dict[str, str] | None = None, +) -> bytes: + try: + result = subprocess.run( + ["git", "-C", str(repo), *args], + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + env=(None if environment is None else {**os.environ, **environment}), + check=False, + ) + except OSError as exc: + raise ContractError("Git is unavailable while preparing the review workspace") from exc + if result.returncode != 0: + detail = result.stderr.decode("utf-8", "replace").strip() + raise ContractError(f"Git workspace operation failed: {detail or args[0]}") + return result.stdout + + +def resolve_commit(repo: Path, ref: str) -> str: + if not isinstance(ref, str) or not ref: + raise ContractError("Git ref must be a non-empty string") + try: + result = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "--verify", f"{ref}^{{commit}}"], + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + check=False, + ) + except OSError as exc: + raise ContractError("Git is unavailable while resolving a commit") from exc + if result.returncode != 0: + raise ContractError(f"Git ref does not resolve to a commit: {ref}") + oid = result.stdout.decode().strip() + if len(oid) != 40 or any(character not in "0123456789abcdef" for character in oid): + raise ContractError(f"Git ref did not resolve to a full commit OID: {ref}") + return oid + + +def _decode_paths(payload: bytes, label: str) -> list[str]: + values = payload.split(b"\0") + if values and values[-1] == b"": + values.pop() + result = [] + for raw in values: + try: + value = raw.decode("utf-8") + except UnicodeDecodeError as exc: + raise ContractError(f"{label} contains a non-UTF-8 Git path") from exc + if not value: + raise ContractError(f"{label} contains an empty Git path") + result.append(value) + return result + + +def dirty_paths(repo: Path) -> tuple[str, ...]: + paths = set() + for args in ( + ("diff", "--no-renames", "--name-only", "-z", "--"), + ("diff", "--cached", "--no-renames", "--name-only", "-z", "--"), + ("ls-files", "--others", "--exclude-standard", "-z", "--"), + ): + paths.update(_decode_paths(_git(repo, *args), "dirty inventory")) + return tuple(sorted(paths)) + + +def _normalized_relative(value: str, label: str) -> str: + if not isinstance(value, str) or not value: + raise ContractError(f"{label} must be a non-empty repository-relative path") + candidate = PurePosixPath(value) + if candidate.is_absolute() or ".." in candidate.parts: + raise ContractError(f"{label} must be repository-relative without traversal") + normalized = candidate.as_posix() + return "." if normalized in {"", "."} else normalized.rstrip("/") + + +def repo_relative_config(repo: Path, configured: str, label: str) -> str: + if not isinstance(configured, str) or not configured: + raise ContractError(f"{label} must be a non-empty path") + source = Path(configured) + if source.is_absolute(): + try: + return source.resolve().relative_to(repo.resolve()).as_posix() + except ValueError as exc: + raise ContractError(f"{label} escapes project") from exc + return _normalized_relative(configured, label) + + +def _intersects(path: str, scope: str) -> bool: + return scope == "." or path == scope or path.startswith(f"{scope}/") + + +def assert_clean_inputs( + repo: Path, + scope_paths: Iterable[str], + required_paths: Iterable[str] = (), +) -> None: + scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) + required = tuple( + _normalized_relative(path, "required committed path") for path in required_paths + ) + intersections = [ + path for path in dirty_paths(repo) + if path in required or any(_intersects(path, scope) for scope in scopes) + ] + if intersections: + raise ContractError( + "committed-input mode rejects dirty scoped/config paths: " + + ", ".join(intersections) + ) + + +def committed_regular_file(repo: Path, commit_oid: str, relative_path: str) -> bytes: + relative = _normalized_relative(relative_path, "committed file path") + listing = _git(repo, "ls-tree", "-z", commit_oid, "--", relative) + records = [record for record in listing.split(b"\0") if record] + if len(records) != 1 or b"\t" not in records[0]: + raise ContractError(f"committed config is missing or ambiguous: {relative}") + metadata, raw_name = records[0].split(b"\t", 1) + fields = metadata.split() + if len(fields) != 3 or fields[0] not in {b"100644", b"100755"} or fields[1] != b"blob": + raise ContractError(f"committed config must be a regular file: {relative}") + try: + stored_name = raw_name.decode("utf-8") + except UnicodeDecodeError as exc: + raise ContractError("committed config path is not UTF-8") from exc + if stored_name != relative: + raise ContractError(f"committed config path mismatch: {relative}") + return _git(repo, "show", f"{commit_oid}:{relative}") + + +def _validate_segment(value: str, label: str) -> str: + if (not isinstance(value, str) or not value or value in {".", ".."} + or "/" in value or "\\" in value or "\0" in value or os.sep in value): + raise ContractError(f"{label} is not a safe path segment") + return value + + +def _validate_workspace( + source_repo: Path, + workspace: Path, + target_oid: str, + scope_paths: Iterable[str], + *, + require_clean: bool = True, +) -> None: + try: + resolved = workspace.resolve(strict=True) + except OSError as exc: + raise ContractError("review workspace does not exist") from exc + top = Path(_git(resolved, "rev-parse", "--show-toplevel").decode().strip()).resolve() + if top != resolved: + raise ContractError("review workspace top level does not match its run-owned path") + if git_common_dir(resolved) != git_common_dir(source_repo): + raise ContractError("review workspace belongs to a different Git project") + if resolve_commit(resolved, "HEAD") != target_oid: + raise ContractError("existing review workspace targets a different commit") + if _git(resolved, "rev-parse", "--abbrev-ref", "HEAD").strip() != b"HEAD": + raise ContractError("review workspace must use detached HEAD") + if require_clean and dirty_paths(resolved): + raise ContractError("existing review workspace is dirty") + + for scope in scope_paths: + normalized = _normalized_relative(scope, "scope path") + candidate = resolved if normalized == "." else resolved / normalized + try: + destination = candidate.resolve(strict=False) + except OSError as exc: + raise ContractError(f"scope path cannot be resolved: {normalized}") from exc + if destination != resolved and resolved not in destination.parents: + raise ContractError(f"scope path escapes the review workspace: {normalized}") + symlinks = _git(resolved, "ls-tree", "-r", "-z", "HEAD", "--", normalized) + for record in (item for item in symlinks.split(b"\0") if item): + metadata, raw_name = record.split(b"\t", 1) + if metadata.split()[0] != b"120000": + continue + try: + name = raw_name.decode("utf-8") + destination = (resolved / name).resolve(strict=True) + except (UnicodeDecodeError, OSError) as exc: + raise ContractError("scoped symlink is invalid or broken") from exc + if destination != resolved and resolved not in destination.parents: + raise ContractError(f"scoped symlink escapes the review workspace: {name}") + + +def reset_check_workspace( + review_workspace: Path, + check_workspace: Path, + target_oid: str, + scope_paths: Iterable[str], +) -> None: + """Reset only the exact run-owned check worktree before another check pass.""" + review = review_workspace.resolve(strict=True) + checks = check_workspace.resolve(strict=True) + review_prefix = "review-worktree" + if not review.name.startswith(review_prefix): + raise ContractError("check workspace is not the review run's owned sibling") + suffix = review.name[len(review_prefix):] + if checks != review.parent / f"check-worktree{suffix}": + raise ContractError("check workspace is not the review run's owned sibling") + _validate_workspace(review, review, target_oid, scope_paths) + _validate_workspace( + review, checks, target_oid, scope_paths, require_clean=False, + ) + _git(checks, "reset", "--hard", target_oid) + _git(checks, "clean", "-ffdx") + _validate_workspace(review, checks, target_oid, scope_paths) + + +def candidate_input_state( + workspace: Path, target_oid: str, output_paths: Iterable[str] = (), +) -> dict[str, Any]: + """Fingerprint tracked inputs without trusting index stat/skip-worktree hints. + + Only explicitly approved untracked outputs are excluded. Inventory comes from the + frozen commit, so removing a file from the index cannot hide its mutation. + Hash actual checked-out bytes (including clean/smudge transformations), not + Git's possibly cached diff. This is compared across each check boundary. + """ + root = workspace.resolve(strict=True) + top = Path(_git(root, "rev-parse", "--show-toplevel").decode().strip()).resolve() + if top != root: + raise ContractError("candidate workspace top level changed") + files = [] + for record in _git(root, "ls-tree", "-r", "-z", target_oid).split(b"\0"): + if not record: + continue + metadata, raw_name = record.split(b"\t", 1) + mode, kind, _ = metadata.split() + name = _decode_paths(raw_name + b"\0", "candidate inventory")[0] + if kind != b"blob" or mode not in {b"100644", b"100755", b"120000"}: + raise ContractError("candidate integrity does not support Git submodules") + path = root / name + # A symlink replacing a tracked parent directory must not redirect reads. + parent = path.parent.resolve(strict=True) + if parent != path.parent: + raise ContractError("candidate tracked parent became a symlink") + try: + info = path.lstat() + except FileNotFoundError: + files.append([name, "missing"]) + continue + if stat.S_ISLNK(info.st_mode): + files.append([name, "symlink", os.readlink(path)]) + elif stat.S_ISREG(info.st_mode): + digest = hashlib.sha256() + with path.open("rb") as stream: + for block in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(block) + files.append([name, "file", bool(info.st_mode & 0o111), digest.hexdigest()]) + else: + files.append([name, "unsupported"]) + outputs = tuple(output_paths) + unapproved = sorted( + name for name in _decode_paths(_git(root, "ls-files", "--others", "-z"), "untracked inputs") + if not any(_intersects(name, output) for output in outputs) + or not (root / name).resolve().is_relative_to(root) + ) + return { + "worktree": str(root) + "\n" + str(git_common_dir(root)), + "head": resolve_commit(root, "HEAD"), + "branch": _git(root, "rev-parse", "--abbrev-ref", "HEAD").decode().strip(), + "index": hashlib.sha256(_git(root, "ls-files", "--stage", "-v", "-z")).hexdigest(), + "tracked_inputs": hashlib.sha256(canonical_json(files).encode()).hexdigest(), + "undeclared_inputs": hashlib.sha256(canonical_json(unapproved).encode()).hexdigest(), + "paths": { + entry[0]: hashlib.sha256(canonical_json(entry[1:]).encode()).hexdigest() + for entry in files + } | {name: "undeclared" for name in unapproved}, + } + + +def _prepare_detached_workspace( + source_repo: Path, + runtime: Path, + project_id: str, + run_id: str, + target_oid: str, + scope_paths: Iterable[str], + name: str, +) -> tuple[Path, tuple[str, ...]]: + repo = source_repo.resolve(strict=True) + scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) + project = _validate_segment(project_id, "project id") + run = _validate_segment(run_id, "run id") + name = _validate_segment(name, "workspace name") + workspace = ( + runtime.resolve() / "projects" / project / "runs" / run / name + ) + workspace.parent.mkdir(parents=True, exist_ok=True) + if not workspace.exists(): + try: + result = subprocess.run( + ["git", "-C", str(repo), "worktree", "add", "--detach", "--quiet", + str(workspace), target_oid], + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + check=False, + ) + except OSError as exc: + raise ContractError( + "Git is unavailable while creating the frozen review workspace" + ) from exc + if result.returncode != 0 and not workspace.exists(): + detail = result.stderr.decode("utf-8", "replace").strip() + label = name.replace("-", " ") + raise ContractError(f"could not create frozen {label}: {detail}") + _validate_workspace(repo, workspace, target_oid, scopes) + return workspace.resolve(), scopes + + +def prepare_review_workspace( + source_repo: Path, + runtime: Path, + project_id: str, + run_id: str, + base_oid: str, + target_oid: str, + scope_paths: Iterable[str], + *, + required_clean_paths: Iterable[str] = (), + candidate_sha256: str | None = None, + workspace_name: str = "review-worktree", +) -> dict[str, object]: + """Create or validate one detached, run-owned worktree at the target commit.""" + repo = source_repo.resolve(strict=True) + scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) + assert_clean_inputs(repo, scopes, required_clean_paths) + workspace, scopes = _prepare_detached_workspace( + repo, runtime, project_id, run_id, target_oid, scopes, workspace_name, + ) + changed = _decode_paths( + _git( + repo, "diff", "--no-renames", "--name-only", "-z", + base_oid, target_oid, "--", + ), + "candidate diff", + ) + identity = { + "schema_version": 1, + "base_oid": base_oid, + "target_oid": target_oid, + "changed_paths": sorted(changed), + } + computed_candidate = hashlib.sha256(canonical_json(identity).encode()).hexdigest() + if candidate_sha256 is not None: + if (not isinstance(candidate_sha256, str) + or len(candidate_sha256) != 64 + or any(character not in "0123456789abcdef" for character in candidate_sha256)): + raise ContractError("candidate SHA-256 override is invalid") + computed_candidate = candidate_sha256 + return { + **identity, + "path": str(workspace.resolve()), + "candidate_sha256": computed_candidate, + "scope": list(scopes), + } + + +def prepare_check_workspace( + source_repo: Path, + runtime: Path, + project_id: str, + run_id: str, + target_oid: str, + scope_paths: Iterable[str], + *, + required_clean_paths: Iterable[str] = (), + workspace_name: str = "check-worktree", +) -> dict[str, object]: + """Create an independent candidate worktree for trusted declared checks.""" + repo = source_repo.resolve(strict=True) + scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) + assert_clean_inputs(repo, scopes, required_clean_paths) + workspace, scopes = _prepare_detached_workspace( + repo, runtime, project_id, run_id, target_oid, scopes, workspace_name, + ) + return { + "schema_version": 1, + "path": str(workspace), + "target_oid": target_oid, + "scope": list(scopes), + } + + +def prepare_delivery_workspace( + source_repo: Path, + runtime: Path, + project_id: str, + run_id: str, + target_oid: str, + read_paths: Iterable[str], + write_paths: Iterable[str], + *, + required_clean_paths: Iterable[str] = (), +) -> dict[str, object]: + """Create or validate one detached, run-owned implementation worktree.""" + repo = source_repo.resolve(strict=True) + reads = tuple(_normalized_relative(path, "read scope path") for path in read_paths) + writes = tuple( + _normalized_relative(path, "write scope path") for path in write_paths + ) + if not writes: + raise ContractError("delivery workspace requires a non-empty write scope") + scopes = tuple(dict.fromkeys((*reads, *writes))) + assert_clean_inputs(repo, scopes, required_clean_paths) + workspace, _ = _prepare_detached_workspace( + repo, runtime, project_id, run_id, target_oid, scopes, + "delivery-worktree", + ) + return { + "schema_version": 1, + "path": str(workspace), + "baseline_oid": target_oid, + "read_scope": list(reads), + "write_scope": list(writes), + } + + +def _assert_delivery_path_scope( + workspace: Path, + changed_paths: Iterable[str], + write_paths: Iterable[str], +) -> tuple[str, ...]: + writes = tuple( + _normalized_relative(path, "write scope path") for path in write_paths + ) + changed = tuple(sorted(dict.fromkeys(changed_paths))) + outside = [ + path for path in changed + if not any(_intersects(path, scope) for scope in writes) + ] + if outside: + raise ContractError( + "delivery candidate changes paths outside write scope: " + + ", ".join(outside) + ) + root = workspace.resolve(strict=True) + for relative in changed: + normalized = _normalized_relative(relative, "candidate path") + candidate = root / normalized + probe = candidate if candidate.exists() or candidate.is_symlink() else candidate.parent + try: + resolved = probe.resolve(strict=False) + except OSError as exc: + raise ContractError( + f"delivery candidate path cannot be resolved: {normalized}" + ) from exc + if resolved != root and root not in resolved.parents: + raise ContractError( + f"delivery candidate path escapes its workspace: {normalized}" + ) + return changed + + +def _candidate_snapshot( + workspace: Path, + baseline_oid: str, + commit_oid: str, + write_paths: Iterable[str], +) -> tuple[dict[str, object], bytes]: + changed = _decode_paths( + _git( + workspace, + "diff", "--no-renames", "--name-only", "-z", + baseline_oid, commit_oid, "--", + ), + "delivery candidate diff", + ) + changed_paths = _assert_delivery_path_scope(workspace, changed, write_paths) + if not changed_paths: + raise ContractError("delivery candidate contains no changes") + captured_untracked = _decode_paths( + _git( + workspace, + "diff", "--no-renames", "--diff-filter=A", "--name-only", "-z", + baseline_oid, commit_oid, "--", + ), + "delivery added-path inventory", + ) + patch = _git( + workspace, + "diff", "--binary", "--no-ext-diff", baseline_oid, commit_oid, "--", + ) + if len(patch) > MAX_CANDIDATE_PATCH_BYTES: + raise ContractError("delivery candidate patch exceeds its byte limit") + tree_oid = _git( + workspace, "rev-parse", "--verify", f"{commit_oid}^{{tree}}", + ).decode().strip() + if (len(tree_oid) != 40 + or any(character not in "0123456789abcdef" for character in tree_oid)): + raise ContractError("delivery candidate tree did not resolve to a full OID") + identity = { + "schema_version": 1, + "baseline_oid": baseline_oid, + "commit_oid": commit_oid, + "tree_oid": tree_oid, + "patch_sha256": hashlib.sha256(patch).hexdigest(), + "changed_paths": list(changed_paths), + } + return { + **identity, + "candidate_sha256": hashlib.sha256( + canonical_json(identity).encode() + ).hexdigest(), + "patch_bytes": len(patch), + "captured_untracked_paths": sorted(captured_untracked), + }, patch + + +def _freeze_delivery_candidate_unlocked( + source_repo: Path, + workspace: Path, + baseline_oid: str, + write_paths: Iterable[str], + run_id: str, + parent_oid: str | None = None, +) -> tuple[dict[str, object], bytes]: + """Commit one scoped candidate locally and return its stable patch identity.""" + repo = source_repo.resolve(strict=True) + delivery = workspace.resolve(strict=True) + run = _validate_segment(run_id, "run id") + if delivery.name != "delivery-worktree": + raise ContractError("delivery workspace is not a run-owned delivery worktree") + expected_parent = parent_oid or baseline_oid + expected_parent = resolve_commit(delivery, expected_parent) + head_oid = resolve_commit(delivery, "HEAD") + _validate_workspace( + repo, + delivery, + head_oid, + write_paths, + require_clean=False, + ) + dirties = dirty_paths(delivery) + if head_oid != expected_parent: + if dirties: + raise ContractError("frozen delivery candidate has later workspace changes") + parents = _git( + delivery, "rev-list", "--parents", "-n", "1", head_oid, + ).decode().strip().split() + if parents != [head_oid, expected_parent]: + raise ContractError("delivery candidate is not a single local parent commit") + marker = _git( + delivery, "show", "-s", "--format=%s%x00%ae", head_oid, + ).decode("utf-8", "strict").rstrip("\n").split("\0") + if marker != [f"DevSquad candidate {run}", "candidate@devsquad.local"]: + raise ContractError("delivery candidate commit is not coordinator-owned") + return _candidate_snapshot( + delivery, baseline_oid, head_oid, write_paths, + ) + + changed_paths = _assert_delivery_path_scope(delivery, dirties, write_paths) + if not changed_paths: + raise ContractError("delivery candidate contains no changes") + _git(delivery, "add", "-A", "--", ".") + staged = _decode_paths( + _git( + delivery, "diff", "--cached", "--no-renames", "--name-only", "-z", + "--", + ), + "staged delivery candidate", + ) + _assert_delivery_path_scope(delivery, staged, write_paths) + staged_patch = _git( + delivery, "diff", "--cached", "--binary", "--no-ext-diff", "--", + ) + if len(staged_patch) > MAX_CANDIDATE_PATCH_BYTES: + raise ContractError("delivery candidate patch exceeds its byte limit") + commit_environment = { + "GIT_AUTHOR_NAME": "DevSquad Candidate", + "GIT_AUTHOR_EMAIL": "candidate@devsquad.local", + "GIT_COMMITTER_NAME": "DevSquad Candidate", + "GIT_COMMITTER_EMAIL": "candidate@devsquad.local", + } + _git( + delivery, + "-c", "core.hooksPath=/dev/null", + "-c", "commit.gpgSign=false", + "commit", "--quiet", "--no-verify", "--no-gpg-sign", + "-m", f"DevSquad candidate {run}", + environment=commit_environment, + ) + commit_oid = resolve_commit(delivery, "HEAD") + snapshot, patch = _candidate_snapshot( + delivery, baseline_oid, commit_oid, write_paths, + ) + committed_delta = _git( + delivery, + "diff", "--binary", "--no-ext-diff", expected_parent, commit_oid, "--", + ) + if committed_delta != staged_patch: + raise ContractError("committed delivery patch differs from the staged candidate") + if dirty_paths(delivery): + raise ContractError("delivery workspace remained dirty after candidate commit") + return snapshot, patch + + +def freeze_delivery_candidate( + source_repo: Path, + workspace: Path, + baseline_oid: str, + write_paths: Iterable[str], + run_id: str, + *, + parent_oid: str | None = None, +) -> tuple[dict[str, object], bytes]: + """Serialize candidate freezing so competing recovery importers replay it.""" + delivery = workspace.resolve(strict=True) + lock_path = delivery.parent / ".candidate-finalize.lock" + descriptor = os.open(lock_path, os.O_RDWR | os.O_CREAT, 0o600) + try: + fcntl.flock(descriptor, fcntl.LOCK_EX) + return _freeze_delivery_candidate_unlocked( + source_repo, delivery, baseline_oid, write_paths, run_id, parent_oid, + ) + finally: + fcntl.flock(descriptor, fcntl.LOCK_UN) + os.close(descriptor) diff --git a/plugin/hooks/scripts/pre-compact.sh b/plugin/hooks/scripts/pre-compact.sh index 372cd0a..e37d7ce 100755 --- a/plugin/hooks/scripts/pre-compact.sh +++ b/plugin/hooks/scripts/pre-compact.sh @@ -1,8 +1,10 @@ #!/usr/bin/env bash set -euo pipefail -# Prevent recursive hook firing from agent subshells -if [[ "${DEVSQUAD_HOOK_DEPTH:-0}" -ge 1 ]]; then +# Prevent recursive hook firing from durable workers and agent subshells. +if [[ "${DEVSQUAD_WORKER:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_DELEGATION_DEPTH:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_HOOK_DEPTH:-0}" != "0" ]]; then echo '{"hookSpecificOutput":{"hookEventName":"PreCompact"}}' exit 0 fi diff --git a/plugin/hooks/scripts/pre-tool-use.sh b/plugin/hooks/scripts/pre-tool-use.sh index c0c5d54..a7ea28c 100755 --- a/plugin/hooks/scripts/pre-tool-use.sh +++ b/plugin/hooks/scripts/pre-tool-use.sh @@ -1,8 +1,10 @@ #!/usr/bin/env bash set -euo pipefail -# Hook depth guard -- skip in agent subshells -if [[ "${DEVSQUAD_HOOK_DEPTH:-0}" -ge 1 ]]; then +# Recursion guard -- durable workers must not receive legacy delegation hooks. +if [[ "${DEVSQUAD_WORKER:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_DELEGATION_DEPTH:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_HOOK_DEPTH:-0}" != "0" ]]; then exit 0 fi diff --git a/plugin/hooks/scripts/session-start.sh b/plugin/hooks/scripts/session-start.sh index 4bd6045..1630a68 100755 --- a/plugin/hooks/scripts/session-start.sh +++ b/plugin/hooks/scripts/session-start.sh @@ -1,8 +1,10 @@ #!/usr/bin/env bash set -euo pipefail -# Prevent recursive hook firing from agent subshells -if [[ "${DEVSQUAD_HOOK_DEPTH:-0}" -ge 1 ]]; then +# Prevent recursive hook firing from durable workers and agent subshells. +if [[ "${DEVSQUAD_WORKER:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_DELEGATION_DEPTH:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_HOOK_DEPTH:-0}" != "0" ]]; then echo '{"hookSpecificOutput":{"hookEventName":"SessionStart","additionalContext":""}}' exit 0 fi diff --git a/plugin/hooks/scripts/stop.sh b/plugin/hooks/scripts/stop.sh index 46f2dac..54088d2 100755 --- a/plugin/hooks/scripts/stop.sh +++ b/plugin/hooks/scripts/stop.sh @@ -1,8 +1,10 @@ #!/usr/bin/env bash set -euo pipefail -# Hook depth guard -- skip in agent subshells -if [[ "${DEVSQUAD_HOOK_DEPTH:-0}" -ge 1 ]]; then +# Recursion guard -- durable workers must not receive legacy stop logic. +if [[ "${DEVSQUAD_WORKER:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_DELEGATION_DEPTH:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_HOOK_DEPTH:-0}" != "0" ]]; then exit 0 fi diff --git a/plugin/lib/adapter.sh b/plugin/lib/adapter.sh index 989ae16..1847357 100644 --- a/plugin/lib/adapter.sh +++ b/plugin/lib/adapter.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash # lib/adapter.sh -- Shared CLI adapter core for DevSquad wrappers (D4). -# Sourced by gemini-/codex-/grok-wrapper.sh. Do not execute directly. +# Sourced by gemini-/codex-/grok-/claude-wrapper.sh. Do not execute directly. # # Contract (enforced by test/test_wrapper_contract.sh): # success: response on stdout, exit 0 @@ -32,6 +32,60 @@ set -euo pipefail _ADAPTER_LIB_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # shellcheck source=/dev/null source "${_ADAPTER_LIB_DIR}/model-catalog.sh" +# shellcheck source=/dev/null +source "${_ADAPTER_LIB_DIR}/../core/adapters/classification-policy.conf" + +# Terminate a bounded subprocess tree without requiring GNU timeout, setsid, +# or job-control process groups (all absent on a stock macOS Bash 3.2 host). +# Descendants are collected before the parent so an exiting parent cannot +# orphan children between discovery and signalling. +_adapter_snapshot_tree() { + local root_pid="$1" child + for child in $(pgrep -P "$root_pid" 2>/dev/null || true); do + _adapter_snapshot_tree "$child" + done + printf '%s\n' "$root_pid" +} + +_adapter_signal_snapshot() { + local snapshot_file="$1" signal="${2:-TERM}" pid + [[ -f "$snapshot_file" ]] || return 0 + while IFS= read -r pid; do + [[ -n "$pid" ]] && kill -"$signal" "$pid" 2>/dev/null || true + done < "$snapshot_file" +} + +# Bash's own job table tracks our directly-owned children without spawning +# ps/tr for every poll. Timer cancellation below never signals a numeric PID. +_adapter_job_running() { + local pid="$1" running + running=$(jobs -pr) + case $'\n'"$running"$'\n' in + *$'\n'"$pid"$'\n'*) return 0 ;; + *) return 1 ;; + esac +} + +_adapter_stop_timer() { + local timer_pid="${1:-}" control_file="${2:-}" + [[ -n "$timer_pid" ]] || return 0 + # Duplex open cannot block if the timer has already exited. Keep it open + # through wait so an early cancellation remains buffered until the timer + # opens its reader. The command-local descriptor restores the caller's FD9. + # No numeric PID signal is ever sent to a timer that may have exited/reaped. + { + printf 'cancel\n' >&9 + wait "$timer_pid" 2>/dev/null || true + } 9<>"$control_file" +} + +_adapter_cleanup_timer() { + _adapter_stop_timer "${timer_pid:-}" "${timer_control_file:-}" + if [[ -n "${timer_dir:-}" ]]; then + rm -f "$timer_dir/control" "$timer_dir/deadline" "$timer_dir/processes" + rmdir "$timer_dir" 2>/dev/null || true + fi +} # Resolve model: agent-specific (agent_models.) > # global (.preferences.) > "" (CLI default). @@ -98,6 +152,12 @@ _adapter_invoke() { local cli cli=$(_adapter_resolve_cli) + if [[ -n "${DEVSQUAD_TEST_ADAPTER_EXECUTABLE:-}" ]]; then + case "$DEVSQUAD_TEST_ADAPTER_EXECUTABLE" in + /*) [[ -x "$DEVSQUAD_TEST_ADAPTER_EXECUTABLE" ]] && cli="$DEVSQUAD_TEST_ADAPTER_EXECUTABLE" ;; + *) _adapter_fail "CLI_ERROR: test adapter executable must be absolute"; return 1 ;; + esac + fi if [[ -z "$cli" ]]; then _adapter_fail "CLI_ERROR: ${ADAPTER_MISSING_MSG}" return 1 @@ -109,14 +169,17 @@ _adapter_invoke() { _adapter_build_args "$final_prompt" "$model" "$timeout_secs" local timeout_cmd="" - if command -v timeout &>/dev/null; then timeout_cmd="timeout" - elif command -v gtimeout &>/dev/null; then timeout_cmd="gtimeout" + if [[ "${DEVSQUAD_FORCE_PORTABLE_TIMEOUT:-0}" != "1" ]] && command -v timeout &>/dev/null; then timeout_cmd="timeout" + elif [[ "${DEVSQUAD_FORCE_PORTABLE_TIMEOUT:-0}" != "1" ]] && command -v gtimeout &>/dev/null; then timeout_cmd="gtimeout" fi local stderr_file stdout_file stderr_file=$(mktemp) stdout_file=$(mktemp) - trap 'rm -f "${stderr_file:-}" "${stdout_file:-}"' EXIT + local timer_pid="" timer_dir="" timer_control_file="" + local previous_exit_trap + previous_exit_trap=$(builtin trap -p EXIT) + trap '_adapter_cleanup_timer; rm -f "${stderr_file:-}" "${stdout_file:-}"' EXIT local exit_code=0 if [[ -n "$timeout_cmd" ]]; then @@ -129,30 +192,79 @@ _adapter_invoke() { # Portable watchdog: every call is bounded even on hosts with no # timeout/gtimeout binary (observed live: an unauthenticated CLI # waiting on OAuth blocks forever) + timer_dir=$(mktemp -d) + timer_control_file="$timer_dir/control" + local timer_deadline_file="$timer_dir/deadline" + mkfifo -m 600 "$timer_control_file" + # Bash read's real deadline is independent of polling/inspection cost. + # Its private FIFO is cancellation authority: no timer-PID signalling, + # orphan sleep, inherited ignored TERM, or retained capture descriptors. + ( + trap - EXIT + local timer_message="" timer_status=0 + if IFS= read -r -t "$timeout_secs" timer_message <>"$timer_control_file"; then + [[ "$timer_message" == "cancel" ]] || printf 'error\n' >"$timer_deadline_file" + else + timer_status=$? + if [[ "$timer_status" -eq 1 || "$timer_status" -gt 128 ]]; then + printf 'timeout\n' >"$timer_deadline_file" + else + printf 'error\n' >"$timer_deadline_file" + fi + fi + ) /dev/null 2>&1 & + timer_pid=$! if [[ -n "${ADAPTER_STDIN_FILE:-}" ]]; then "$cli" "${ADAPTER_ARGS[@]}" <"$ADAPTER_STDIN_FILE" >"$stdout_file" 2>"$stderr_file" & else "$cli" "${ADAPTER_ARGS[@]}" >"$stdout_file" 2>"$stderr_file" & fi local cli_pid=$! - ( sleep "$timeout_secs"; kill "$cli_pid" 2>/dev/null ) & - local watchdog_pid=$! + local process_snapshot="$timer_dir/processes" timed_out="false" monitor_failed="false" + while _adapter_job_running "$cli_pid"; do + # Observe completion, not an in-progress marker write. The completed + # timer has published either its actual deadline or a monitor failure. + if ! _adapter_job_running "$timer_pid"; then + if [[ "$(cat "$timer_deadline_file" 2>/dev/null || true)" == "timeout" ]]; then + timed_out="true" + else + monitor_failed="true" + fi + _adapter_snapshot_tree "$cli_pid" > "$process_snapshot" + _adapter_signal_snapshot "$process_snapshot" TERM + sleep 0.1 + _adapter_signal_snapshot "$process_snapshot" KILL + break + fi + sleep 0.05 + done + _adapter_stop_timer "$timer_pid" "$timer_control_file" + timer_pid="" if wait "$cli_pid"; then exit_code=0 else exit_code=$? fi - kill "$watchdog_pid" 2>/dev/null || true - wait "$watchdog_pid" 2>/dev/null || true - # SIGTERM from the watchdog surfaces as 143 — normalize to timeout's 124 - if [[ $exit_code -eq 143 ]]; then + if [[ "$timed_out" == "true" ]]; then + _adapter_signal_snapshot "$process_snapshot" KILL exit_code=124 + elif [[ "$monitor_failed" == "true" ]]; then + exit_code=125 fi + _adapter_cleanup_timer + timer_dir="" + timer_control_file="" fi local stdout stderr_content stdout=$(cat "$stdout_file" 2>/dev/null) stderr_content=$(cat "$stderr_file" 2>/dev/null) + rm -f "$stderr_file" "$stdout_file" + # Cleanup uses invocation locals only while they are live. Restore the + # caller's shell-generated trap before any classification/return unwinds + # that scope, so same-named caller globals can never become cleanup targets. + builtin trap - EXIT + if [[ -n "$previous_exit_trap" ]]; then eval "$previous_exit_trap"; fi # CLI-specific auth signal (may appear on stdout with exit 0, e.g. grok's # sign-in banner) — checked before the success path @@ -160,7 +272,8 @@ _adapter_invoke() { _adapter_fail "AUTH_ERROR: ${agent} CLI is not authenticated. ${ADAPTER_AUTH_HINT}" elif [[ $exit_code -eq 0 ]]; then if [[ -z "$stdout" ]]; then - echo "WARNING: ${agent} returned empty response" >&2 + _adapter_fail "CLI_ERROR: ${agent} returned an empty response. ${ADAPTER_FALLBACK}" + return 1 fi update_agent_stats "$state_dir" "$agent" "true" record_usage "$agent" "$chars_in" "${#stdout}" @@ -169,9 +282,9 @@ _adapter_invoke() { return 0 elif [[ $exit_code -eq 124 ]]; then _adapter_fail "TIMEOUT: ${agent} did not respond within ${timeout_secs}s. ${ADAPTER_FALLBACK}" - elif echo "$stderr_content" | grep -qiE 'auth|401|403|ineligible|unauthorized'; then + elif echo "$stderr_content" | grep -qiE "$DEVSQUAD_AUTH_ERROR_PATTERN"; then _adapter_fail "AUTH_ERROR: ${agent} CLI authentication failed. ${ADAPTER_AUTH_HINT}" - elif echo "$stderr_content" | grep -qiE '429|rate.?limit|quota|resource.?exhausted|too many requests'; then + elif echo "$stderr_content" | grep -qiE "$DEVSQUAD_RATE_LIMIT_PATTERN"; then record_rate_limit "$state_dir" "$agent" _adapter_fail "RATE_LIMITED: ${agent} hit a rate limit. 2-minute cooldown started. ${ADAPTER_FALLBACK}" else diff --git a/plugin/lib/claude-wrapper.sh b/plugin/lib/claude-wrapper.sh new file mode 100644 index 0000000..2ef632d --- /dev/null +++ b/plugin/lib/claude-wrapper.sh @@ -0,0 +1,70 @@ +#!/usr/bin/env bash +# lib/claude-wrapper.sh -- Claude Code headless adapter configuration. +# Sourced by agent system prompts. Do not execute directly. +# Shared invocation core lives in lib/adapter.sh (D4 contract). +set -euo pipefail + +_CLAUDE_LIB_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +# shellcheck source=/dev/null +source "${_CLAUDE_LIB_DIR}/adapter.sh" + +_resolve_claude_model() { + _adapter_resolve_model "claude_model" +} + +_claude_configure_adapter() { + ADAPTER_AGENT="claude" + ADAPTER_PREF_MODEL_KEY="claude_model" + ADAPTER_AUTH_HINT="Run 'claude auth login' through the normal subscription flow, then retry." + ADAPTER_FALLBACK="Fallback: use a separately qualified DevSquad profile." + ADAPTER_MISSING_MSG="Claude Code CLI not installed. Install it through the official Claude Code setup." + ADAPTER_EXTRA_AUTH_RE='not logged in|please run /login|login required' + ADAPTER_STDIN_FILE="" + ADAPTER_EXTRA_CHARS_IN=0 + + _adapter_resolve_cli() { + if command -v claude &>/dev/null; then + echo "claude" + else + echo "" + fi + } + + _adapter_build_args() { + ADAPTER_ARGS=( + "--print" + "--output-format" "text" + "--safe-mode" + "--disable-slash-commands" + "--no-session-persistence" + "--strict-mcp-config" + "--mcp-config" '{"mcpServers":{}}' + "--no-chrome" + "--permission-mode" "plan" + "--tools" "Read,Glob,Grep" + ) + if [[ -n "$2" ]]; then + ADAPTER_ARGS+=("--model" "$2") + fi + ADAPTER_ARGS+=("$1") + } +} + +# Usage: invoke_claude "prompt" [word_limit] [timeout_secs] +# This compatibility entry point is deliberately read-only. The durable M5 +# implementer uses the manifest-driven workspace_write profile instead. +invoke_claude() { + local prompt="$1" + local word_limit="${2:-300}" + local timeout_secs="${3:-180}" + local final_prompt="$prompt" + + if [[ "$word_limit" -gt 0 ]] 2>/dev/null; then + if ! echo "$prompt" | grep -qiE 'under [0-9]+ (words|lines)|[0-9]+ (words|lines) max'; then + final_prompt="${prompt}. Under ${word_limit} words." + fi + fi + + _claude_configure_adapter + _adapter_invoke "$final_prompt" "$timeout_secs" +} diff --git a/plugin/lib/gemini-wrapper.sh b/plugin/lib/gemini-wrapper.sh index 46a1e1f..cc96fdd 100755 --- a/plugin/lib/gemini-wrapper.sh +++ b/plugin/lib/gemini-wrapper.sh @@ -104,9 +104,10 @@ invoke_gemini() { _adapter_invoke "$final_prompt" "$timeout_secs" } -# File-based invocation: concatenates file/dir contents and pipes them via -# stdin (bypasses Antigravity's workspace sandbox restriction on @file refs). -# Usage: invoke_gemini_with_files "@src/auth/ @src/models/user.ts" "prompt" [word_limit] [timeout_secs] +# File-based invocation. Context arguments are newline-delimited so paths with +# spaces remain intact. A single legacy whitespace-delimited argument remains +# accepted only when every token resolves, preserving existing callers. +# Usage: invoke_gemini_with_files $'@src/auth/\n@src/models/user.ts' "prompt" [word_limit] [timeout_secs] invoke_gemini_with_files() { local files_arg="$1" local prompt="$2" @@ -117,22 +118,78 @@ invoke_gemini_with_files() { # expand backslash escapes INSIDE file contents, corrupting code) local nl=$'\n' local file_content="" - local token path f - for token in $files_arg; do + local token path f manifest="$files_arg" resolved_parent + local project_root="${CLAUDE_PROJECT_DIR:-.}" + local max_bytes="${DEVSQUAD_CONTEXT_MAX_BYTES:-1048576}" + local max_file_bytes="${DEVSQUAD_CONTEXT_MAX_FILE_BYTES:-262144}" + local used_bytes=0 file_bytes header_bytes + project_root=$(cd "$project_root" 2>/dev/null && pwd -P) || { + echo "CONTEXT_OMITTED: project root is unavailable: ${project_root}" >&2 + return 1 + } + if [[ "$files_arg" != *$'\n'* ]]; then + local all_tokens_resolve="true" + for token in $files_arg; do + path="${token#@}" + [[ "$token" == @* && ( -f "$project_root/$path" || -d "$project_root/$path" ) ]] || all_tokens_resolve="false" + done + if [[ "$all_tokens_resolve" == "true" ]]; then + manifest=$(printf '%s\n' $files_arg) + fi + fi + while IFS= read -r token; do if [[ "$token" == @* ]]; then path="${token#@}" - if [[ -f "$path" ]]; then - file_content+="=== ${path} ===${nl}$(cat "$path")${nl}${nl}" - elif [[ -d "$path" ]]; then - while IFS= read -r f; do - file_content+="=== ${f} ===${nl}$(cat "$f")${nl}${nl}" - done < <(find "$path" -type f \( \ - -name "*.ts" -o -name "*.js" -o -name "*.sh" -o -name "*.py" \ - -o -name "*.go" -o -name "*.rs" -o -name "*.md" -o -name "*.json" \ - \) 2>/dev/null | sort) + case "$path" in + ""|/*|../*|*/../*|*/..) echo "CONTEXT_OMITTED: path escapes project scope: ${path}" >&2; continue ;; + esac + path="${path#./}" + if [[ -L "$project_root/$path" ]]; then + echo "CONTEXT_OMITTED: symlink input is not followed: ${path}" >&2 + continue + fi + + local matched="false" + while IFS= read -r -d '' f; do + matched="true" + [[ -L "$project_root/$f" ]] && { echo "CONTEXT_OMITTED: symlink input is not followed: ${f}" >&2; continue; } + resolved_parent=$(cd "$(dirname "$project_root/$f")" 2>/dev/null && pwd -P) || { + echo "CONTEXT_OMITTED: file parent is unavailable: ${f}" >&2; continue; + } + case "${resolved_parent}/" in + "${project_root}/"*) ;; + *) echo "CONTEXT_OMITTED: symlink ancestor escapes project scope: ${f}" >&2; continue ;; + esac + case "/$f" in + */.devsquad/*|*/.env|*/.env.*|*/credentials.json|*.pem|*.key) + echo "CONTEXT_OMITTED: sensitive or runtime path excluded: ${f}" >&2; continue ;; + esac + if git -C "$project_root" check-ignore --no-index -q -- "$f" 2>/dev/null; then + echo "CONTEXT_OMITTED: ignored path excluded: ${f}" >&2 + continue + fi + if ! grep -Iq . "$project_root/$f" 2>/dev/null && [[ -s "$project_root/$f" ]]; then + echo "CONTEXT_OMITTED: binary file excluded: ${f}" >&2 + continue + fi + file_bytes=$(wc -c < "$project_root/$f" | tr -d ' ') + if [[ "$file_bytes" -gt "$max_file_bytes" ]]; then + echo "CONTEXT_OMITTED: file exceeds ${max_file_bytes} byte limit: ${f} (${file_bytes} bytes)" >&2 + continue + fi + header_bytes=$(( ${#f} + 10 )) + if [[ $(( used_bytes + file_bytes + header_bytes )) -gt "$max_bytes" ]]; then + echo "CONTEXT_OMITTED: total context exceeds ${max_bytes} byte limit before: ${f}" >&2 + continue + fi + file_content+="=== ${f} ===${nl}$(cat "$project_root/$f")${nl}${nl}" + used_bytes=$(( used_bytes + file_bytes + header_bytes )) + done < <(git -C "$project_root" ls-files -z -- ":(literal)$path" 2>/dev/null) + if [[ "$matched" == "false" ]]; then + echo "CONTEXT_OMITTED: no tracked files in scope: ${path}" >&2 fi fi - done + done <<< "$manifest" local final_prompt final_prompt=$(_gemini_final_prompt "$prompt" "$word_limit") diff --git a/plugin/lib/model-catalog.sh b/plugin/lib/model-catalog.sh index ee45235..f9550f3 100644 --- a/plugin/lib/model-catalog.sh +++ b/plugin/lib/model-catalog.sh @@ -61,15 +61,20 @@ refresh_model_catalog() { if [[ -n "$k_models" ]]; then k_status="ok"; else k_status="error"; fi fi + local old_json='{}' + [[ -f "$CATALOG_FILE" ]] && old_json=$(cat "$CATALOG_FILE" 2>/dev/null || echo '{}') local new_json new_json=$(jq -n \ --arg ts "$ts" \ --arg gs "$g_status" --arg g "$g_models" \ --arg ks "$k_status" --arg k "$k_models" \ + --argjson old "$old_json" \ '{ fetched_at: $ts, - gemini: { status: $gs, models: ($g | split("\n") | map(select(length > 0))) }, - grok: { status: $ks, models: ($k | split("\n") | map(select(length > 0))) }, + gemini: (if $gs == "ok" then { status: "ok", models: ($g | split("\n") | map(select(length > 0))), last_good_at: $ts, last_refresh_error: null } + else (($old.gemini // {models: []}) + {status: ($old.gemini.status // "unavailable"), last_refresh_error: $gs}) end), + grok: (if $ks == "ok" then { status: "ok", models: ($k | split("\n") | map(select(length > 0))), last_good_at: $ts, last_refresh_error: null } + else (($old.grok // {models: []}) + {status: ($old.grok.status // "unavailable"), last_refresh_error: $ks}) end), codex: { status: "unlistable", models: [] } }') @@ -104,11 +109,9 @@ refresh_model_catalog() { } # Map a tier to the best available model for a CLI, from the cached catalog. -# tier:fast -> cheap/fast family (flash|fast|mini|lite|haiku), -# highest version, prefer (Medium) then (Low) -# tier:frontier -> non-fast family matching pro|opus|max|ultra (fallback: -# any non-fast, then anything), highest version, prefer -# (High) then Thinking +# Structured entries declare family and compatible tiers. Legacy string entries +# are constrained to the harness family before applying their old local ranking; +# version numbers are never compared across model families. # Echoes "" when unresolvable — callers fall back to the CLI default. resolve_model_tier() { local cli="$1" tier="$2" @@ -116,19 +119,27 @@ resolve_model_tier() { command -v jq &>/dev/null || { echo ""; return 0; } local models - models=$(jq -r --arg c "$cli" '.[$c].models // [] | .[]' "$CATALOG_FILE" 2>/dev/null || true) + models=$(jq -r --arg c "$cli" --arg t "$tier" ' + .[$c].models // [] | .[] | + if type == "object" then + select((.family == $c) and ((.compatibility.tiers // []) | index($t))) | .id + else + select((ascii_downcase | startswith($c + "-") or startswith($c + " "))) | . + end' "$CATALOG_FILE" 2>/dev/null || true) [[ -n "$models" ]] || { echo ""; return 0; } local pool - if [[ "$tier" == "fast" ]]; then - pool=$(printf '%s\n' "$models" | grep -iE 'flash|fast|mini|lite|haiku' || true) - [[ -n "$pool" ]] || pool="$models" + if jq -e --arg c "$cli" '.[$c].models // [] | any(type == "object")' "$CATALOG_FILE" >/dev/null 2>&1; then + pool="$models" + elif [[ "$tier" == "fast" ]]; then + pool=$(printf '%s\n' "$models" | grep -iE '(^|[- (])(flash|fast|mini|lite|haiku)([- )]|$)' || true) + [[ -n "$pool" ]] || { echo ""; return 0; } else - pool=$(printf '%s\n' "$models" | grep -ivE 'flash|fast|mini|lite|haiku' | grep -iE 'pro|opus|max|ultra' || true) + pool=$(printf '%s\n' "$models" | grep -ivE '(^|[- (])(flash|fast|mini|lite|haiku)([- )]|$)' | grep -iE '(^|[- (])(pro|opus|max|ultra)([- )]|$)' || true) if [[ -z "$pool" ]]; then - pool=$(printf '%s\n' "$models" | grep -ivE 'flash|fast|mini|lite|haiku' || true) + pool=$(printf '%s\n' "$models" | grep -ivE '(^|[- (])(flash|fast|mini|lite|haiku)([- )]|$)' || true) fi - [[ -n "$pool" ]] || pool="$models" + [[ -n "$pool" ]] || { echo ""; return 0; } fi printf '%s\n' "$pool" | awk -v tier="$tier" ' diff --git a/scripts/generate-core-reference.py b/scripts/generate-core-reference.py new file mode 100755 index 0000000..5a99ada --- /dev/null +++ b/scripts/generate-core-reference.py @@ -0,0 +1,127 @@ +#!/usr/bin/env python3 +"""Generate the checked-in DevSquad CLI and schema reference.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +from pathlib import Path +import sys + + +ROOT = Path(__file__).resolve().parents[1] +CORE = ROOT / "plugin/core" +OUTPUT = ROOT / "docs/generated/core-reference.md" +sys.path.insert(0, str(CORE / "src")) + +from devsquad.cli import parser as build_parser # noqa: E402 + + +def command_usages() -> list[str]: + result: list[str] = [] + + def visit(command_parser: argparse.ArgumentParser) -> None: + for action in command_parser._actions: + if not isinstance(action, argparse._SubParsersAction): + continue + for name in sorted(action.choices): + child = action.choices[name] + usage = child.format_usage().strip().removeprefix("usage: ") + result.append(" ".join(usage.split())) + visit(child) + + visit(build_parser()) + return result + + +def schema_rows() -> list[tuple[str, str, str, str]]: + rows = [] + for path in sorted((CORE / "schemas").glob("*.schema.json")): + raw = path.read_bytes() + value = json.loads(raw) + required = value.get("required", []) + rows.append(( + path.name, + value.get("$id", "—"), + ", ".join(f"`{field}`" for field in required) or "—", + hashlib.sha256(raw).hexdigest(), + )) + return rows + + +def render() -> str: + task_example = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + lines = [ + "", + "# DevSquad core command and schema reference", + "", + "Regenerate with `python3 scripts/generate-core-reference.py`; verify with", + "`python3 scripts/generate-core-reference.py --check`.", + "", + "## Command forms", + "", + ] + lines.extend(f"- `{usage}`" for usage in command_usages()) + lines.extend([ + "", + "## Common command examples", + "", + "```bash", + "squad --version", + "squad doctor --project-dir \"$PWD\" --json", + "squad start --task-file task.json --idempotency-key issue-123 --json", + "squad status RUN_ID --json", + "squad events RUN_ID --after 0 --limit 100 --json", + "squad result RUN_ID --json", + "squad cancel RUN_ID --json", + "squad resume RUN_ID --recovery-file recovery.json --json", + "squad setup --host codex --dry-run --json", + "```", + "", + "`RUN_ID`, task paths and recovery evidence are operator-supplied values; the", + "runtime never infers them from chat history.", + "", + "## Packaged JSON schemas", + "", + "| File | Identifier | Required top-level fields | SHA-256 |", + "|---|---|---|---|", + ]) + for name, identifier, required, digest in schema_rows(): + lines.append(f"| `{name}` | `{identifier}` | {required} | `{digest}` |") + lines.extend([ + "", + "## Task-shape example", + "", + "This generated copy demonstrates the strict v1 task shape. Replace the", + "fixture repository, refs, routing files and checks before starting a run.", + "", + "```json", + json.dumps(task_example, indent=2, sort_keys=True), + "```", + "", + ]) + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + arguments = argparse.ArgumentParser() + arguments.add_argument("--check", action="store_true") + parsed = arguments.parse_args(argv) + content = render() + if parsed.check: + if not OUTPUT.is_file() or OUTPUT.read_text() != content: + print(f"generated reference is stale: {OUTPUT}", file=sys.stderr) + return 1 + print(f"generated reference is current: {OUTPUT}") + return 0 + OUTPUT.parent.mkdir(parents=True, exist_ok=True) + OUTPUT.write_text(content) + print(OUTPUT) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/install-core.sh b/scripts/install-core.sh new file mode 100755 index 0000000..ed18173 --- /dev/null +++ b/scripts/install-core.sh @@ -0,0 +1,524 @@ +#!/usr/bin/env bash +# Install DevSquad's dependency-free core into an immutable local release. +# Bash 3.2 compatible. The default path performs no network access. +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd -P)" +REPO_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" +SOURCE_CORE="${DEVSQUAD_SOURCE_CORE:-${REPO_ROOT}/plugin/core}" +INSTALL_ROOT="${DEVSQUAD_INSTALL_ROOT:-${HOME:?HOME is required}/.devsquad}" +BIN_DIR="${DEVSQUAD_BIN_DIR:-${HOME:?HOME is required}/.local/bin}" +PYTHON_REQUEST="${DEVSQUAD_PYTHON:-python3}" +MCP_WHEELHOUSE="${DEVSQUAD_MCP_WHEELHOUSE:-}" +WITH_MCP=0 +STATUS_ONLY=0 +JSON_OUTPUT=0 + +usage() { + cat <<'EOF' +Usage: scripts/install-core.sh [options] + +Install the local DevSquad core without requiring Claude or network access. + +Options: + --source-core PATH Core source directory (default: plugin/core) + --install-root PATH Release root (default: ~/.devsquad) + --bin-dir PATH Stable launcher directory (default: ~/.local/bin) + --python PATH Python 3.11+ used for the isolated runtime + --with-mcp Install the optional MCP environment offline + --mcp-wheelhouse PATH Directory containing every locked MCP wheel + --status Report source/plugin/installed drift; do not install + --json Emit one JSON object + -h, --help Show this help + +The MCP option never downloads packages. It requires --mcp-wheelhouse (or +DEVSQUAD_MCP_WHEELHOUSE) and installs requirements-mcp.lock with --no-index. +EOF +} + +fail() { + printf 'devsquad install: %s\n' "$*" >&2 + exit 1 +} + +while [ "$#" -gt 0 ]; do + case "$1" in + --source-core) + [ "$#" -ge 2 ] || fail "--source-core requires a path" + SOURCE_CORE="$2"; shift 2 ;; + --install-root) + [ "$#" -ge 2 ] || fail "--install-root requires a path" + INSTALL_ROOT="$2"; shift 2 ;; + --bin-dir) + [ "$#" -ge 2 ] || fail "--bin-dir requires a path" + BIN_DIR="$2"; shift 2 ;; + --python) + [ "$#" -ge 2 ] || fail "--python requires a path" + PYTHON_REQUEST="$2"; shift 2 ;; + --with-mcp) WITH_MCP=1; shift ;; + --mcp-wheelhouse) + [ "$#" -ge 2 ] || fail "--mcp-wheelhouse requires a path" + MCP_WHEELHOUSE="$2"; shift 2 ;; + --status) STATUS_ONLY=1; shift ;; + --json) JSON_OUTPUT=1; shift ;; + -h|--help) usage; exit 0 ;; + *) fail "unknown option: $1" ;; + esac +done + +case "$INSTALL_ROOT" in /*) ;; *) fail "install root must be absolute" ;; esac +case "$BIN_DIR" in /*) ;; *) fail "bin directory must be absolute" ;; esac +[ -d "$SOURCE_CORE/src/devsquad" ] || fail "core source is incomplete: $SOURCE_CORE" +[ -f "$SOURCE_CORE/pyproject.toml" ] || fail "core source has no pyproject.toml: $SOURCE_CORE" + +if [ -x "$PYTHON_REQUEST" ]; then + PYTHON="$PYTHON_REQUEST" +else + PYTHON="$(command -v "$PYTHON_REQUEST" 2>/dev/null || true)" +fi +[ -n "$PYTHON" ] && [ -x "$PYTHON" ] || fail "Python executable not found: $PYTHON_REQUEST" + +PYTHON_INFO="$("$PYTHON" - <<'PY' +import json +import platform +import re +import sys + +if sys.version_info < (3, 11): + raise SystemExit("DevSquad requires Python 3.11+") +cache_tag = sys.implementation.cache_tag or "python" +safe_tag = re.sub(r"[^A-Za-z0-9._-]+", "-", cache_tag) +print(json.dumps({ + "executable": sys.executable, + "version": platform.python_version(), + "version_id": "%d%d%d" % sys.version_info[:3], + "cache_tag": safe_tag, +}, sort_keys=True)) +PY +)" || fail "Python 3.11+ is required" + +core_info() { + "$PYTHON" - "$1" <<'PY' +import hashlib +import json +from pathlib import Path +import re +import sys +import tomllib + +root = Path(sys.argv[1]).resolve(strict=True) +excluded_names = {".DS_Store", ".pytest_cache", "build", "dist", "__pycache__"} +members = [] +for path in root.rglob("*"): + relative = path.relative_to(root) + if any(part in excluded_names or part.endswith(".egg-info") for part in relative.parts): + continue + if path.is_symlink(): + raise SystemExit(f"core source may not contain symlinks: {relative}") + if path.is_file() and path.suffix != ".pyc": + members.append((relative, path)) +digest = hashlib.sha256() +for relative, path in sorted(members, key=lambda item: item[0].as_posix()): + digest.update(relative.as_posix().encode("utf-8") + b"\0") + digest.update(path.read_bytes()) +configuration = tomllib.loads((root / "pyproject.toml").read_text()) +version = configuration["project"]["version"] +init_text = (root / "src/devsquad/__init__.py").read_text() +match = re.search(r'^__version__\s*=\s*["\']([^"\']+)["\']', init_text, re.M) +if not match or match.group(1) != version: + raise SystemExit("pyproject and package versions differ") +if not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._-]*", version): + raise SystemExit("core version is not release-path safe") +print(json.dumps({ + "digest": digest.hexdigest(), + "path": str(root), + "version": version, +}, sort_keys=True)) +PY +} + +SOURCE_INFO="$(core_info "$SOURCE_CORE")" || fail "cannot fingerprint core source" +PLUGIN_CORE="${REPO_ROOT}/plugin/core" +if [ -d "$PLUGIN_CORE/src/devsquad" ]; then + PLUGIN_INFO="$(core_info "$PLUGIN_CORE")" || fail "cannot fingerprint plugin core" +else + PLUGIN_INFO='null' +fi + +emit_report() { + "$PYTHON" - "$1" "$SOURCE_INFO" "$PLUGIN_INFO" "$INSTALL_ROOT" "$BIN_DIR" "$PYTHON_INFO" "$2" "$3" <<'PY' +import hashlib +import json +from pathlib import Path +import sys + +mode, source_raw, plugin_raw, install_raw, bin_raw, python_raw, changed_raw, error = sys.argv[1:] +source = json.loads(source_raw) +plugin = json.loads(plugin_raw) +python = json.loads(python_raw) +install_root = Path(install_raw) +current = install_root / "current" +launcher = Path(bin_raw) / "squad" +installed = None +installed_payload_digest = None +current_target = None +manifest_error = None + +def payload_digest(root): + excluded_names = {".DS_Store", ".pytest_cache", "build", "dist", "__pycache__"} + members = [] + for path in root.rglob("*"): + relative = path.relative_to(root) + if any(part in excluded_names or part.endswith(".egg-info") for part in relative.parts): + continue + if path.is_symlink(): + raise OSError(f"installed core contains a symlink: {relative}") + if path.is_file() and path.suffix != ".pyc": + members.append((relative, path)) + digest = hashlib.sha256() + for relative, path in sorted(members, key=lambda item: item[0].as_posix()): + digest.update(relative.as_posix().encode("utf-8") + b"\0") + digest.update(path.read_bytes()) + return digest.hexdigest() + +try: + if current.is_symlink(): + current_target = str(current.resolve(strict=True)) + installed = json.loads((current / "release.json").read_text()) + installed_payload_digest = payload_digest(current / "core") + if installed.get("source_digest") != installed_payload_digest: + manifest_error = "installed release payload differs from its manifest" + elif current.exists(): + manifest_error = "current selector is not a symlink" +except (OSError, UnicodeError, json.JSONDecodeError) as exc: + manifest_error = f"cannot read installed release: {type(exc).__name__}" +source_digest = source["digest"] +plugin_digest = plugin["digest"] if plugin else None +installed_digest = installed_payload_digest +report = { + "schema_version": 1, + "mode": mode, + "changed": changed_raw == "1", + "source": source, + "plugin": plugin, + "installed": installed, + "installed_payload_digest": installed_payload_digest, + "installed_manifest_matches": bool( + installed is not None + and installed.get("source_digest") == installed_payload_digest + ), + "current_target": current_target, + "launcher": str(launcher), + "launcher_ready": launcher.is_file() and launcher.stat().st_mode & 0o111 != 0, + "python": python, + "drift": { + "source_plugin": plugin_digest is not None and source_digest != plugin_digest, + "source_installed": installed_digest is None or source_digest != installed_digest, + "plugin_installed": plugin_digest is None or installed_digest is None or plugin_digest != installed_digest, + }, + "error": error or manifest_error, +} +print(json.dumps(report, sort_keys=True, separators=(",", ":"))) +PY +} + +if [ "$STATUS_ONLY" -eq 1 ]; then + REPORT="$(emit_report status 0 '')" + if [ "$JSON_OUTPUT" -eq 1 ]; then + printf '%s\n' "$REPORT" + else + "$PYTHON" - "$REPORT" <<'PY' +import json, sys +r = json.loads(sys.argv[1]) +state = "not installed" if r["installed"] is None else r["installed"]["release_id"] +print(f"DevSquad standalone: {state}") +print(f" launcher: {r['launcher']} ({'ready' if r['launcher_ready'] else 'missing'})") +print(" drift: source/plugin={source_plugin} source/installed={source_installed} plugin/installed={plugin_installed}".format(**r["drift"])) +if r["error"]: + print(f" error: {r['error']}") +PY + fi + exit 0 +fi + +if [ "$WITH_MCP" -eq 1 ]; then + [ -n "$MCP_WHEELHOUSE" ] || fail "--with-mcp requires --mcp-wheelhouse" + [ -d "$MCP_WHEELHOUSE" ] || fail "MCP wheelhouse is not a directory: $MCP_WHEELHOUSE" +fi + +mkdir -p "$INSTALL_ROOT/releases" "$BIN_DIR" +INSTALL_ROOT="$(cd "$INSTALL_ROOT" && pwd -P)" +BIN_DIR="$(cd "$BIN_DIR" && pwd -P)" +SOURCE_CORE="$(cd "$SOURCE_CORE" && pwd -P)" + +LOCK_DIR="$INSTALL_ROOT/.install.lock" +if ! mkdir "$LOCK_DIR" 2>/dev/null; then + fail "another install is active or left $LOCK_DIR behind" +fi +TEMP_DIR="" +cleanup() { + if [ -n "$TEMP_DIR" ] && [ -d "$TEMP_DIR" ]; then + rm -rf "$TEMP_DIR" + fi + rmdir "$LOCK_DIR" 2>/dev/null || true +} +trap cleanup EXIT HUP INT TERM + +SOURCE_DIGEST="$($PYTHON -c 'import json,sys; print(json.loads(sys.argv[1])["digest"])' "$SOURCE_INFO")" +SOURCE_VERSION="$($PYTHON -c 'import json,sys; print(json.loads(sys.argv[1])["version"])' "$SOURCE_INFO")" +PYTHON_VERSION_ID="$($PYTHON -c 'import json,sys; print(json.loads(sys.argv[1])["version_id"])' "$PYTHON_INFO")" +PYTHON_CACHE_TAG="$($PYTHON -c 'import json,sys; print(json.loads(sys.argv[1])["cache_tag"])' "$PYTHON_INFO")" +FLAVOR="core" +MCP_ENV_DIGEST="" +if [ "$WITH_MCP" -eq 1 ]; then + MCP_ENV_DIGEST="$($PYTHON - "$SOURCE_CORE/requirements-mcp.lock" "$MCP_WHEELHOUSE" <<'PY' +import hashlib +from pathlib import Path +import sys + +digest = hashlib.sha256() +for root in (Path(sys.argv[1]), Path(sys.argv[2])): + paths = [root] if root.is_file() else sorted(p for p in root.rglob("*") if p.is_file()) + for path in paths: + digest.update(path.name.encode("utf-8") + b"\0" + path.read_bytes()) +print(digest.hexdigest()) +PY +)" + FLAVOR="mcp-${MCP_ENV_DIGEST%${MCP_ENV_DIGEST#????????????}}" +fi +SHORT_DIGEST="${SOURCE_DIGEST%${SOURCE_DIGEST#????????????}}" +RELEASE_ID="${SOURCE_VERSION}-py${PYTHON_VERSION_ID}-${SHORT_DIGEST}-${FLAVOR}" +RELEASE_DIR="$INSTALL_ROOT/releases/$RELEASE_ID" +RELEASE_CREATED=0 + +if [ -e "$RELEASE_DIR" ]; then + [ -d "$RELEASE_DIR" ] && [ -f "$RELEASE_DIR/release.json" ] || fail "release path is incomplete: $RELEASE_DIR" + RELEASE_CORE_INFO="$(core_info "$RELEASE_DIR/core")" || fail "existing release payload is unreadable" + RELEASE_CORE_DIGEST="$($PYTHON -c 'import json,sys; print(json.loads(sys.argv[1])["digest"])' "$RELEASE_CORE_INFO")" + [ "$RELEASE_CORE_DIGEST" = "$SOURCE_DIGEST" ] || fail "existing release payload digest differs: $RELEASE_DIR" + "$PYTHON" - "$RELEASE_DIR/release.json" "$RELEASE_ID" "$SOURCE_DIGEST" "$PYTHON_CACHE_TAG" "$WITH_MCP" "$MCP_ENV_DIGEST" <<'PY' +import json +from pathlib import Path +import sys + +path, release_id, digest, cache_tag, with_mcp, mcp_digest = sys.argv[1:] +value = json.loads(Path(path).read_text()) +expected = { + "release_id": release_id, + "source_digest": digest, + "python_cache_tag": cache_tag, + "mcp": with_mcp == "1", + "mcp_environment_digest": mcp_digest or None, +} +for key, item in expected.items(): + if value.get(key) != item: + raise SystemExit(f"existing release manifest mismatch: {key}") +venv_python = Path(path).parent / "venv/bin/python" +if not venv_python.is_file(): + raise SystemExit("existing release has no runtime Python") +PY +else + TEMP_DIR="$(mktemp -d "$INSTALL_ROOT/.install.XXXXXX")" + BUILD_RELEASE="$TEMP_DIR/release" + mkdir -p "$BUILD_RELEASE" + "$PYTHON" -m venv "$BUILD_RELEASE/venv" + VENV_PYTHON="$BUILD_RELEASE/venv/bin/python" + [ -x "$VENV_PYTHON" ] || fail "virtual environment did not provide bin/python" + PURELIB="$($VENV_PYTHON -c 'import sysconfig; print(sysconfig.get_path("purelib"))')" + "$VENV_PYTHON" - "$SOURCE_CORE" "$BUILD_RELEASE/core" "$PURELIB" "$SOURCE_VERSION" <<'PY' +from pathlib import Path +import os +import shutil +import sys + +source = Path(sys.argv[1]) +installed_core = Path(sys.argv[2]) +purelib = Path(sys.argv[3]) +version = sys.argv[4] +ignore = shutil.ignore_patterns( + "__pycache__", "*.pyc", ".DS_Store", ".pytest_cache", "build", "dist", "*.egg-info", +) +shutil.copytree(source, installed_core, ignore=ignore) +relative_source = os.path.relpath(installed_core / "src", purelib) +(purelib / "devsquad-core.pth").write_text(relative_source + "\n") +dist = purelib / f"devsquad_core-{version}.dist-info" +dist.mkdir() +(dist / "METADATA").write_text( + "Metadata-Version: 2.1\nName: devsquad-core\nVersion: " + version + "\n" +) +(dist / "WHEEL").write_text( + "Wheel-Version: 1.0\nGenerator: devsquad-install-core\nRoot-Is-Purelib: true\nTag: py3-none-any\n" +) +(dist / "entry_points.txt").write_text("[console_scripts]\nsquad = devsquad.cli:main\n") +(dist / "RECORD").write_text("") +PY + if [ "$WITH_MCP" -eq 1 ]; then + "$VENV_PYTHON" -m pip install --quiet --disable-pip-version-check --no-index \ + --only-binary=:all: --find-links "$MCP_WHEELHOUSE" \ + -r "$SOURCE_CORE/requirements-mcp.lock" + "$VENV_PYTHON" -m pip check >/dev/null + "$VENV_PYTHON" - <<'PY' +from importlib import metadata +import mcp +assert metadata.version("mcp") == "2.2.0" +PY + fi + "$VENV_PYTHON" -P "$BUILD_RELEASE/core/bin/squad" --version >/dev/null + "$VENV_PYTHON" -P - <<'PY' +from devsquad.integrations import load_integrations +assert {item.id for item in load_integrations()} == {"codex", "claude-code", "antigravity", "grok"} +PY + "$PYTHON" - "$BUILD_RELEASE/release.json" "$RELEASE_ID" "$SOURCE_VERSION" "$SOURCE_DIGEST" "$PYTHON_INFO" "$WITH_MCP" "$MCP_ENV_DIGEST" <<'PY' +import json +from pathlib import Path +import sys + +path, release_id, version, digest, python_raw, with_mcp, mcp_digest = sys.argv[1:] +python = json.loads(python_raw) +value = { + "schema_version": 1, + "release_id": release_id, + "version": version, + "source_digest": digest, + "python_version": python["version"], + "python_cache_tag": python["cache_tag"], + "mcp": with_mcp == "1", + "mcp_environment_digest": mcp_digest or None, +} +Path(path).write_text(json.dumps(value, sort_keys=True, separators=(",", ":")) + "\n") +PY + mv "$BUILD_RELEASE" "$RELEASE_DIR" + RELEASE_CREATED=1 +fi + +# Never advance the ledger while the old release is still selected. Check +# upgrade readiness and replace the selector under one ledger lock; actual +# migration is lazy, through the new release's guarded Store constructor. +UPGRADE_RUNTIME="${DEVSQUAD_RUNTIME_DIR:-${INSTALL_ROOT}/runtime}" +case "$UPGRADE_RUNTIME" in /*) ;; *) fail "runtime directory must be absolute" ;; esac +activate_current() { + "$RELEASE_DIR/venv/bin/python" -P - "$1" "$INSTALL_ROOT/current" "$UPGRADE_RUNTIME" "$REPO_ROOT/plugin/core/src/devsquad/release_activation.py" <<'PY' || fail "runtime upgrade deferred; previous current selector and launcher are unchanged" +from pathlib import Path +import runpy +import sys +from devsquad.store import SUPPORTED_SCHEMA_VERSION + +temporary, selector, runtime, helper = map(Path, sys.argv[1:]) +activate_release = runpy.run_path(str(helper))["activate_release"] +try: + activate_release(temporary, selector, runtime, supported_schema_version=SUPPORTED_SCHEMA_VERSION) +except Exception as exc: + if temporary.is_symlink(): + temporary.unlink() + print(str(exc), file=sys.stderr) + raise SystemExit(1) +PY +} + +CURRENT_CHANGED=0 +CURRENT_TARGET="releases/$RELEASE_ID" +if [ -L "$INSTALL_ROOT/current" ] && [ "$(readlink "$INSTALL_ROOT/current")" = "$CURRENT_TARGET" ]; then + : +elif [ -e "$INSTALL_ROOT/current" ] || [ -L "$INSTALL_ROOT/current" ]; then + [ -L "$INSTALL_ROOT/current" ] || fail "current selector is not a symlink: $INSTALL_ROOT/current" + ln -s "$CURRENT_TARGET" "$INSTALL_ROOT/.current.$$" + activate_current "$INSTALL_ROOT/.current.$$" + CURRENT_CHANGED=1 +else + ln -s "$CURRENT_TARGET" "$INSTALL_ROOT/.current.$$" + activate_current "$INSTALL_ROOT/.current.$$" + CURRENT_CHANGED=1 +fi + +LAUNCHER="$BIN_DIR/squad" +if { [ -e "$LAUNCHER" ] || [ -L "$LAUNCHER" ]; } \ + && ! grep -q '^# managed-by: devsquad-install-core-v1$' "$LAUNCHER" 2>/dev/null; then + if [ -L "$LAUNCHER" ]; then + "$PYTHON" - "$LAUNCHER" "$INSTALL_ROOT/releases" <<'PY' || fail "refusing to replace unmanaged launcher: $LAUNCHER" +from pathlib import Path +import sys + +launcher = Path(sys.argv[1]) +releases = Path(sys.argv[2]).resolve(strict=True) +target = launcher.resolve(strict=True) +try: + relative = target.relative_to(releases) +except ValueError as exc: + raise SystemExit("legacy launcher target is outside the release root") from exc +if len(relative.parts) != 4 or relative.parts[-2:] != ("bin", "squad") or relative.parts[-3] != "venv": + raise SystemExit("legacy launcher target is not a DevSquad release command") +PY + else + fail "refusing to replace unmanaged launcher: $LAUNCHER" + fi +fi +LAUNCHER_TEMP="$BIN_DIR/.squad.$$" +"$PYTHON" - "$LAUNCHER_TEMP" "$INSTALL_ROOT" <<'PY' +from pathlib import Path +import shlex +import sys + +path = Path(sys.argv[1]) +python = Path(sys.argv[2]) / "current/venv/bin/python" +entry = Path(sys.argv[2]) / "current/core/bin/squad" +path.write_text( + "#!/bin/sh\n" + "# managed-by: devsquad-install-core-v1\n" + "exec " + shlex.quote(str(python)) + " -P " + shlex.quote(str(entry)) + " \"$@\"\n" +) +path.chmod(0o755) +PY +LAUNCHER_CHANGED=0 +if [ -f "$LAUNCHER" ] && cmp -s "$LAUNCHER_TEMP" "$LAUNCHER"; then + rm -f "$LAUNCHER_TEMP" +else + mv -f "$LAUNCHER_TEMP" "$LAUNCHER" + LAUNCHER_CHANGED=1 +fi + +CHANGED=0 +if [ "$RELEASE_CREATED" -eq 1 ] || [ "$CURRENT_CHANGED" -eq 1 ] || [ "$LAUNCHER_CHANGED" -eq 1 ]; then + CHANGED=1 +fi +REPORT="$(emit_report install "$CHANGED" '')" +STATE_TEMP="$INSTALL_ROOT/.install-state.$$" +"$PYTHON" - "$STATE_TEMP" "$REPORT" <<'PY' +import json +from pathlib import Path +import sys + +path = Path(sys.argv[1]) +report = json.loads(sys.argv[2]) +state = { + "schema_version": report["schema_version"], + "source": report["source"], + "plugin": report["plugin"], + "installed": report["installed"], + "installed_payload_digest": report["installed_payload_digest"], + "installed_manifest_matches": report["installed_manifest_matches"], + "current_target": report["current_target"], + "launcher": report["launcher"], + "python": report["python"], + "drift": report["drift"], +} +path.write_text(json.dumps(state, sort_keys=True, separators=(",", ":")) + "\n") +PY +if [ -f "$INSTALL_ROOT/install-state.json" ] && cmp -s "$STATE_TEMP" "$INSTALL_ROOT/install-state.json"; then + rm -f "$STATE_TEMP" +else + mv -f "$STATE_TEMP" "$INSTALL_ROOT/install-state.json" +fi + +if [ "$JSON_OUTPUT" -eq 1 ]; then + printf '%s\n' "$REPORT" +else + "$PYTHON" - "$REPORT" <<'PY' +import json, sys +r = json.loads(sys.argv[1]) +print(f"DevSquad {r['installed']['version']} installed") +print(f" release: {r['current_target']}") +print(f" launcher: {r['launcher']}") +print(f" changed: {str(r['changed']).lower()}") +print(" drift: source/plugin={source_plugin} source/installed={source_installed} plugin/installed={plugin_installed}".format(**r["drift"])) +PY +fi diff --git a/scripts/run-core-tests.py b/scripts/run-core-tests.py new file mode 100644 index 0000000..0114167 --- /dev/null +++ b/scripts/run-core-tests.py @@ -0,0 +1,39 @@ +#!/usr/bin/env python3 +"""Spawn-safe core gate with monotonic timing and SQLite finalizer diagnostics.""" + +from datetime import datetime, timezone +import gc +import json +from pathlib import Path +import sys +import time +import unittest + + +def main(): + root = Path(__file__).resolve().parents[1] + sys.path[:0] = [str(root / "plugin/core/src"), str(root / "test/core")] + unraisable = [] + original_hook = sys.unraisablehook + def record_unraisable(event): + unraisable.append(type(event.exc_value).__name__) + original_hook(event) + sys.unraisablehook = record_unraisable + started, utc = time.monotonic(), datetime.now(timezone.utc) + suite = unittest.defaultTestLoader.discover(str(root / "test/core")) + result = unittest.TextTestRunner( + verbosity=2 if "--verbose" in sys.argv else 1, + failfast="--failfast" in sys.argv, + ).run(suite) + gc.collect() + print(json.dumps({ + "tests": result.testsRun, "errors": len(result.errors), "failures": len(result.failures), + "skips": len(result.skipped), "unraisable": unraisable, + "monotonic_seconds": time.monotonic() - started, + "utc_seconds": (datetime.now(timezone.utc) - utc).total_seconds(), + })) + return 0 if result.wasSuccessful() and not unraisable else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/test/core/devsquad_test_fixtures.py b/test/core/devsquad_test_fixtures.py new file mode 100644 index 0000000..54c16ee --- /dev/null +++ b/test/core/devsquad_test_fixtures.py @@ -0,0 +1,44 @@ +import json + + +def branch_review_routing_documents(): + profiles = { + "schema_version": 1, + "profiles": [{ + "id": "fixture-reviewer", + "harness": "fixture", + "model_family": "fixture-family-a", + "model_id": "fixture-review-model", + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "fixture-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["tracked-fixture"], + }], + "bindings": { + "review.deep": {"profile_id": "fixture-reviewer", "version": 1}, + }, + } + policy = { + "schema_version": 1, + "id": "fixture-policy", + "version": 1, + "roles": {"reviewer": [{"kind": "alias", "id": "review.deep"}]}, + "task_classes": {"fixture-review-small": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "fixture-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + return ( + json.dumps(profiles, sort_keys=True) + "\n", + json.dumps(policy, sort_keys=True) + "\n", + ) diff --git a/test/core/experiment_runtime_fixture.py b/test/core/experiment_runtime_fixture.py new file mode 100644 index 0000000..2ef8489 --- /dev/null +++ b/test/core/experiment_runtime_fixture.py @@ -0,0 +1,260 @@ +"""Real offline experiment runs through the explicit public trial controller. + +This is not native-provider quality proof. +Profiles and reviews are explicit fixtures, but preparation, worker processes, +checks, host decisions and outcomes use the actual Service/Store workflow. +""" + +from __future__ import annotations + +from contextlib import ExitStack +import copy +from pathlib import Path +import subprocess +import sys +import time +from unittest.mock import patch + +from test_experiment_provenance import digest, execution_digest +from test_lifecycle import profile, review_task, routing_policy + +from devsquad.experiment_provenance import paired_input_identity, selected_execution_fingerprint +from devsquad.codex_review_worker import freeze_codex_reviewer +from devsquad.router import load_routing +from devsquad.service import Service +from devsquad.store import Store, canonical_json, git_common_dir, request_hash + + +class ExperimentRuntimeFixture: + def __init__( + self, root: Path, *, with_fallback=False, fail_candidate=False, + service=None, repo=None, experiment_id="saved-run-review-pair", + candidate_succeeds=True, case_splits=None, profiles=None, workflow="branch-review", + policy=None, task_class=None, native_review=False, + ): + self.root = root.resolve() + self.root.mkdir(parents=True, exist_ok=True) + self.with_fallback = with_fallback + self.fail_candidate = fail_candidate + self.candidate_succeeds = candidate_succeeds + self.experiment_id = experiment_id + self.workflow = workflow + self.task_class = task_class + self.native_review = native_review + self.role = "implementer" if workflow == "issue-delivery" else "reviewer" + self.case_splits = case_splits or [("eval-1", "evaluation"), ("hold-1", "held_out")] + self.repo = repo or self.root / "repo" + self.service = service or Service(self.root / "runtime") + self.runs = {} + self.git("init", "-q", str(self.repo), outside=True) + self.git("config", "user.email", "test@example.invalid") + self.git("config", "user.name", "Test") + (self.repo / "README").write_text("base\n") + self.git("add", "README") + self.git("commit", "-qm", "baseline") + self.base = self.git("rev-parse", "HEAD").strip() + self.targets = {} + for case_id, _ in self.case_splits: + (self.repo / "README").write_text(f"candidate {case_id}\n") + self.git("add", "README") + self.git("commit", "-qm", f"candidate {case_id}") + self.targets[case_id] = self.git("rev-parse", "HEAD").strip() + self.common = str(git_common_dir(self.repo)) + self.profiles = copy.deepcopy(profiles) if profiles is not None else { + arm: profile(f"profile-{letter}", f"model-{letter}") + for arm, letter in (("control", "a"), ("candidate", "b")) + } + if with_fallback or fail_candidate: + # The real offline worker deliberately errors on this suffix. + self.profiles["candidate"]["id"] = "candidate-fixture-fail" + self.registry = { + "schema_version": 1, "profiles": list(self.profiles.values()), + "bindings": {"review.deep": {"profile_id": self.profiles["control"]["id"], "version": 7}}, + } + self.policy = copy.deepcopy(policy) if policy is not None else routing_policy() + if workflow == "issue-delivery": + for value in self.profiles.values(): + value.update(permission_policy="workspace_write", required_tools=["read", "write"]) + reviewer = profile("fixture-reviewer", "independent-review-model") + self.registry["profiles"].append(reviewer) + self.policy["roles"] = { + "implementer": [{"kind": "alias", "id": "review.deep"}], + "reviewer": [{"kind": "profile", "id": reviewer["id"]}], + } + self.policy["task_classes"] = {"fixture-delivery-small": "proven"} + if with_fallback: + fallback = profile("profile-fallback", "model-fallback") + self.registry["profiles"].append(fallback) + self.policy["roles"]["reviewer"] = [{"kind": "profile", "id": fallback["id"]}] + self.review_adapters = {} + if native_review: + for value in self.profiles.values(): + selected = {"profile_id": value["id"], "profile": value, "profile_sha256": digest(value)} + self.review_adapters[value["id"]] = freeze_codex_reviewer(selected) + self.package_path, self.package_digest = self.service._freeze_package() + cases = [] + for case_id, split in self.case_splits: + identities = paired_input_identity( + self.declaration_snapshot(case_id, "control"), + role=self.role, package_digest=self.package_digest, + ) + cases.append({ + "case_id": case_id, "split": split, + "control_outcome_id": self.outcome_id(case_id, "control"), + "candidate_outcome_id": self.outcome_id(case_id, "candidate"), + **identities, + }) + self.spec = { + "schema_version": 2, "experiment_id": experiment_id, + "project_path": str(self.repo), + "question": "Does the candidate fixture improve paired acceptance?", + "hypothesis": "Candidate outcomes improve in evaluation and held-out cases.", + "evidence_availability": "tracked_fixture", + "variable": { + "kind": "profile_binding", "alias": "review.deep", "role": self.role, + **{f"{arm}_profile_id": value["id"] for arm, value in self.profiles.items()}, + **{f"{arm}_profile_sha256": digest(value) for arm, value in self.profiles.items()}, + **{f"{arm}_execution_sha256": selected_execution_fingerprint(self.declaration_snapshot(self.case_splits[0][0], arm), role=self.role) + for arm in self.profiles}, + }, + "cases": cases, + "gate": { + "min_evaluation_pairs": 1, "min_held_out_pairs": 1, + "noninferiority_margin": 0.0, "minimum_success_gain": 1.0, + "max_candidate_escaped_defects": 0, + }, + "budget": {"max_cases": len(cases), "max_worker_invocations": len(cases) * (4 if with_fallback or workflow == "issue-delivery" else 2), "wall_seconds": 600}, + "rollback_target": {"profile_id": self.profiles["control"]["id"], "binding_version": 7}, + } + + def outcome_id(self, case_id, arm): + suffix = "" if self.experiment_id == "saved-run-review-pair" else f"-{self.experiment_id}" + return f"{arm}-{case_id}{suffix}" + + def git(self, *args, outside=False): + command = ["git"] if outside else ["git", "-C", str(self.repo)] + return subprocess.run(command + list(args), check=True, capture_output=True, text=True).stdout + + def task(self, case_id, arm): + task = review_task(self.repo) + task["project"].update({"base_ref": self.base, "target_ref": self.targets[case_id]}) + task["goal"] = "Review the exact README candidate and pass the declared check." + task["routing"] = { + "profiles": copy.deepcopy(self.registry), "policy": copy.deepcopy(self.policy), + "overrides": {self.role: { + "profile_id": self.profiles[arm]["id"], + "fallback": "policy" if self.with_fallback else "none", + }}, + } + task["checks"] = [{ + "id": "read-candidate", "argv": [sys.executable, "-c", "from pathlib import Path; assert Path('README').read_text().startswith('candidate')"], + "cwd": ".", "timeout_seconds": 10, "required_to_pass": True, + }] + task["budget"]["wall_seconds"] = 120 + if self.task_class is not None: + task["task_class"] = self.task_class + if self.with_fallback: + task["budget"].update(max_worker_invocations=2, max_fallbacks_per_step=1) + if self.workflow == "issue-delivery": + baseline = self.targets[case_id] + task.update(workflow=self.workflow, task_class="fixture-delivery-small") + task["project"].update(base_ref=baseline, target_ref=baseline) + task["goal"] = f"Implement the bounded README repair for {case_id}." + task["scope"]["write_paths"] = ["README"] + task["checks"][0]["argv"][-1] = "from pathlib import Path; assert Path('README').read_text().startswith('fixed')" + task["budget"].update(max_worker_invocations=2) + return task + + def declaration_snapshot(self, case_id, arm): + task = self.task(case_id, arm) + if self.workflow == "issue-delivery": + return { + "task": task, "base_oid": self.targets[case_id], "target_oid": self.targets[case_id], + "delivery_workspace": {"baseline_oid": self.targets[case_id]}, + "configs": {"policy_file": {"sha256": digest(self.policy)}}, + "routing": load_routing(task, canonical_json(self.registry), canonical_json(self.policy)), + } + identity = { + "schema_version": 1, "base_oid": self.base, + "target_oid": self.targets[case_id], "changed_paths": ["README"], + } + return { + "task": task, "base_oid": self.base, "target_oid": self.targets[case_id], + "workspace": {**identity, "candidate_sha256": digest(identity)}, + "configs": {"policy_file": {"sha256": digest(self.policy)}}, + "routing": load_routing(task, canonical_json(self.registry), canonical_json(self.policy)), + **({"review_adapters": copy.deepcopy(self.review_adapters)} if self.native_review else {}), + } + + def store(self): + return Store(self.service.database, self.service.artifacts) + + def wait(self, run_id, *, candidate_ready=True): + deadline = time.monotonic() + 20 + while time.monotonic() < deadline: + status = self.service.status(run_id) + if (status["state"] in {"awaiting_host", "failed", "succeeded", "cancelled"} + or candidate_ready and status.get("next_action") == "resume_candidate_review"): + return status + time.sleep(0.05) + raise AssertionError(f"saved-run fixture did not reach a gate: {self.service.status(run_id)}") + + def run_arm(self, case_id, arm, *, no_attempt=False): + with ExitStack() as stack: + if no_attempt: + stack.enter_context(patch.object(self.service, "_spawn_daemon", return_value=0)) + fixture_args = {} + if self.workflow == "issue-delivery": + fixture_args["_internal_implementation_fixture"] = { + "writes": [{"path": "README", "content": f"fixed {case_id}\n"}], + "delay_seconds": 0, + **({"fail_profile_ids": [self.profiles["candidate"]["id"]]} if self.fail_candidate else {}), + } + if not self.native_review: + fixture_args["_internal_review_fixture"] = {"verdict": "clean", "summary": "Fixture review of the frozen candidate.", "findings": []} + started = self.service.trial_start( + self.spec, case_id, arm, self.task(case_id, arm), self.outcome_id(case_id, arm), + **fixture_args, + ) + run_id = started["run_id"] + self.runs[(case_id, arm)] = run_id + if started["state"] != "queued": + raise AssertionError(f"public fixture preparation failed: {started}") + if no_attempt: + self.service.cancel(run_id) + verdict = "cancelled" + else: + waiting = self.wait(run_id) + if waiting.get("next_action") == "resume_candidate_review": + self.service.resume(run_id) + waiting = self.wait(run_id, candidate_ready=False) + if waiting["state"] == "failed" and self.fail_candidate and arm == "candidate": + verdict = "failed" + else: + if waiting["state"] != "awaiting_host": + raise AssertionError(f"public fixture worker failed: {waiting}") + claimed = self.service.handoff_claim(run_id, waiting["version"], "experiment-fixture-host") + packet = claimed["handoff"]["packet"] + body = { + "schema_version": 1, "submission_id": f"finish-{arm}-{case_id}", + "disposition": "accept" if (arm == "candidate") == self.candidate_succeeds else "reject", + "reason": "Predeclared offline fixture disposition.", + "evidence_refs": [{"artifact_id": reference["artifact_id"], "sha256": reference["sha256"]} for reference in packet["artifacts"]], + } + completed = self.service.handoff_complete(run_id, claimed["claim"], {**body, "submission_hash": request_hash(body)}) + verdict = "succeeded" if (arm == "candidate") == self.candidate_succeeds else "failed" + if completed["state"] != verdict: + raise AssertionError(f"public fixture disposition failed: {completed}") + return run_id + + def run_all(self, *, skip=None, no_attempt=None): + for case in self.spec["cases"]: + for arm in ("control", "candidate"): + key = (case["case_id"], arm) + if key != skip: + self.run_arm(*key, no_attempt=key == no_attempt) + + def close(self): + for run_id in self.runs.values(): + if self.service.status(run_id)["state"] not in {"succeeded", "failed", "cancelled"}: + self.service.cancel(run_id) diff --git a/test/core/fakes/claude_delivery_cli.py b/test/core/fakes/claude_delivery_cli.py new file mode 100755 index 0000000..5f820ca --- /dev/null +++ b/test/core/fakes/claude_delivery_cli.py @@ -0,0 +1,35 @@ +#!/usr/bin/env python3 +"""Offline Claude stream fixture: repair only the installed-usability seed.""" + +import json +import os +from pathlib import Path +import sys + + +if "--version" in sys.argv: + print("2.1.220 (Claude Code)") + raise SystemExit(0) +if sys.argv[1:3] == ["auth", "status"]: + print(json.dumps({"loggedIn": True, "authMethod": "claude.ai", "apiProvider": "firstParty"})) + raise SystemExit(0) +if os.environ.get("DEVSQUAD_WORKER") != "1" or "--print" not in sys.argv or "--" not in sys.argv: + raise SystemExit("fixture accepts only a fenced offline writer invocation") + +path = Path("src/app.py") +seed = "def add(a, b):\n return a - b\n" +if path.is_symlink() or not path.is_file() or path.read_text() != seed: + raise SystemExit("fixture writer requires the exact seeded defect") +path.write_text("def add(a, b):\n return a + b\n") +model, session = "claude-sonnet-4-6", "offline-installed-writer-session" +records = [{ + "type": "assistant", "session_id": session, "parent_tool_use_id": None, + "message": {"role": "assistant", "model": model}, +}, { + "type": "result", "subtype": "success", "is_error": False, + "session_id": session, "result": "Corrected the bounded addition defect.", + "usage": {"input_tokens": 12, "output_tokens": 7}, + "modelUsage": {model: {"inputTokens": 12, "outputTokens": 7, "provider": "firstParty"}}, +}] +for record in records: + print(json.dumps(record), flush=True) diff --git a/test/core/fakes/codex_app_server.py b/test/core/fakes/codex_app_server.py new file mode 100644 index 0000000..9722d57 --- /dev/null +++ b/test/core/fakes/codex_app_server.py @@ -0,0 +1,27 @@ +#!/usr/bin/env python3 +import json, sys, time +initialized = False +for line in sys.stdin: + request = json.loads(line) + method = request.get("method") + if method == "initialize": + print(json.dumps({"id":request["id"],"result":{"serverInfo":{"name":"fake","version":"1"}}}), flush=True) + elif method == "initialized": + initialized = True + elif method == "model/list": + if not initialized: + print(json.dumps({"id":request["id"],"error":{"code":-32002,"message":"not initialized"}}), flush=True); continue + cursor = request.get("params",{}).get("cursor") + result = {"data":[{"id":"gpt-fake","supportedReasoningEfforts":[{"reasoningEffort":"low","description":"fixture"}]}],"nextCursor":"two"} if cursor is None else {"data":[],"nextCursor":None} + print(json.dumps({"method":"account/updated","params":{"reason":"fixture"}}), flush=True) + print(json.dumps({"id":request["id"],"result":result}), flush=True) + elif method == "turn/start": + if not initialized: + print(json.dumps({"id":request["id"],"error":{"code":-32002,"message":"not initialized"}}), flush=True); continue + print(json.dumps({"id":request["id"],"result":{"turn":{"id":"turn-1"}}}), flush=True) + time.sleep(0.02) + print(json.dumps({"method":"turn/started","params":{"threadId":request["params"]["threadId"],"turn":{"id":"turn-1","status":"inProgress"}}}), flush=True) + time.sleep(0.02) + print(json.dumps({"method":"item/agentMessage/delta","params":{"threadId":request["params"]["threadId"],"turnId":"turn-1","delta":"ok"}}), flush=True) + time.sleep(0.02) + print(json.dumps({"method":"turn/completed","params":{"threadId":request["params"]["threadId"],"turn":{"id":"turn-1","status":"completed"}}}), flush=True) diff --git a/test/core/fakes/codex_review_cli.py b/test/core/fakes/codex_review_cli.py new file mode 100755 index 0000000..3e8db33 --- /dev/null +++ b/test/core/fakes/codex_review_cli.py @@ -0,0 +1,210 @@ +#!/usr/bin/env python3 +"""Small Codex app-server fixture used by the public M3 native review tests.""" + +import json +import re +import sys + + +if "--version" in sys.argv: + print("codex-cli 0.153.4") + raise SystemExit(0) + + +model = "gpt-fake-review" +for index, argument in enumerate(sys.argv[:-1]): + if argument == "-c": + match = re.fullmatch(r'model="([^"]+)"', sys.argv[index + 1]) + if match: + model = match.group(1) + + +initialized = False +thread_id = "fixture-thread" +turn_id = "fixture-turn" +for line in sys.stdin: + request = json.loads(line) + method = request.get("method") + request_id = request.get("id") + params = request.get("params", {}) + if method == "initialize": + print(json.dumps({ + "id": request_id, + "result": {"serverInfo": {"name": "fake-codex", "version": "0.153.4"}}, + }), flush=True) + elif method == "initialized": + initialized = True + elif method == "account/read": + print(json.dumps({"id": request_id, "result": { + "account": {"type": "chatgpt", "id": "offline-fixture-account", "planType": "plus"}, + "requiresOpenaiAuth": True, + }}), flush=True) + elif method == "config/read": + print(json.dumps({"id": request_id, "result": { + "config": {"model_provider": "openai"}, + }}), flush=True) + elif method == "account/rateLimits/read": + print(json.dumps({"id": request_id, "result": {"rateLimits": None}}), flush=True) + elif method == "model/list": + if not initialized: + print(json.dumps({ + "id": request_id, + "error": {"code": -32002, "message": "not initialized"}, + }), flush=True) + continue + print(json.dumps({ + "id": request_id, + "result": { + "data": [{ + "id": model, + "supportedReasoningEfforts": [{ + "reasoningEffort": "low", + "description": "fixture", + }], + }], + "nextCursor": None, + }, + }), flush=True) + elif method == "thread/start": + sandbox = ( + {"type": "workspaceWrite", "writableRoots": [params["cwd"]], "networkAccess": False} + if model.endswith("identity-drift") + else {"type": "readOnly", "networkAccess": False} + ) + print(json.dumps({ + "id": request_id, + "result": { + "thread": { + "id": thread_id, + "cliVersion": "0.153.4", + "modelProvider": "openai", + }, + "model": params["model"], + "modelProvider": "openai", + "reasoningEffort": "low", + "cwd": params["cwd"], + "sandbox": sandbox, + "approvalPolicy": "never", + "activePermissionProfile": None, + }, + }), flush=True) + elif method == "turn/start": + if model.endswith("denied"): + print(json.dumps({ + "id": request_id, + "error": {"code": -32000, "message": "permission denied by fixture"}, + }), flush=True) + continue + prompt = params["input"][0]["text"] + assignment = json.loads(prompt.splitlines()[-1]) + if "single read-only lead" in prompt: + review = { + "schema_version": 1, + "candidate_sha256": assignment["candidate_sha256"], + "disposition": "reject" if model.endswith("lead-reject") else "accept", + "reason": "The frozen review and checks support this disposition.", + } + else: + review = { + "schema_version": 1, + "candidate_sha256": assignment["candidate_sha256"], + "base_oid": assignment["base_oid"], + "target_oid": assignment["target_oid"], + "review_mode": assignment["review_mode"], + "verdict": "clean", + "summary": "The bounded native fixture found no supported defect.", + "findings": [], + } + output = ( + "" if model.endswith("empty") + else "{}" if model.endswith("malformed") + else json.dumps(review, sort_keys=True, separators=(",", ":")) + ) + print(json.dumps({ + "id": request_id, + "result": {"turn": {"id": turn_id}}, + }), flush=True) + if model.endswith("disconnect"): + break + print(json.dumps({ + "method": "turn/started", + "params": { + "threadId": thread_id, + "turn": {"id": turn_id, "status": "inProgress"}, + }, + }), flush=True) + if model == "gpt-fake-review": + print(json.dumps({ + "method": "item/agentMessage/delta", + "params": { + "threadId": thread_id, + "turnId": turn_id, + "itemId": "fixture-draft", + "delta": '{"draft":true}', + }, + }), flush=True) + print(json.dumps({ + "method": "item/completed", + "params": { + "threadId": thread_id, + "turnId": turn_id, + "item": { + "id": "fixture-draft", + "type": "agentMessage", + "text": '{"draft":true}', + }, + }, + }), flush=True) + midpoint = len(output) // 2 + for part in (output[:midpoint], output[midpoint:]): + print(json.dumps({ + "method": "item/agentMessage/delta", + "params": { + "threadId": thread_id, + "turnId": turn_id, + "delta": part, + }, + }), flush=True) + print(json.dumps({ + "method": "item/completed", + "params": { + "threadId": thread_id, + "turnId": turn_id, + "item": { + "id": "fixture-final", + "type": "agentMessage", + "text": output, + }, + }, + }), flush=True) + print(json.dumps({ + "method": "thread/tokenUsage/updated", + "params": { + "threadId": thread_id, + "turnId": turn_id, + "tokenUsage": { + "total": { + "inputTokens": 120, + "cachedInputTokens": 0, + "outputTokens": 40, + "reasoningOutputTokens": 10, + "totalTokens": 160, + }, + "last": { + "inputTokens": 120, + "cachedInputTokens": 0, + "outputTokens": 40, + "reasoningOutputTokens": 10, + "totalTokens": 160, + }, + "modelContextWindow": 1000, + }, + }, + }), flush=True) + print(json.dumps({ + "method": "turn/completed", + "params": { + "threadId": thread_id, + "turn": {"id": turn_id, "status": "completed"}, + }, + }), flush=True) diff --git a/test/core/fakes/decision_adapter.py b/test/core/fakes/decision_adapter.py new file mode 100644 index 0000000..4efc218 --- /dev/null +++ b/test/core/fakes/decision_adapter.py @@ -0,0 +1,54 @@ +"""Deterministic typed decision adapter used only by offline core tests.""" + +from __future__ import annotations + +import hashlib +import json + + +def _canonical(value): + return json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False) + + +class FakeDecisionAdapter: + def __init__(self, rankings=None, *, confidence=0.9, elapsed_ms=3): + self.rankings = rankings or {} + self.confidence = confidence + self.elapsed_ms = elapsed_ms + self.calls = 0 + + def decide(self, request): + self.calls += 1 + recommendations = {} + for role, candidates in request["candidates"].items(): + ranking = list(self.rankings.get(role, candidates)) + denominator = sum(range(1, len(ranking) + 1)) + descending = list(range(len(ranking), 0, -1)) + scores = { + profile_id: score / denominator + for profile_id, score in zip(ranking, descending) + } + recommendations[role] = { + "ranking": ranking, + "probabilities": scores, + "confidence": self.confidence, + "abstain_reason": None, + } + return { + "schema_version": 1, + "request_sha256": hashlib.sha256( + _canonical(request).encode(), + ).hexdigest(), + "adapter": request["adapter"], + "language": request["language"], + "truncation": {"occurred": False, "detail": None}, + "recommendations": recommendations, + "usage": { + "source": "fake", + "billable_requests": 1, + "input_tokens": 10, + "output_tokens": 2, + "cost_usd": 0.0, + }, + "elapsed_ms": self.elapsed_ms, + } diff --git a/test/core/probes/jev_decision_eval.py b/test/core/probes/jev_decision_eval.py new file mode 100644 index 0000000..f04a371 --- /dev/null +++ b/test/core/probes/jev_decision_eval.py @@ -0,0 +1,425 @@ +#!/usr/bin/env python3 +"""Run one bounded, synthetic Jev decision-classifier pilot. + +The probe is deliberately separate from the DevSquad runtime. It makes exactly +one billable request, performs no retries, accepts the API key through the +environment or an explicitly selected private env file, and writes a redacted +result containing no task text. Dry run is the default and makes no API call. +""" + +from __future__ import annotations + +import argparse +from datetime import datetime, timezone +import hashlib +import json +import math +import os +from pathlib import Path +import re +import ssl +import stat +import sys +import tempfile +import time +from typing import Any +from urllib.error import HTTPError, URLError +from urllib.request import Request, urlopen + + +ROOT = Path(__file__).resolve().parents[3] +DEFAULT_SPEC = ( + ROOT + / "docs" + / "plans" + / "engineering-team" + / "experiments" + / "jev-pilot-v1.json" +) +CONTEXT_LIMIT_TOKENS = 64_000 +OFFICIAL_ENDPOINT = "https://api.typesafe.ai/v1/systemone" +PINNED_MODEL = "jev-1.13.0" +MAX_RESPONSE_BYTES = 1_048_576 +MAX_ENV_BYTES = 16_384 + + +class ProbeError(RuntimeError): + """A safe, user-facing probe failure.""" + + +def _key_value(value: str) -> str | None: + if any(character.isspace() or ord(character) < 32 or ord(character) == 127 + for character in value): + raise ProbeError("TYPESAFE_API_KEY must be a single non-whitespace value") + return value or None + + +def load_api_key(env_file: Path | None = None) -> str | None: + """Read only the Jev key; never source shell code or mutate process env.""" + exported = os.environ.get("TYPESAFE_API_KEY") + if exported: + return _key_value(exported) + if env_file is None: + return None + try: + descriptor = os.open(env_file, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + with os.fdopen(descriptor, "r", encoding="utf-8") as handle: + info = os.fstat(handle.fileno()) + if not stat.S_ISREG(info.st_mode): + raise ProbeError("Jev env file must be a regular file") + if info.st_mode & 0o077: + raise ProbeError("Jev env file permissions must be private (chmod 600)") + if info.st_size > MAX_ENV_BYTES: + raise ProbeError("Jev env file exceeded the size limit") + contents = handle.read(MAX_ENV_BYTES + 1) + except (OSError, UnicodeError): + raise ProbeError("Jev env file could not be read; use a private UTF-8 regular file") from None + if len(contents) > MAX_ENV_BYTES: + raise ProbeError("Jev env file exceeded the size limit") + key = None + found = False + for line in contents.splitlines(): + match = re.fullmatch(r"(?:export[ \t]+)?TYPESAFE_API_KEY[ \t]*=[ \t]*(.*)", line.strip()) + if match is None: + continue + if found: + raise ProbeError("Jev env file defines TYPESAFE_API_KEY more than once") + found = True + value = match[1].strip() + if value.startswith(("'", '"')): + if len(value) < 2 or value[-1] != value[0]: + raise ProbeError("Jev env file has an invalid quoted TYPESAFE_API_KEY") + value = value[1:-1] + else: + value = value.partition(" #")[0].rstrip() + key = _key_value(value) + return key + + +def load_spec(path: Path) -> dict[str, Any]: + with path.open(encoding="utf-8") as handle: + spec = json.load(handle) + required = { + "schema_version", + "experiment_id", + "status", + "model", + "endpoint", + "pricing", + "budget", + "purposes", + "cases", + } + if set(spec) != required: + raise ProbeError("pilot spec fields do not match the version-1 contract") + if spec["schema_version"] != 1: + raise ProbeError("unsupported pilot spec version") + if spec["status"] != "ready_for_live_run": + raise ProbeError("pilot spec is not approved for a live run") + if spec["endpoint"] != OFFICIAL_ENDPOINT or spec["model"] != PINNED_MODEL: + raise ProbeError("pilot endpoint and model must remain pinned") + pricing = spec["pricing"] + if not isinstance(pricing, dict) or set(pricing) != { + "checked_at", + "usd_per_million_input_tokens", + "max_cost_usd", + }: + raise ProbeError("pilot pricing fields are invalid") + if pricing["checked_at"] != "2026-09-26": + raise ProbeError("pilot pricing must be rechecked before changing its date") + if not all( + type(pricing[name]) in {int, float} + and math.isfinite(pricing[name]) + and pricing[name] > 0 + for name in ("usd_per_million_input_tokens", "max_cost_usd") + ): + raise ProbeError("pilot pricing values must be finite and positive") + if pricing["max_cost_usd"] != 0.01: + raise ProbeError("pilot cost ceiling must remain exactly $0.01") + if spec["budget"] != { + "max_billable_requests": 1, + "retries": 0, + "timeout_seconds": 30, + "data_class": "synthetic_public_fixture", + }: + raise ProbeError("pilot budget must remain one request with no retries") + if not spec["cases"] or not spec["purposes"]: + raise ProbeError("pilot requires cases and purposes") + case_ids = [case.get("id") for case in spec["cases"]] + purpose_ids = [purpose.get("id") for purpose in spec["purposes"]] + if len(case_ids) != len(set(case_ids)) or not all( + isinstance(value, str) and value for value in case_ids + ): + raise ProbeError("case IDs must be unique non-empty strings") + if len(purpose_ids) != len(set(purpose_ids)) or not all( + isinstance(value, str) and value for value in purpose_ids + ): + raise ProbeError("purpose IDs must be unique non-empty strings") + for case in spec["cases"]: + if set(case) != {"id", "task", "expected"}: + raise ProbeError(f"case {case.get('id')} has invalid fields") + if set(case["expected"]) != set(purpose_ids): + raise ProbeError(f"case {case['id']} lacks an expected purpose label") + purpose_map = {item["id"]: item for item in spec["purposes"]} + for purpose_id, purpose in purpose_map.items(): + if set(purpose) != {"id", "instructions", "criteria"}: + raise ProbeError(f"purpose {purpose_id} has invalid fields") + if not isinstance(purpose["instructions"], str) or not purpose["instructions"]: + raise ProbeError(f"purpose {purpose_id} lacks instructions") + if ( + not isinstance(purpose["criteria"], dict) + or len(purpose["criteria"]) < 2 + or not all( + isinstance(key, str) + and key + and isinstance(value, str) + and value + for key, value in purpose["criteria"].items() + ) + ): + raise ProbeError(f"purpose {purpose_id} criteria are invalid") + for case in spec["cases"]: + if not isinstance(case["task"], str) or not case["task"]: + raise ProbeError(f"case {case['id']} lacks task text") + for purpose_id, expected in case["expected"].items(): + if expected not in purpose_map[purpose_id]["criteria"]: + raise ProbeError(f"case {case['id']} has an unknown expected label") + return spec + + +def build_request(spec: dict[str, Any]) -> dict[str, Any]: + state = [ + {"task_id": case["id"], "task": case["task"]} + for case in spec["cases"] + ] + questions: dict[str, Any] = {} + for case in spec["cases"]: + for purpose in spec["purposes"]: + key = f"{case['id']}__{purpose['id']}" + questions[key] = { + "type": "choice", + "instructions": ( + f"For task_id {case['id']} only: {purpose['instructions']}" + ), + "criteria": purpose["criteria"], + } + return {"state": state, "model": spec["model"], "questions": questions} + + +def _finite_probability(value: Any) -> bool: + return ( + type(value) in {int, float} + and math.isfinite(value) + and 0.0 <= float(value) <= 1.0 + ) + + +def validate_response( + spec: dict[str, Any], request_body: dict[str, Any], response: Any +) -> dict[str, Any]: + if not isinstance(response, dict) or set(response) != {"model", "answers", "usage"}: + raise ProbeError("Jev response envelope is invalid") + if response["model"] != spec["model"]: + raise ProbeError("Jev response model does not match the concrete pin") + answers = response["answers"] + if not isinstance(answers, dict) or set(answers) != set(request_body["questions"]): + raise ProbeError("Jev answer IDs do not match the frozen questions") + usage = response["usage"] + if not isinstance(usage, dict) or set(usage) != {"input_tokens", "output_tokens"}: + raise ProbeError("Jev usage is invalid") + if not all(type(usage[name]) is int and usage[name] >= 0 for name in usage): + raise ProbeError("Jev usage counts must be non-negative integers") + + results = [] + purpose_map = {item["id"]: item for item in spec["purposes"]} + for case in spec["cases"]: + predictions: dict[str, Any] = {} + for purpose_id, purpose in purpose_map.items(): + key = f"{case['id']}__{purpose_id}" + answer = answers[key] + criteria = set(purpose["criteria"]) + if not isinstance(answer, dict) or set(answer) != { + "type", + "choice", + "confidence", + "probabilities", + }: + raise ProbeError(f"answer {key} has invalid fields") + probabilities = answer["probabilities"] + if answer["type"] != "choice" or answer["choice"] not in criteria: + raise ProbeError(f"answer {key} has an unknown choice") + if not _finite_probability(answer["confidence"]): + raise ProbeError(f"answer {key} has invalid confidence") + if not isinstance(probabilities, dict) or set(probabilities) != criteria: + raise ProbeError(f"answer {key} probability labels are invalid") + if not all(_finite_probability(value) for value in probabilities.values()): + raise ProbeError(f"answer {key} probabilities are invalid") + if not math.isclose(sum(probabilities.values()), 1.0, abs_tol=0.02): + raise ProbeError(f"answer {key} probabilities do not sum to one") + predictions[purpose_id] = { + "expected": case["expected"][purpose_id], + "choice": answer["choice"], + "correct": answer["choice"] == case["expected"][purpose_id], + "confidence": answer["confidence"], + "probabilities": probabilities, + } + results.append({"case_id": case["id"], "predictions": predictions}) + return {"model": response["model"], "usage": usage, "cases": results} + + +def summarize( + spec: dict[str, Any], + validated: dict[str, Any], + elapsed_ms: int, + request_sha256: str, +) -> dict[str, Any]: + per_purpose: dict[str, dict[str, int]] = { + purpose["id"]: {"correct": 0, "total": 0} + for purpose in spec["purposes"] + } + for case in validated["cases"]: + for purpose_id, prediction in case["predictions"].items(): + per_purpose[purpose_id]["total"] += 1 + per_purpose[purpose_id]["correct"] += int(prediction["correct"]) + for counts in per_purpose.values(): + counts["accuracy_percent"] = round( + 100.0 * counts["correct"] / counts["total"], 1 + ) + input_tokens = validated["usage"]["input_tokens"] + rate = spec["pricing"]["usd_per_million_input_tokens"] + cost = input_tokens * rate / 1_000_000 + return { + "schema_version": 1, + "experiment_id": spec["experiment_id"], + "executed_at": datetime.now(timezone.utc).isoformat(), + "data_class": spec["budget"]["data_class"], + "request_sha256": request_sha256, + "billable_requests": 1, + "requested_model": spec["model"], + "observed_model": validated["model"], + "elapsed_ms": elapsed_ms, + "usage": validated["usage"], + "pricing": { + "usd_per_million_input_tokens": rate, + "estimated_cost_usd": round(cost, 8), + "max_cost_usd": spec["pricing"]["max_cost_usd"], + }, + "per_purpose": per_purpose, + "cases": validated["cases"], + "interpretation": ( + "Synthetic smoke result only; it does not establish a production " + "routing improvement or authorize advisory mode." + ), + } + + +def write_json(path: Path, value: dict[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with tempfile.NamedTemporaryFile( + "w", encoding="utf-8", dir=path.parent, delete=False + ) as handle: + json.dump(value, handle, indent=2, sort_keys=True) + handle.write("\n") + temporary = Path(handle.name) + temporary.replace(path) + + +def execute( + spec: dict[str, Any], request_body: dict[str, Any], *, env_file: Path | None = None, +) -> dict[str, Any]: + key = load_api_key(env_file) + if not key: + raise ProbeError("TYPESAFE_API_KEY is not set") + encoded = json.dumps(request_body, separators=(",", ":")).encode("utf-8") + request = Request( + spec["endpoint"], + data=encoded, + headers={ + "Authorization": f"Bearer {key}", + "Content-Type": "application/json", + "User-Agent": "devsquad-jev-pilot/1", + }, + method="POST", + ) + start = time.monotonic() + try: + with urlopen( + request, + timeout=spec["budget"]["timeout_seconds"], + context=ssl.create_default_context(), + ) as response: + raw = response.read(MAX_RESPONSE_BYTES + 1) + if len(raw) > MAX_RESPONSE_BYTES: + raise ProbeError("TypeSafe API response exceeded the size limit") + payload = json.loads(raw.decode("utf-8")) + except HTTPError as exc: + raise ProbeError(f"TypeSafe API returned HTTP {exc.code}; no retry attempted") from None + except URLError as exc: + raise ProbeError(f"TypeSafe API was unreachable ({exc.reason}); no retry attempted") from None + except (TimeoutError, json.JSONDecodeError): + raise ProbeError("TypeSafe API timed out or returned invalid JSON; no retry attempted") from None + elapsed_ms = round((time.monotonic() - start) * 1000) + validated = validate_response(spec, request_body, payload) + request_sha256 = hashlib.sha256(encoded).hexdigest() + return summarize(spec, validated, elapsed_ms, request_sha256) + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--spec", type=Path, default=DEFAULT_SPEC) + parser.add_argument("--execute", action="store_true") + parser.add_argument("--output", type=Path) + parser.add_argument( + "--env-file", type=Path, + help="Read TYPESAFE_API_KEY from a private file; an exported key takes precedence", + ) + args = parser.parse_args(argv) + try: + spec = load_spec(args.spec) + request_body = build_request(spec) + price = spec["pricing"]["usd_per_million_input_tokens"] + max_cost = spec["pricing"]["max_cost_usd"] + documented_max_cost = CONTEXT_LIMIT_TOKENS * price / 1_000_000 + if documented_max_cost > max_cost: + raise ProbeError( + "one maximum-context request exceeds the frozen cost ceiling; " + "stop and evaluate Laya" + ) + dry_run = { + "experiment_id": spec["experiment_id"], + "cases": len(spec["cases"]), + "questions": len(request_body["questions"]), + "billable_requests": 1, + "retries": 0, + "request_bytes": len( + json.dumps(request_body, separators=(",", ":")).encode("utf-8") + ), + "request_sha256": hashlib.sha256( + json.dumps(request_body, separators=(",", ":")).encode("utf-8") + ).hexdigest(), + "documented_max_request_cost_usd": round(documented_max_cost, 8), + "cost_ceiling_usd": max_cost, + "data_class": spec["budget"]["data_class"], + } + if not args.execute: + if args.env_file is not None: + dry_run["api_key_configured"] = bool(load_api_key(args.env_file)) + print(json.dumps(dry_run, indent=2, sort_keys=True)) + return 0 + if args.output is None: + raise ProbeError("--output is required for a live run") + if args.output.resolve().is_relative_to(ROOT): + raise ProbeError("live output must remain outside the Git repository") + result = execute(spec, request_body, env_file=args.env_file) + write_json(args.output, result) + print(json.dumps({**dry_run, "result": str(args.output)}, indent=2, sort_keys=True)) + if result["pricing"]["estimated_cost_usd"] > max_cost: + raise ProbeError("actual reported usage exceeded the cost ceiling; switch to Laya") + return 0 + except ProbeError as exc: + print(f"JEV_PROBE_ERROR: {exc}", file=sys.stderr) + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/test/core/probes/m3_codex_review.py b/test/core/probes/m3_codex_review.py new file mode 100755 index 0000000..80e84d3 --- /dev/null +++ b/test/core/probes/m3_codex_review.py @@ -0,0 +1,358 @@ +#!/usr/bin/env python3 +"""Opt-in live proof for the public M3 native Codex branch-review path.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +from pathlib import Path +import re +import subprocess +import sys +import time +from typing import Any + + +ROOT = Path(__file__).resolve().parents[3] +CORE_SRC = ROOT / "plugin" / "core" / "src" +sys.path.insert(0, str(CORE_SRC)) + +from devsquad.service import Service +from devsquad.store import request_hash + + +TERMINAL_STATES = {"succeeded", "failed", "cancelled", "timed_out"} +SAFE_MODEL = re.compile(r"[A-Za-z0-9._-]+\Z") + + +def _git(repo: Path, *arguments: str) -> str: + return subprocess.run( + ["git", "-C", str(repo), *arguments], + text=True, + capture_output=True, + check=True, + ).stdout.strip() + + +def _write_json(path: Path, value: dict[str, Any]) -> None: + path.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(65536), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _wait(service: Service, run_id: str, timeout: int) -> dict[str, Any]: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status = service.status(run_id) + if status["state"] in TERMINAL_STATES | {"awaiting_host", "blocked"}: + return status + time.sleep(0.1) + raise TimeoutError("public M3 review did not reach a bounded handoff or terminal state") + + +def _make_repository(repo: Path, model: str, effort: str) -> tuple[str, str]: + repo.mkdir() + _git(repo, "init", "-q") + _git(repo, "config", "user.email", "devsquad-probe@example.invalid") + _git(repo, "config", "user.name", "DevSquad Probe") + (repo / "src").mkdir() + (repo / "tests").mkdir() + (repo / "devsquad").mkdir() + (repo / "src/__init__.py").write_text("") + (repo / "src/ratio.py").write_text( + "def safe_ratio(numerator, denominator):\n" + " if denominator == 0:\n" + " return None\n" + " return numerator / denominator\n" + ) + (repo / "tests/test_ratio.py").write_text( + "import unittest\n" + "from src.ratio import safe_ratio\n\n" + "class RatioTest(unittest.TestCase):\n" + " def test_positive_ratio(self):\n" + " self.assertEqual(safe_ratio(6, 3), 2)\n" + ) + profiles = { + "schema_version": 1, + "profiles": [{ + "id": "m3-live-codex-reviewer", + "harness": "codex", + "model_family": "gpt", + "model_id": model, + "effort": {"value": effort, "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "codex-subscription", + "billing_mode": "subscription", + "quality_status": "trial", + "evidence_refs": ["m3-live-probe"], + }], + "bindings": { + "review.deep": { + "profile_id": "m3-live-codex-reviewer", + "version": 1, + }, + }, + } + policy = { + "schema_version": 1, + "id": "m3-live-probe-policy", + "version": 1, + "roles": {"reviewer": [{"kind": "alias", "id": "review.deep"}]}, + "task_classes": {"live-review-small": "trial"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "codex-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + _write_json(repo / "devsquad/profiles.json", profiles) + _write_json(repo / "devsquad/policy.json", policy) + _git(repo, "add", ".") + _git(repo, "commit", "-qm", "probe base") + base = _git(repo, "rev-parse", "HEAD") + + # Deliberately introduce a small reviewable regression while leaving the + # declared positive-path check green. The integration proof does not + # require a particular verdict, but a supported finding is useful evidence. + (repo / "src/ratio.py").write_text( + "def safe_ratio(numerator, denominator):\n" + " if numerator == 0:\n" + " return None\n" + " return numerator / denominator\n" + ) + _git(repo, "add", "src/ratio.py") + _git(repo, "commit", "-qm", "candidate regression") + return base, _git(repo, "rev-parse", "HEAD") + + +def _task(repo: Path, base: str, target: str, wall_seconds: int) -> dict[str, Any]: + return { + "schema_version": 1, + "project": { + "repo_path": str(repo), + "base_ref": base, + "target_ref": target, + }, + "workflow": "branch-review", + "goal": "Review the exact candidate for correctness regressions with file-and-line evidence.", + "task_class": "live-review-small", + "acceptance": [{ + "id": "candidate-bound-review", + "description": "Return a candidate-bound review with supported file locations.", + "evidence_kind": "review", + }, { + "id": "declared-check", + "description": "Record the trusted declared check outcome.", + "evidence_kind": "check", + }], + "checks": [{ + "id": "unit-tests", + "argv": [sys.executable, "-m", "unittest", "discover", "-s", "tests", "-q"], + "cwd": ".", + "timeout_seconds": 30, + "required_to_pass": True, + }], + "scope": {"read_paths": ["src", "tests"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + }, + "budget": { + "wall_seconds": wall_seconds, + "max_worker_invocations": 1, + "max_revisions": 0, + "max_fallbacks_per_step": 0, + }, + "review": {"mode": "standard"}, + "origin": {"surface": "live-probe"}, + } + + +def _decision(packet: dict[str, Any]) -> dict[str, Any]: + body = { + "schema_version": 1, + "submission_id": "m3-live-host-accept", + "disposition": "accept", + "reason": "Accept the bounded reviewer and required-check evidence.", + "evidence_refs": [{ + "artifact_id": reference["artifact_id"], + "sha256": reference["sha256"], + } for reference in packet["artifacts"]], + } + return {**body, "submission_hash": request_hash(body)} + + +def _lock_down(root: Path) -> None: + for path in sorted(root.rglob("*"), reverse=True): + try: + if path.is_dir(): + path.chmod(0o700) + elif path.is_file(): + path.chmod(0o600) + except OSError: + pass + root.chmod(0o700) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--run-live", action="store_true", + help="required acknowledgement for one subscription-backed model turn", + ) + parser.add_argument( + "--output-dir", type=Path, + default=Path.home() / ".devsquad" / "private-probes", + ) + parser.add_argument("--model", default="gpt-5.5") + parser.add_argument( + "--codex-binary", type=Path, + help="optional absolute codex binary; its directory is prepended for this probe only", + ) + parser.add_argument( + "--effort", default="low", + choices=("minimal", "low", "medium", "high", "xhigh"), + ) + parser.add_argument("--timeout", type=int, default=180) + args = parser.parse_args() + if not args.run_live: + parser.error("--run-live is required") + if not SAFE_MODEL.fullmatch(args.model): + parser.error("--model must contain only letters, numbers, dot, underscore or hyphen") + if args.timeout < 30 or args.timeout > 600: + parser.error("--timeout must be between 30 and 600 seconds") + if args.codex_binary is not None: + try: + binary = args.codex_binary.expanduser().resolve(strict=True) + except OSError as exc: + parser.error(f"--codex-binary cannot be resolved: {exc}") + if (not binary.is_file() or binary.name != "codex" + or not os.access(binary, os.X_OK)): + parser.error("--codex-binary must be an executable absolute path named codex") + os.environ["PATH"] = f"{binary.parent}{os.pathsep}{os.environ.get('PATH', '')}" + + stamp = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime()) + revision = _git(ROOT, "rev-parse", "HEAD") + run_dir = ( + args.output_dir.expanduser().resolve() + / f"m3-codex-review-{stamp}-{revision[:12]}" + ) + run_dir.mkdir(parents=True, mode=0o700, exist_ok=False) + run_dir.chmod(0o700) + receipt: dict[str, Any] = { + "schema_version": 1, + "status": "failed", + "started_at": stamp, + "revision": revision, + "requested": {"model": args.model, "effort": args.effort}, + } + service: Service | None = None + run_id: str | None = None + try: + repo = run_dir / "repository" + runtime = run_dir / "runtime" + base, target = _make_repository(repo, args.model, args.effort) + service = Service(runtime) + started = service.start( + _task(repo, base, target, args.timeout), + f"m3-live-{stamp}", + ) + run_id = started["run_id"] + receipt["run_id"] = run_id + if started["state"] == "failed": + raise RuntimeError(f"public start failed: {started.get('error')}") + waiting = _wait(service, run_id, args.timeout + 30) + if waiting["state"] != "awaiting_host": + raise RuntimeError(f"review did not produce a host handoff: {waiting}") + claimed = service.handoff_claim( + run_id, waiting["version"], "m3-live-probe-host", + ) + packet = claimed["handoff"]["packet"] + attempt = packet["attempt"] + observed = attempt["observed_identity"] + if (observed["harness"] != "codex" + or observed["model_id"] != args.model + or observed["effort"] != args.effort + or observed["permission_policy"] != "read_only" + or observed["verification"] != "verified"): + raise RuntimeError(f"observed reviewer identity drifted: {observed}") + if attempt["usage"]["source"] != "native_reported": + raise RuntimeError("Codex did not report native token usage") + if packet["evaluation"]["accept_allowed"] is not True: + raise RuntimeError(f"required evidence blocked acceptance: {packet['evaluation']}") + completed = service.handoff_complete( + run_id, claimed["claim"], _decision(packet), + ) + if completed["state"] != "succeeded": + raise RuntimeError(f"host completion did not succeed: {completed}") + result = service.result(run_id) + artifacts = {item["name"]: item for item in result["artifacts"]} + required_reports = { + "receipt.json", "receipt.md", "events.jsonl", + "artifact-manifest.json", "result-receipt.json", + } + if not result["ready"] or not required_reports <= set(artifacts): + raise RuntimeError("terminal result is missing required M3 reports") + receipt.update({ + "status": "passed", + "observed": observed, + "native_ids_present": { + key: bool(attempt["native_ids"].get(key)) + for key in ("thread_id", "turn_id") + }, + "usage": attempt["usage"], + "review": { + "verdict": packet["review"]["verdict"], + "finding_count": len(packet["review"]["findings"]), + }, + "checks": [{ + "id": check["id"], "status": check["status"], + "required_to_pass": check["required_to_pass"], + } for check in packet["checks"]], + "candidate_sha256": packet["candidate_sha256"], + "terminal_state": completed["state"], + "report_sha256": { + name: artifacts[name]["sha256"] for name in sorted(required_reports) + }, + }) + except Exception as exc: + receipt["error_type"] = type(exc).__name__ + receipt["error"] = str(exc) + finally: + if service is not None and run_id is not None: + try: + status = service.status(run_id) + if status["state"] not in TERMINAL_STATES: + receipt["cleanup"] = service.cancel(run_id) + receipt["cleanup_terminal"] = _wait(service, run_id, 15)["state"] + except Exception as cleanup_error: + receipt["cleanup_error"] = str(cleanup_error) + receipt["finished_at"] = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime()) + receipt_path = run_dir / "probe-receipt.json" + _write_json(receipt_path, receipt) + _lock_down(run_dir) + print(json.dumps({ + "status": receipt["status"], + "run_dir": str(run_dir), + "receipt_sha256": _sha256(receipt_path), + }, sort_keys=True)) + return 0 if receipt["status"] == "passed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/test/core/probes/m4_cross_surface.py b/test/core/probes/m4_cross_surface.py new file mode 100644 index 0000000..ea977a7 --- /dev/null +++ b/test/core/probes/m4_cross_surface.py @@ -0,0 +1,379 @@ +#!/usr/bin/env python3 +"""Prepare and finish one private M4 cross-surface MCP acceptance run.""" + +from __future__ import annotations + +import argparse +import asyncio +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import time +from typing import Any + +from devsquad.service import Service +from devsquad.store import request_hash + + +TERMINAL_STATES = {"succeeded", "failed", "cancelled", "timed_out"} + + +def _git(repo: Path, *arguments: str) -> str: + return subprocess.run( + ["git", "-C", str(repo), *arguments], + check=True, + text=True, + capture_output=True, + ).stdout.strip() + + +def _write_json(path: Path, value: dict[str, Any]) -> None: + path.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") + path.chmod(0o600) + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(65536), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _wait(service: Service, run_id: str, states: set[str], timeout: int = 20) -> dict[str, Any]: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status = service.status(run_id) + if status["state"] in states: + return status + time.sleep(0.05) + raise TimeoutError(f"run did not reach {sorted(states)}: {service.status(run_id)}") + + +def _routing_documents() -> tuple[dict[str, Any], dict[str, Any]]: + profiles = { + "schema_version": 1, + "profiles": [{ + "id": "m4-offline-reviewer", + "harness": "fixture", + "model_family": "offline-fixture", + "model_id": "m4-offline-review", + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "m4-offline", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["m4-cross-surface-probe"], + }], + "bindings": { + "review.deep": {"profile_id": "m4-offline-reviewer", "version": 1}, + }, + } + policy = { + "schema_version": 1, + "id": "m4-cross-surface-policy", + "version": 1, + "roles": {"reviewer": [{"kind": "alias", "id": "review.deep"}]}, + "task_classes": {"m4-cross-surface": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "m4-offline": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + return profiles, policy + + +def _make_repository(repo: Path) -> tuple[str, str]: + repo.mkdir() + _git(repo, "init", "-q") + _git(repo, "config", "user.email", "devsquad-probe@example.invalid") + _git(repo, "config", "user.name", "DevSquad Probe") + (repo / "src").mkdir() + (repo / "tests").mkdir() + (repo / "devsquad").mkdir() + (repo / "src/value.py").write_text("VALUE = 'base'\n") + (repo / "tests/test_value.py").write_text("# bounded fixture\n") + profiles, policy = _routing_documents() + _write_json(repo / "devsquad/profiles.json", profiles) + _write_json(repo / "devsquad/policy.json", policy) + _git(repo, "add", ".") + _git(repo, "commit", "-qm", "probe base") + base = _git(repo, "rev-parse", "HEAD") + (repo / "src/value.py").write_text("VALUE = 'candidate'\n") + _git(repo, "add", "src/value.py") + _git(repo, "commit", "-qm", "probe candidate") + return base, _git(repo, "rev-parse", "HEAD") + + +def _task(repo: Path, base: str, target: str) -> dict[str, Any]: + return { + "schema_version": 1, + "project": { + "repo_path": str(repo), "base_ref": base, "target_ref": target, + }, + "workflow": "branch-review", + "goal": "Prove that local MCP clients share one durable saved run.", + "task_class": "m4-cross-surface", + "acceptance": [{ + "id": "shared-run", + "description": "Each client observes the same run and bound evidence.", + "evidence_kind": "review", + }], + "checks": [{ + "id": "offline-check", + "argv": [sys.executable, "-c", "print('m4 check passed')"], + "cwd": ".", + "timeout_seconds": 15, + "required_to_pass": True, + }], + "scope": {"read_paths": ["src", "tests"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + }, + "budget": { + "wall_seconds": 120, + "max_worker_invocations": 1, + "max_revisions": 0, + "max_fallbacks_per_step": 0, + }, + "review": {"mode": "standard"}, + "origin": {"surface": "terminal-m4-probe"}, + } + + +def prepare(args: argparse.Namespace) -> int: + stamp = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime()) + run_dir = args.output_dir.expanduser().resolve() / f"m4-cross-surface-{stamp}" + run_dir.mkdir(parents=True, mode=0o700, exist_ok=False) + run_dir.chmod(0o700) + repo = run_dir / "repository" + base, target = _make_repository(repo) + task = _task(repo, base, target) + fixture = { + "verdict": "findings", + "summary": "The bounded fixture changes the configured value.", + "findings": [{ + "id": "M4-1", + "severity": "low", + "title": "Fixture value changed", + "description": "The candidate intentionally changes the fixture value.", + "path": "src/value.py", + "start_line": 1, + "end_line": 1, + "evidence": "The candidate contains VALUE = 'candidate'.", + }], + } + service = Service(args.runtime) + started = service.start( + task, + f"m4-cross-surface-{stamp}", + _internal_review_fixture=fixture, + ) + waiting = _wait(service, started["run_id"], {"awaiting_host", "failed"}) + if waiting["state"] != "awaiting_host": + raise RuntimeError(f"offline run did not produce a handoff: {waiting}") + state = { + "schema_version": 1, + "run_dir": str(run_dir), + "runtime": str(args.runtime.resolve()), + "squad_executable": str(args.squad.resolve(strict=True)), + "run_id": started["run_id"], + "terminal_start": started, + "waiting": waiting, + "base_oid": base, + "target_oid": target, + } + state_path = run_dir / "probe-state.json" + _write_json(state_path, state) + print(json.dumps({ + "state_file": str(state_path), + "run_id": started["run_id"], + "state": waiting["state"], + "version": waiting["version"], + }, sort_keys=True)) + return 0 + + +def _data(result: Any) -> dict[str, Any]: + payload = result.structured_content + if not isinstance(payload, dict) or payload.get("ok") is not True: + raise RuntimeError(f"MCP operation failed: {payload}") + return payload["data"] + + +async def _client( + state: dict[str, Any], surface: str, +): + from mcp import Client, StdioServerParameters + + parameters = StdioServerParameters( + command=state["squad_executable"], + args=[ + "mcp", "serve", "--runtime-dir", state["runtime"], + "--surface", surface, + ], + ) + return Client(parameters) + + +async def _complete(state: dict[str, Any]) -> dict[str, Any]: + run_id = state["run_id"] + codex = await _client(state, "codex-app") + async with codex: + codex_status = _data(await codex.call_tool("squad_status", {"run_id": run_id})) + codex_events = _data(await codex.call_tool( + "squad_events", {"run_id": run_id, "after": 0, "limit": 100}, + )) + + claude = await _client(state, "claude-code") + async with claude: + claude_status = _data(await claude.call_tool("squad_status", {"run_id": run_id})) + claude_events = _data(await claude.call_tool( + "squad_events", {"run_id": run_id, "after": 0, "limit": 100}, + )) + claimed = _data(await claude.call_tool("squad_handoff_claim", { + "run_id": run_id, + "expected_version": claude_status["version"], + "owner": "m4-claude-client", + })) + + competing = await _client(state, "antigravity") + async with competing: + fenced_result = await competing.call_tool("squad_handoff_claim", { + "run_id": run_id, + "expected_version": claude_status["version"], + "owner": "m4-competing-client", + }) + fenced = fenced_result.structured_content + + packet = claimed["handoff"]["packet"] + decision_body = { + "schema_version": 1, + "submission_id": "m4-cross-surface-accept", + "disposition": "accept", + "reason": "Accept the bounded offline evidence after cross-client inspection.", + "evidence_refs": [{ + "artifact_id": item["artifact_id"], "sha256": item["sha256"], + } for item in packet["artifacts"]], + } + decision = {**decision_body, "submission_hash": request_hash(decision_body)} + claude_complete = await _client(state, "claude-code") + async with claude_complete: + completed = _data(await claude_complete.call_tool( + "squad_handoff_complete", + {"run_id": run_id, "claim": claimed["claim"], "decision": decision}, + )) + + codex_result = await _client(state, "codex-app") + async with codex_result: + result = _data(await codex_result.call_tool( + "squad_result", {"run_id": run_id, "preview_bytes": 0}, + )) + final_events = _data(await codex_result.call_tool( + "squad_events", {"run_id": run_id, "after": 0, "limit": 100}, + )) + return { + "codex_status": codex_status, + "claude_status": claude_status, + "preclaim_ledgers_identical": codex_events == claude_events, + "claim": claimed["claim"], + "packet_sha256": claimed["handoff"]["packet_sha256"], + "candidate_sha256": packet["candidate_sha256"], + "competing_claim": fenced, + "completed": completed, + "result": result, + "final_events": final_events, + } + + +def complete(args: argparse.Namespace) -> int: + state_path = args.state_file.expanduser().resolve(strict=True) + state = json.loads(state_path.read_text()) + evidence = asyncio.run(_complete(state)) + if not evidence["preclaim_ledgers_identical"]: + raise RuntimeError("Codex and Claude clients observed different ledgers") + fenced = evidence["competing_claim"] + if (not isinstance(fenced, dict) or fenced.get("ok") is not False + or fenced.get("error", {}).get("code") != "CONFLICT"): + raise RuntimeError(f"competing host was not fenced: {fenced}") + if evidence["completed"]["state"] != "succeeded": + raise RuntimeError(f"completion did not succeed: {evidence['completed']}") + result = evidence["result"] + if result["run_id"] != state["run_id"] or not result["ready"]: + raise RuntimeError(f"terminal result is not ready: {result}") + receipt = { + "schema_version": 1, + "status": "passed", + "run_id": state["run_id"], + "terminal_start": state["terminal_start"], + "clients": ["codex-app", "claude-code", "antigravity"], + "same_run_id": ( + evidence["codex_status"]["run_id"] + == evidence["claude_status"]["run_id"] + == result["run_id"] + == state["run_id"] + ), + "preclaim_ledgers_identical": True, + "packet_sha256": evidence["packet_sha256"], + "candidate_sha256": evidence["candidate_sha256"], + "competing_claim_error": fenced["error"]["code"], + "terminal_state": result["state"], + "event_count": len(evidence["final_events"]["events"]), + "artifacts": [{ + key: artifact[key] + for key in ("id", "name", "sha256", "byte_size") + } for artifact in result["artifacts"]], + } + receipt_path = Path(state["run_dir"]) / "probe-receipt.json" + _write_json(receipt_path, receipt) + print(json.dumps({ + "status": "passed", + "run_id": state["run_id"], + "receipt": str(receipt_path), + "receipt_sha256": _sha256(receipt_path), + }, sort_keys=True)) + return 0 + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--runtime", type=Path, + default=Path.home() / ".devsquad/runtime", + ) + parser.add_argument( + "--squad", type=Path, + default=Path.home() / ".local/bin/squad", + ) + parser.add_argument( + "--output-dir", type=Path, + default=Path.home() / ".devsquad/private-probes", + ) + subparsers = parser.add_subparsers(dest="operation", required=True) + prepare_parser = subparsers.add_parser("prepare") + prepare_parser.set_defaults(function=prepare) + complete_parser = subparsers.add_parser("complete") + complete_parser.add_argument("--state-file", type=Path, required=True) + complete_parser.set_defaults(function=complete) + args = parser.parse_args() + args.runtime = args.runtime.expanduser().resolve() + args.squad = args.squad.expanduser().resolve(strict=True) + args.output_dir = args.output_dir.expanduser().resolve() + args.output_dir.mkdir(parents=True, mode=0o700, exist_ok=True) + return args.function(args) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/test/core/probes/native_codex_smoke.py b/test/core/probes/native_codex_smoke.py new file mode 100644 index 0000000..3b253bf --- /dev/null +++ b/test/core/probes/native_codex_smoke.py @@ -0,0 +1,221 @@ +#!/usr/bin/env python3 +"""Opt-in, bounded live smoke for the production Codex native M1 path.""" + +from __future__ import annotations + +import argparse +import errno +import hashlib +import json +import os +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile +import time +from typing import Any + +CORE_SRC = Path(__file__).resolve().parents[3] / "plugin" / "core" / "src" +sys.path.insert(0, str(CORE_SRC)) + +from devsquad.adapters import AdapterManifest, harness_version, prepare_native_codex_from_catalog +from devsquad.catalog import update_last_good +from devsquad.contracts import ExecutionIdentity, LaunchSpec, SCHEMA_VERSION +from devsquad.codex_protocol import ( + JsonLinePeer, + NativeTurnState, + discover_models, + initialize_request, + initialized_notification, + receive_response, + thread_start_request, + turn_start_request, +) + + +class RecordingPeer: + def __init__(self, peer: JsonLinePeer, transcript: Any): + self.peer, self.transcript = peer, transcript + + def send(self, message: dict[str, Any]) -> None: + self.transcript.write(json.dumps({"direction": "send", "message": message}) + "\n") + self.transcript.flush() + self.peer.send(message) + + def receive(self, timeout_seconds: float) -> dict[str, Any]: + message = self.peer.receive(timeout_seconds) + self.transcript.write(json.dumps({"direction": "receive", "message": message}) + "\n") + self.transcript.flush() + return message + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(65536), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _stop(process: subprocess.Popen[str]) -> dict[str, Any]: + pgid = os.getpgid(process.pid) if process.poll() is None else process.pid + if process.poll() is None: + os.killpg(pgid, 15) + try: + process.wait(timeout=2) + except subprocess.TimeoutExpired: + os.killpg(pgid, 9) + process.wait(timeout=2) + try: + os.killpg(pgid, 0) + except OSError as exc: + if exc.errno != errno.ESRCH: + raise + absent = True + else: + absent = False + os.killpg(pgid, 9) + deadline = time.monotonic() + 2 + while time.monotonic() < deadline: + try: + os.killpg(pgid, 0) + except OSError as exc: + if exc.errno != errno.ESRCH: + raise + absent = True + break + time.sleep(0.05) + return {"exit_code": process.returncode, "process_group": pgid, "group_absent": absent} + + +def _server(spec: Any, stderr_path: Path, transcript_path: Path): + stderr_stream = stderr_path.open("w", encoding="utf-8") + transcript = transcript_path.open("w", encoding="utf-8") + environment = os.environ.copy() + environment.update(spec.environment) + process = subprocess.Popen( + list(spec.argv), cwd=spec.cwd, env=environment, stdin=subprocess.PIPE, + stdout=subprocess.PIPE, stderr=stderr_stream, text=True, bufsize=1, + start_new_session=True, + ) + assert process.stdin is not None and process.stdout is not None + peer = RecordingPeer(JsonLinePeer(process.stdout, process.stdin), transcript) + return process, peer, stderr_stream, transcript + + +def _initialize(peer: RecordingPeer, timeout: float) -> None: + peer.send(initialize_request(1)) + response = receive_response(peer, 1, timeout_seconds=timeout) + if "error" in response: + raise RuntimeError(f"initialize failed: {response['error']}") + peer.send(initialized_notification()) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--run-live", action="store_true", help="required acknowledgement for a live subscription-backed probe") + parser.add_argument("--output-dir", type=Path, default=Path.home() / ".devsquad" / "private-probes") + parser.add_argument("--timeout", type=int, default=30) + args = parser.parse_args() + if not args.run_live: + parser.error("--run-live is required") + if args.timeout < 5 or args.timeout > 60: + parser.error("--timeout must be between 5 and 60 seconds") + + stamp = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime()) + revision = subprocess.run(["git", "rev-parse", "HEAD"], cwd=CORE_SRC, text=True, capture_output=True, check=True).stdout.strip() + run_dir = args.output_dir.expanduser().resolve() / f"native-codex-{stamp}-{revision[:12]}" + run_dir.mkdir(parents=True, mode=0o700, exist_ok=False) + os.chmod(run_dir, 0o700) + receipt: dict[str, Any] = {"started_at": stamp, "revision": revision, "status": "failed", "cleanup": []} + manifest_path = CORE_SRC.parent / "adapters" / "codex" / "adapter.json" + manifest = AdapterManifest.load(manifest_path) + binary = manifest.resolve_binary() + if not binary: + raise RuntimeError("codex executable is unavailable") + version = harness_version(binary) + if not version: + raise RuntimeError("could not determine codex version") + + try: + with tempfile.TemporaryDirectory(prefix="devsquad-native-smoke-") as workspace_name: + workspace = Path(workspace_name) + subprocess.run(["git", "init", "-q", str(workspace)], check=True) + bootstrap = LaunchSpec( + SCHEMA_VERSION, "codex", "native_protocol", + (binary, "app-server", "--listen", "stdio://"), str(workspace), None, + args.timeout, + ExecutionIdentity("codex", version, "openai", None, None, None, (), "read_only", None, "verified"), + {"DEVSQUAD_WORKER": "1"}, + ) + first, peer, stderr_stream, transcript = _server(bootstrap, run_dir / "discovery.stderr.log", run_dir / "discovery.jsonl") + try: + _initialize(peer, args.timeout) + models = discover_models(peer, first_request_id=10, timeout_seconds=args.timeout) + finally: + receipt["cleanup"].append({"discovery": _stop(first)}) + transcript.close(); stderr_stream.close() + + snapshot = update_last_good(run_dir / "catalog.json", harness="codex", version=version, models=models, complete=True) + candidates = [m for m in snapshot["models"] if m.get("supported_efforts")] + if not candidates: + raise RuntimeError("discovery returned no model with verified effort metadata") + selected = next((m for m in candidates if m.get("is_default")), candidates[0]) + supported = selected["supported_efforts"] + effort = next((name for name in ("minimal", "low", "medium", "high", "xhigh") if name in supported), supported[0]) + spec = prepare_native_codex_from_catalog( + manifest, snapshot, cwd=str(workspace), model=selected["id"], effort=effort, + permission="read_only", timeout_seconds=args.timeout, harness_version_value=version, + ) + second, peer, stderr_stream, transcript = _server(spec, run_dir / "turn.stderr.log", run_dir / "turn.jsonl") + try: + _initialize(peer, args.timeout) + peer.send(thread_start_request(20, cwd=str(workspace), model=selected["id"], permission="read_only")) + thread_response = receive_response(peer, 20, timeout_seconds=args.timeout) + result = thread_response.get("result", {}) + thread = result.get("thread", {}) if isinstance(result, dict) else {} + thread_id = thread.get("id") or result.get("threadId") + if not isinstance(thread_id, str) or not thread_id: + raise RuntimeError(f"thread/start returned no thread id: {thread_response}") + peer.send(turn_start_request(21, thread_id=thread_id, prompt="Reply with exactly DEVSQUAD_M1_NATIVE_OK. Do not use tools.", model=selected["id"], effort=effort, cwd=str(workspace), permission="read_only")) + early_notifications: list[dict[str, Any]] = [] + turn_response = receive_response(peer, 21, timeout_seconds=args.timeout, on_notification=early_notifications.append) + turn_result = turn_response.get("result", {}) + turn = turn_result.get("turn", {}) if isinstance(turn_result, dict) else {} + turn_id = turn.get("id") or turn_result.get("turnId") + if not isinstance(turn_id, str) or not turn_id: + raise RuntimeError(f"turn/start returned no turn id: {turn_response}") + state = NativeTurnState(thread_id=thread_id, turn_id=turn_id) + for notification in early_notifications: + state.consume(notification) + deadline = time.monotonic() + args.timeout + while not state.terminal: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("native turn did not reach correlated terminal state") + state.consume(peer.receive(remaining)) + output = state.final_output().strip() + if state.terminal_status != "completed" or output != "DEVSQUAD_M1_NATIVE_OK": + raise RuntimeError(f"native verdict was not successful: status={state.terminal_status!r}, output={output!r}") + receipt.update({"status": "passed", "harness_version": version, "model": selected["id"], "effort": effort, "permission": "read_only", "terminal_status": state.terminal_status}) + finally: + receipt["cleanup"].append({"turn": _stop(second)}) + transcript.close(); stderr_stream.close() + except Exception as exc: + receipt["error_type"] = type(exc).__name__ + receipt["error"] = str(exc) + finally: + receipt["finished_at"] = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime()) + for path in run_dir.iterdir(): + if path.is_file(): os.chmod(path, 0o600) + receipt["private_artifacts"] = {p.name: _sha256(p) for p in sorted(run_dir.iterdir()) if p.is_file()} + receipt_path = run_dir / "receipt.json" + receipt_path.write_text(json.dumps(receipt, indent=2, sort_keys=True) + "\n") + os.chmod(receipt_path, 0o600) + print(json.dumps({"status": receipt["status"], "run_dir": str(run_dir), "receipt_sha256": _sha256(receipt_path)})) + return 0 if receipt["status"] == "passed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/test/core/test_adapters.py b/test/core/test_adapters.py new file mode 100644 index 0000000..0d7fc0c --- /dev/null +++ b/test/core/test_adapters.py @@ -0,0 +1,137 @@ +from __future__ import annotations + +import json +import os +import stat +import sys +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +CORE = Path(__file__).resolve().parents[2] / "plugin" / "core" +sys.path.insert(0, str(CORE / "src")) + +from devsquad.adapters import AdapterManifest, classify_cli, prepare_cli +from devsquad.contracts import ProfileUnsupported + + +class ClaudeAdapterTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.binary = Path(self.temp.name) / "claude" + self.binary.write_text("#!/bin/sh\nexit 0\n") + self.binary.chmod(self.binary.stat().st_mode | stat.S_IXUSR) + self.manifest = AdapterManifest.load( + CORE / "adapters" / "claude" / "adapter.json" + ) + + def prepare( + self, + *, + permission: str = "read_only", + model: str | None = None, + effort: str | None = None, + version: str | None = None, + ): + manifest = self.manifest + if effort is not None: + manifest = manifest.with_model_efforts({model: (effort,)}) + with patch.dict(os.environ, {"PATH": str(self.binary.parent)}): + return prepare_cli( + manifest, + prompt="Fix src/My Parser.ts without delegating.", + cwd=self.temp.name, + model=model, + effort=effort, + permission=permission, + timeout_seconds=17, + harness_version_value=version, + ) + + def test_read_only_argv_is_structured_and_has_no_bypass_or_delegation(self): + spec = self.prepare() + self.assertEqual(spec.adapter, "claude") + self.assertEqual(spec.transport, "cli_exec") + self.assertEqual(spec.requested.permissions, "read_only") + self.assertEqual(spec.requested.tools, ("Read", "Glob", "Grep")) + self.assertIn("plan", spec.argv) + self.assertIn("Read,Glob,Grep", spec.argv) + self.assertIn('{"mcpServers":{}}', spec.argv) + self.assertIn("Fix src/My Parser.ts without delegating.", spec.argv) + self.assertEqual(spec.argv[-2:], ("--", "Fix src/My Parser.ts without delegating.")) + self.assertNotIn("--dangerously-skip-permissions", spec.argv) + self.assertNotIn("Agent", ",".join(spec.argv)) + self.assertNotIn("Bash", ",".join(spec.argv)) + + def test_workspace_write_argv_and_version_are_exactly_verified(self): + spec = self.prepare( + permission="workspace_write", + model="claude-fixture-1", + effort="high", + version="2.1.220 (Claude Code)", + ) + self.assertIn("acceptEdits", spec.argv) + self.assertIn("Read,Glob,Grep,Edit,Write", spec.argv) + self.assertEqual(spec.requested.model, "claude-fixture-1") + self.assertEqual(spec.requested.effort, "high") + self.assertEqual(spec.requested.harness_version, "2.1.220 (Claude Code)") + self.assertEqual(spec.requested.verification, "verified") + with self.assertRaisesRegex(ProfileUnsupported, "unverified claude"): + self.prepare(version="2.2.0 (Claude Code)") + + def test_structured_result_and_faults_are_not_conflated_with_acceptance(self): + spec = self.prepare() + success = classify_cli( + spec, + returncode=0, + stdout=json.dumps({ + "type": "result", + "subtype": "success", + "is_error": False, + "result": "bounded implementation summary", + "session_id": "fixture-session", + }), + stderr="", + ) + self.assertEqual(success.execution_status, "succeeded") + self.assertEqual(success.acceptance_status, "not_evaluated") + + auth = classify_cli( + spec, + returncode=0, + stdout=json.dumps({ + "type": "result", + "is_error": True, + "result": "401 quota authorization required", + }), + stderr="", + ) + self.assertEqual(auth.execution_status, "failed") + self.assertEqual(auth.error_code, "AUTH_ERROR") + + denied = classify_cli( + spec, + returncode=0, + stdout=json.dumps({ + "type": "result", + "is_error": True, + "result": "tool use denied", + }), + stderr="", + ) + self.assertEqual(denied.execution_status, "denied") + self.assertEqual(denied.error_code, "CLI_ERROR") + + malformed = classify_cli( + spec, + returncode=0, + stdout=json.dumps({"type": "system", "subtype": "init"}), + stderr="", + ) + self.assertEqual(malformed.execution_status, "malformed") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py new file mode 100644 index 0000000..a91b5a6 --- /dev/null +++ b/test/core/test_capacity.py @@ -0,0 +1,346 @@ +import copy +from concurrent.futures import ThreadPoolExecutor +from datetime import datetime, timedelta, timezone +import math +from pathlib import Path +import subprocess +import sys +import tempfile +import threading +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.capacity import derive_pool_capacity, validate_observation +from devsquad.contracts import ContractError +from devsquad.store import ConflictError, SchemaVersionError, Store, SUPPORTED_SCHEMA_VERSION + + +NOW = datetime(2026, 9, 27, 15, 0, tzinfo=timezone.utc) + + +def observation( + observation_id="obs-short", + *, + window_id="short", + observed_at=NOW - timedelta(minutes=1), + expires_at=NOW + timedelta(minutes=10), + source="native_reported", + used=20, + limit=100, + unit="percent", + confidence="confirmed", + applies_to=None, +): + return { + "schema_version": 1, + "observation_id": observation_id, + "pool_id": "shared-pool", + "window_id": window_id, + "applies_to": applies_to or { + "harnesses": [], + "model_families": [], + "model_ids": [], + }, + "observed_at": observed_at.isoformat(), + "expires_at": expires_at.isoformat(), + "source": source, + "used": used, + "limit": limit, + "unit": unit, + "resets_at": (NOW + timedelta(hours=1)).isoformat(), + "confidence": confidence, + } + + +class CapacityContractTest(unittest.TestCase): + @staticmethod + def make_repository(path): + subprocess.run(["git", "init", "-q", str(path)], check=True) + subprocess.run( + ["git", "-C", str(path), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(path), "config", "user.name", "Test"], + check=True, + ) + (path / "README").write_text("fixture\n") + subprocess.run(["git", "-C", str(path), "add", "README"], check=True) + subprocess.run(["git", "-C", str(path), "commit", "-qm", "base"], check=True) + + def test_validation_is_strict_and_normalizes_scopes(self): + value = observation(applies_to={ + "harnesses": ["grok", "codex"], + "model_families": [], + "model_ids": ["model-b", "model-a"], + }) + normalized = validate_observation(value, now=NOW) + self.assertEqual(normalized["applies_to"]["harnesses"], ["codex", "grok"]) + self.assertEqual(normalized["applies_to"]["model_ids"], ["model-a", "model-b"]) + value["applies_to"]["harnesses"].append("claude") + self.assertEqual(normalized["applies_to"]["harnesses"], ["codex", "grok"]) + + invalid = copy.deepcopy(normalized) + invalid["extra"] = True + with self.assertRaisesRegex(ContractError, "fields invalid"): + validate_observation(invalid, now=NOW) + invalid = observation(used=True) + with self.assertRaisesRegex(ContractError, "finite non-negative"): + validate_observation(invalid, now=NOW) + invalid = observation(used=math.inf) + with self.assertRaisesRegex(ContractError, "finite non-negative"): + validate_observation(invalid, now=NOW) + invalid = observation(used=101) + with self.assertRaisesRegex(ContractError, "at most 100"): + validate_observation(invalid, now=NOW) + invalid = observation() + invalid["source"] = [] + with self.assertRaisesRegex(ContractError, "source"): + validate_observation(invalid, now=NOW) + invalid = observation(observed_at=NOW + timedelta(minutes=6)) + invalid["expires_at"] = (NOW + timedelta(minutes=20)).isoformat() + with self.assertRaisesRegex(ContractError, "clock skew"): + validate_observation(invalid, now=NOW) + invalid = observation() + invalid["observed_at"] = "2026-09-27T15:00:00" + with self.assertRaisesRegex(ContractError, "timezone"): + validate_observation(invalid, now=NOW) + + def test_fresh_exhausted_weekly_window_beats_available_short_window(self): + short = observation() + weekly = observation( + "obs-weekly", window_id="weekly", used=100, limit=100, + ) + snapshot = derive_pool_capacity( + "shared-pool", [short, weekly], in_flight=1, now=NOW, + ) + self.assertEqual(snapshot["status"], "exhausted") + self.assertEqual(snapshot["in_flight"], 1) + self.assertEqual( + [window["status"] for window in snapshot["windows"]], + ["available", "exhausted"], + ) + self.assertEqual(snapshot["reasons"], ["window_exhausted:weekly"]) + + def test_stale_estimated_or_unknown_window_never_becomes_zero(self): + stale = observation( + "obs-weekly", window_id="weekly", + expires_at=NOW - timedelta(seconds=1), used=100, limit=100, + ) + snapshot = derive_pool_capacity( + "shared-pool", [observation(), stale], now=NOW, + ) + self.assertEqual(snapshot["status"], "unknown") + self.assertIn("stale_observation:weekly", snapshot["reasons"]) + + estimated = observation( + "obs-estimated", source="estimated", confidence="estimated", + used=100, limit=100, + ) + snapshot = derive_pool_capacity("shared-pool", [estimated], now=NOW) + self.assertEqual(snapshot["status"], "unknown") + self.assertFalse(snapshot["windows"][0]["authoritative"]) + + unknown = observation("obs-unknown", used=None, limit=None) + snapshot = derive_pool_capacity("shared-pool", [unknown], now=NOW) + self.assertEqual(snapshot["status"], "unknown") + self.assertEqual(snapshot["windows"][0]["reason"], "unknown_measurement") + + def test_latest_observation_per_scoped_window_is_used(self): + scope = { + "harnesses": ["codex"], + "model_families": ["gpt"], + "model_ids": [], + } + older = observation( + "older", used=100, limit=100, applies_to=scope, + observed_at=NOW - timedelta(minutes=2), + ) + newer = observation( + "newer", used=10, limit=100, applies_to=scope, + observed_at=NOW - timedelta(minutes=1), + ) + target = {"harness": "codex", "model_family": "gpt", "model_id": "gpt-5"} + snapshot = derive_pool_capacity( + "shared-pool", [older, newer], target=target, now=NOW, + ) + self.assertEqual(snapshot["status"], "available") + self.assertEqual([item["observation_id"] for item in snapshot["windows"]], ["newer"]) + + other = {"harness": "claude", "model_family": "claude", "model_id": "opus"} + snapshot = derive_pool_capacity( + "shared-pool", [older, newer], target=other, now=NOW, + ) + self.assertEqual(snapshot["status"], "unknown") + self.assertEqual(snapshot["reasons"], ["no_applicable_observations"]) + + def test_migration_nine_creates_capacity_ledger(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + store = Store(path / "state.sqlite3", path / "artifacts") + self.addCleanup(store.close) + version = store.connection.execute( + "SELECT MAX(version) FROM schema_migrations", + ).fetchone()[0] + self.assertEqual(version, SUPPORTED_SCHEMA_VERSION) + tables = { + row[0] for row in store.connection.execute( + "SELECT name FROM sqlite_master WHERE type='table'", + ) + } + self.assertTrue({"pool_observations", "pool_reservations"} <= tables) + + def test_store_observation_is_replay_safe_and_stale_evidence_stays_visible(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + store = Store(path / "state.sqlite3", path / "artifacts") + self.addCleanup(store.close) + stale = observation( + "persisted-stale", + expires_at=NOW - timedelta(seconds=1), + used=100, + limit=100, + ) + first = store.record_pool_observation(stale, now=NOW) + replay = store.record_pool_observation(stale, now=NOW) + self.assertFalse(first["replayed"]) + self.assertTrue(replay["replayed"]) + snapshot = store.capacity_snapshot("shared-pool", now=NOW) + self.assertEqual(snapshot["status"], "unknown") + self.assertEqual(snapshot["windows"][0]["reason"], "stale_observation") + + changed = copy.deepcopy(stale) + changed["used"] = 99 + with self.assertRaisesRegex(ConflictError, "different evidence"): + store.record_pool_observation(changed, now=NOW) + + def test_non_attempt_reservation_is_bounded_and_explicitly_reconciled(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + repository = path / "repo" + self.make_repository(repository) + store = Store(path / "state.sqlite3", path / "artifacts") + self.addCleanup(store.close) + claim = store.claim_start(repository, "classifier-capacity", {}, "owner") + reserved = store.reserve_pool_capacity( + claim.run_id, "shared-pool", "classifier", now=NOW, + ) + self.assertEqual(store.active_pool_counts(), {"shared-pool": 1}) + with self.assertRaisesRegex(ConflictError, "concurrency"): + store.reserve_pool_capacity( + claim.run_id, "shared-pool", "classifier", now=NOW, + ) + reconciled = store.reconcile_pool_reservation( + reserved["reservation_id"], "classifier_finished", now=NOW, + ) + self.assertFalse(reconciled["replayed"]) + self.assertEqual(store.active_pool_counts(), {}) + replay = store.reconcile_pool_reservation( + reserved["reservation_id"], "classifier_finished", now=NOW, + ) + self.assertTrue(replay["replayed"]) + + def test_two_projects_racing_for_unknown_pool_create_one_reservation(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + repositories = [path / "repo-a", path / "repo-b"] + for repository in repositories: + self.make_repository(repository) + database, artifacts = path / "state.sqlite3", path / "artifacts" + store = Store(database, artifacts) + run_ids = [ + store.claim_start(repository, f"run-{index}", {}, "owner").run_id + for index, repository in enumerate(repositories) + ] + store.close() + barrier = threading.Barrier(2) + + def reserve(run_id): + connection = Store(database, artifacts) + try: + barrier.wait() + return connection.reserve_pool_capacity( + run_id, "shared-pool", "qualification", now=NOW, + )["reservation_id"] + except ConflictError: + return "conflict" + finally: + connection.close() + + with ThreadPoolExecutor(max_workers=2) as executor: + results = list(executor.map(reserve, run_ids)) + self.assertEqual(results.count("conflict"), 1) + winner = next(result for result in results if result != "conflict") + store = Store(database, artifacts) + self.addCleanup(store.close) + self.assertEqual(store.active_pool_counts(), {"shared-pool": 1}) + store.reconcile_pool_reservation(winner, "qualification_finished", now=NOW) + + def test_schema_eight_active_attempt_is_backfilled_and_reconciled(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + database = path / "state.sqlite3" + import sqlite3 + + connection = sqlite3.connect(database) + migrations = ROOT / "plugin/core/src/devsquad/migrations" + for version in range(1, 9): + name = next(migrations.glob(f"{version:03d}_*.sql")) + connection.executescript(name.read_text()) + connection.execute( + "INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)", + (version, NOW.isoformat()), + ) + connection.execute( + "INSERT INTO projects(id,git_common_dir,created_at) VALUES('p','/tmp/p',?)", + (NOW.isoformat(),), + ) + connection.execute( + "INSERT INTO runs(id,project_id,idempotency_key,request_hash," + "submitted_request,state,version,created_at,updated_at,worktree_path) " + "VALUES('r','p','key','hash','{}','running',1,?,?, '/tmp/w')", + (NOW.isoformat(), NOW.isoformat()), + ) + connection.execute( + "INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token," + "status,heartbeat_at,package_digest,created_at,role,account_pool_id,profile_id) " + "VALUES('a','r','p','/tmp/w','token','running',?,'package',?," + "'reviewer','shared-pool','profile-a')", + (NOW.isoformat(), NOW.isoformat()), + ) + connection.commit() + connection.close() + + with self.assertRaisesRegex(SchemaVersionError, "active/recoverable"): + Store(database, path / "artifacts") + # Keep the migration-9 SQL backfill unit coverage independently + # of public upgrades, which must now defer for old active runs. + connection = sqlite3.connect(database) + connection.executescript(next(migrations.glob("009_*.sql")).read_text()) + connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(9,?)", (NOW.isoformat(),)) + connection.commit() + connection.close() + with patch("devsquad.store.SUPPORTED_SCHEMA_VERSION", 9): + store = Store(database, path / "artifacts") + self.addCleanup(store.close) + self.assertEqual(store.active_pool_counts(), {"shared-pool": 1}) + row = store.connection.execute( + "SELECT purpose,profile_id FROM pool_reservations WHERE attempt_id='a'", + ).fetchone() + self.assertEqual(tuple(row), ("attempt", "profile-a")) + store.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id='a'", + (NOW.isoformat(),), + ) + self.assertEqual(store.active_pool_counts(), {}) + store.connection.execute("UPDATE runs SET state='failed' WHERE id='r'") + upgraded = Store(database, path / "artifacts") + self.addCleanup(upgraded.close) + self.assertEqual(upgraded.active_pool_counts(), {}) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_check_integrity_runtime.py b/test/core/test_check_integrity_runtime.py new file mode 100644 index 0000000..dd46409 --- /dev/null +++ b/test/core/test_check_integrity_runtime.py @@ -0,0 +1,342 @@ +"""Candidate-integrity gates exercised through durable public service runs.""" + +from __future__ import annotations + +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import time +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.service import Service +from devsquad.store import ConflictError, Store, request_hash +import test_delivery_workflow as delivery_fixtures + + +class PublicCheckIntegrityTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-check-integrity-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + self.git("config", "user.email", "fixture@example.invalid") + self.git("config", "user.name", "Fixture") + for name in ("src", "tests", "devsquad"): + (self.repo / name).mkdir() + (self.repo / "src/app.py").write_text("VALUE = 'base'\n") + (self.repo / "tests/test_app.py").write_text("# fixture\n") + profiles, policy = delivery_fixtures.DeliveryWorkspaceTest.delivery_routing_documents() + policy["task_classes"]["fixture-review-small"] = "proven" + for name, document in (("profiles", profiles), ("policy", policy)): + (self.repo / "devsquad" / f"{name}.json").write_text(json.dumps(document)) + self.git("add", ".") + self.git("commit", "-qm", "baseline") + self.baseline = self.git("rev-parse", "HEAD").strip() + (self.repo / "src/app.py").write_text("VALUE = 'wrong'\n") + self.git("add", "src/app.py") + self.git("commit", "-qm", "candidate needing a fix") + self.target = self.git("rev-parse", "HEAD").strip() + self.original = { + "head": self.target, + "index": self.git("write-tree"), + "status": self.git("status", "--porcelain"), + "refs": self.git("show-ref"), + "source": (self.repo / "src/app.py").read_bytes(), + } + self.service = Service(self.runtime) + self.run_ids = [] + self.addCleanup(self.cancel_unfinished_runs) + + def git(self, *arguments): + return subprocess.run( + ["git", "-C", str(self.repo), *arguments], check=True, + text=True, capture_output=True, + ).stdout + + def cancel_unfinished_runs(self): + for run_id in self.run_ids: + if self.service.status(run_id)["state"] not in {"succeeded", "failed", "cancelled"}: + self.service.cancel(run_id) + + def task(self, workflow, mode, checks): + task = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + task["project"] = { + "repo_path": str(self.repo), "base_ref": self.baseline, + "target_ref": self.target if workflow == "branch-review" else self.baseline, + } + task["workflow"] = workflow + task["lead"] = {"mode": mode} + task["checks"] = checks + task["budget"] = { + "wall_seconds": 60, "max_worker_invocations": 3, + "max_revisions": 0, "max_fallbacks_per_step": 0, + } + if workflow == "issue-delivery": + task["task_class"] = "fixture-delivery-small" + task["scope"]["write_paths"] = ["src/app.py"] + return task + + @staticmethod + def check(check_id, script, *arguments, required=False, output_paths=None): + result = { + "id": check_id, + "argv": [sys.executable, "-c", script, *arguments], + "cwd": ".", "timeout_seconds": 10, "required_to_pass": required, + } + if output_paths is not None: + result["output_paths"] = output_paths + return result + + def wait_state(self, run_id, states): + deadline = time.monotonic() + 20 + while time.monotonic() < deadline: + status = self.service.status(run_id) + if status["state"] in states: + return status + time.sleep(0.05) + log = self.runtime / "private-logs" / f"{run_id}.supervisor.log" + self.fail( + f"run did not reach {states}: {self.service.status(run_id)}\n" + f"{log.read_text() if log.exists() else 'no supervisor log'}" + ) + + def start(self, workflow, mode, checks): + options = { + "_internal_review_fixture": { + "verdict": "clean", "summary": "Fixture review of the exact candidate.", + "findings": [], + }, + } + if workflow == "issue-delivery": + options["_internal_implementation_fixture"] = { + "writes": [{"path": "src/app.py", "content": "VALUE = 'wrong'\n"}], + "delay_seconds": 0, + } + if mode == "headless": + options["_internal_lead_fixture"] = { + "disposition": "accept", "reason": "Attempt to accept the saved evidence.", + } + started = self.service.start( + self.task(workflow, mode, checks), f"integrity-{workflow}-{mode}", **options, + ) + run_id = started["run_id"] + self.run_ids.append(run_id) + self.assertTrue(started["created"]) + if workflow == "issue-delivery": + deadline = time.monotonic() + 20 + while time.monotonic() < deadline: + status = self.service.status(run_id) + if status["state"] == "queued" and status.get("next_action") == "resume_candidate_review": + break + self.assertNotIn(status["state"], {"failed", "cancelled"}, status) + time.sleep(0.05) + else: + self.fail(f"delivery candidate was not frozen: {self.service.status(run_id)}") + self.assertTrue(self.service.resume(run_id)["launched"]) + return run_id + + @staticmethod + def decision(packet, submission_id, disposition): + body = { + "schema_version": 1, "submission_id": submission_id, + "disposition": disposition, "reason": "Exercise the candidate integrity gate.", + "evidence_refs": [ + {"artifact_id": item["artifact_id"], "sha256": item["sha256"]} + for item in packet["artifacts"] + ], + } + return {**body, "submission_hash": request_hash(body)} + + def receipt(self, run_id): + result = self.service.result(run_id) + artifact = next(( + item for item in result["artifacts"] + if item["name"] == "receipt.json" + ), None) + self.assertIsNotNone(artifact, { + "status": self.service.status(run_id), + "artifact_names": [item["name"] for item in result["artifacts"]], + }) + return json.loads(Path(artifact["path"]).read_text()) + + def snapshot(self, run_id): + store = Store(self.service.database, self.service.artifacts) + try: + return json.loads(store.run(run_id)["mutable_snapshot"]) + finally: + store.close() + + def assert_original_unchanged(self): + self.assertEqual(self.git("rev-parse", "HEAD").strip(), self.original["head"]) + self.assertEqual(self.git("write-tree"), self.original["index"]) + self.assertEqual(self.git("status", "--porcelain"), self.original["status"]) + self.assertEqual(self.git("show-ref"), self.original["refs"]) + self.assertEqual((self.repo / "src/app.py").read_bytes(), self.original["source"]) + + def exercise_mutation(self, workflow, mode, *, add_source=False): + second_marker = self.root / "second-check-ran" + mutated_path = "src/helper.py" if add_source else "src/app.py" + expected_output = { + "before": None if add_source else "VALUE = 'wrong'\n", + "after": "VALUE = 'fixed'\n", + } + checks = [ + self.check( + "mutating-report-only-check", + "from pathlib import Path; import json,sys; p=Path(sys.argv[1]); " + "before=p.read_text() if p.exists() else None; " + "p.write_text(\"VALUE = 'fixed'\\n\"); " + "print(json.dumps({'before':before,'after':p.read_text()}))", + mutated_path, + ), + self.check( + "must-not-use-contaminated-source", + "from pathlib import Path; import sys; " + "Path(sys.argv[1]).write_text(Path(sys.argv[2]).read_text())", + str(second_marker), mutated_path, + ), + ] + run_id = self.start(workflow, mode, checks) + # Headless mode traverses awaiting_host while it queues its own lead. + # Only host mode may inspect that intermediate packet; headless evidence + # is the terminal receipt, which is not published at the transition. + states = {"succeeded", "failed"} + if mode == "host": + states.add("awaiting_host") + status = self.wait_state(run_id, states) + if mode == "host": + self.assertEqual(status["state"], "awaiting_host", status) + claimed = self.service.handoff_claim(run_id, status["version"], "integrity-host") + evidence = claimed["handoff"]["packet"] + else: + evidence = self.receipt(run_id) + self.assertEqual(status["state"], "failed", "headless lead accepted changed source") + + # A successful shell exit is not a successful check of the frozen source. + self.assertIsInstance(evidence["evaluation"], dict, evidence) + self.assertFalse(evidence["evaluation"]["accept_allowed"]) + self.assertFalse(second_marker.exists(), "later check ran on mutated candidate source") + self.assertEqual(evidence["checks"][0]["status"], "invalidated") + self.assertEqual(evidence["checks"][0]["returncode"], 0) + self.assertEqual(evidence["checks"][0]["integrity"]["status"], "violated") + integrity = evidence["checks"][0]["integrity"] + self.assertNotEqual(integrity["before_state_sha256"], integrity["after_state_sha256"]) + self.assertEqual(integrity["changes"][0]["path"], mutated_path) + self.assertNotEqual(integrity["changes"][0]["before"], integrity["changes"][0]["after"]) + self.assertEqual(evidence["checks"][1]["status"], "not_run") + self.assertIsNone(evidence["checks"][1]["returncode"]) + self.assertEqual(evidence["checks"][1]["integrity"]["status"], "not_run") + output = json.loads(evidence["checks"][0]["stdout"]["preview"]) + self.assertEqual(output, expected_output) + snapshot = self.snapshot(run_id) + candidate_oid = snapshot["workspace"]["target_oid"] + self.assertEqual(self.git("show", f"{candidate_oid}:src/app.py"), "VALUE = 'wrong'\n") + if add_source: + self.assertEqual(self.git("ls-tree", candidate_oid, "--", mutated_path), "") + self.assertEqual( + (Path(snapshot["check_workspace"]["path"]) / mutated_path).read_text(), + "VALUE = 'fixed'\n", + ) + + # A new service instance must replay the invalid evidence, not rerun it + # against the altered tree or lose the gate when recovering a handoff. + self.service = Service(self.runtime) + with self.assertRaises(ConflictError): + self.service.resume(run_id) + if mode == "host": + with self.assertRaisesRegex(ContractError, "blocked by required evidence"): + self.service.handoff_complete( + run_id, claimed["claim"], self.decision(evidence, "invalid-accept", "accept"), + ) + terminal = self.service.handoff_complete( + run_id, claimed["claim"], self.decision(evidence, "reject-mutation", "reject"), + ) + self.assertEqual(terminal["state"], "failed") + receipt = self.receipt(run_id) + self.assertFalse(receipt["evaluation"]["accept_allowed"]) + self.assertEqual(receipt["checks"], evidence["checks"]) + with self.assertRaises(ConflictError): + self.service.resume(run_id) + self.assertEqual(self.service.status(run_id)["state"], "failed") + self.assertFalse(second_marker.exists()) + self.assert_original_unchanged() + + def test_branch_review_host_rejects_source_mutating_report_only_check(self): + self.exercise_mutation("branch-review", "host") + + def test_branch_review_headless_rejects_source_mutating_report_only_check(self): + self.exercise_mutation("branch-review", "headless") + + def test_delivery_host_rejects_source_mutating_report_only_check(self): + self.exercise_mutation("issue-delivery", "host") + + def test_delivery_headless_rejects_source_mutating_report_only_check(self): + self.exercise_mutation("issue-delivery", "headless") + + def test_branch_review_host_rejects_new_untracked_source_input(self): + self.exercise_mutation("branch-review", "host", add_source=True) + + def test_delivery_host_rejects_new_untracked_source_input(self): + self.exercise_mutation("issue-delivery", "host", add_source=True) + + def exercise_build_output(self, workflow): + checks = [ + self.check( + "ordinary-build-output", + "from pathlib import Path; Path('tests/build-output.bin').write_bytes(b'build')", + required=True, + output_paths=["tests/build-output.bin"], + ), + self.check( + "clean-source-and-build-output", + "from pathlib import Path; " + "assert Path('src/app.py').read_text() == \"VALUE = 'wrong'\\n\"; " + "assert Path('tests/build-output.bin').read_bytes() == b'build'", + required=True, + ), + ] + run_id = self.start(workflow, "headless", checks) + self.assertEqual(self.wait_state(run_id, {"succeeded", "failed"})["state"], "succeeded") + receipt = self.receipt(run_id) + self.assertTrue(receipt["evaluation"]["accept_allowed"]) + self.assertEqual([check["status"] for check in receipt["checks"]], ["passed", "passed"]) + self.assert_original_unchanged() + + def test_branch_review_permits_untracked_build_output_and_clean_checks(self): + self.exercise_build_output("branch-review") + + def test_delivery_permits_untracked_build_output_and_clean_checks(self): + self.exercise_build_output("issue-delivery") + + def test_terminal_completion_replay_preserves_historical_acceptance(self): + run_id = self.start("branch-review", "host", [self.check("clean", "print('ok')")]) + status = self.wait_state(run_id, {"awaiting_host", "failed"}) + self.assertEqual(status["state"], "awaiting_host") + claimed = self.service.handoff_claim(run_id, status["version"], "replay-host") + decision = self.decision(claimed["handoff"]["packet"], "accept-once", "accept") + first = self.service.handoff_complete(run_id, claimed["claim"], decision) + self.assertEqual(first["state"], "succeeded") + receipt = self.receipt(run_id) + with patch("devsquad.service.require_check_integrity", side_effect=ContractError("legacy check")) as gate: + replay = self.service.handoff_complete(run_id, claimed["claim"], decision) + gate.assert_not_called() + self.assertTrue(replay["replayed"]) + self.assertEqual(self.receipt(run_id), receipt) + conflicting = self.decision(claimed["handoff"]["packet"], "accept-once", "reject") + with self.assertRaises(ConflictError): + self.service.handoff_complete(run_id, claimed["claim"], conflicting) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_claude_identity.py b/test/core/test_claude_identity.py new file mode 100644 index 0000000..f6cfb77 --- /dev/null +++ b/test/core/test_claude_identity.py @@ -0,0 +1,413 @@ +"""Native Claude identity conformance without provider or authentication calls.""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import shlex +import sys +import tempfile +import unittest +from unittest.mock import patch + +CORE = Path(__file__).resolve().parents[2] / "plugin" / "core" +sys.path.insert(0, str(CORE / "src")) + +from devsquad.claude_delivery_worker import ( + freeze_claude_implementer, + run as run_claude_implementer, +) +from devsquad.contracts import ContractError + + +class ClaudeImplementationIdentityTest(unittest.TestCase): + MODEL = "claude-sonnet-4-6" + + def setUp(self): + temporary = tempfile.TemporaryDirectory() + self.addCleanup(temporary.cleanup) + self.root = Path(temporary.name) + self.workspace = self.root / "workspace" + self.workspace.mkdir() + self.output = self.root / "provider-result.json" + self.binary = self.root / "claude" + self.binary.write_text( + "#!/bin/sh\n" + "if [ \"$1\" = \"--version\" ]; then\n" + " printf '%s\\n' '2.1.220 (Claude Code)'\n" + " exit 0\n" + "fi\n" + f"/bin/cat {shlex.quote(str(self.output))}\n" + ) + self.binary.chmod(0o700) + + @staticmethod + def model_usage(): + # Native modelUsage keys identify the model; these values are usage, + # not a requested configuration or an effective-effort observation. + return { + "inputTokens": 12, + "outputTokens": 7, + "cacheReadInputTokens": 0, + "cacheCreationInputTokens": 0, + "webSearchRequests": 0, + "costUSD": 0.0, + "contextWindow": 200000, + "maxOutputTokens": 64000, + } + + def document(self): + return { + "type": "result", + "subtype": "success", + "is_error": False, + "result": "Applied the bounded implementation.", + "session_id": "session-identity-fixture", + "usage": {"input_tokens": 12, "output_tokens": 7}, + "modelUsage": {self.MODEL: self.model_usage()}, + } + + def snapshot(self, model=None): + profile = { + "id": "claude-implementer", + "harness": "claude", + "model_family": "claude-sonnet", + "model_id": model or self.MODEL, + "effort": {"value": "high", "transport": "native"}, + "required_tools": ["read", "write"], + "permission_policy": "workspace_write", + "account_pool_id": "claude-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["identity-fixture"], + } + selected = { + "reference": {"kind": "profile", "id": profile["id"]}, + "binding": None, + "profile_id": profile["id"], + "profile_sha256": hashlib.sha256(json.dumps( + profile, sort_keys=True, separators=(",", ":"), + ).encode()).hexdigest(), + "profile": profile, + } + baseline = "a" * 40 + task = { + "schema_version": 1, + "project": { + "repo_path": str(self.workspace), + "base_ref": baseline, + "target_ref": baseline, + }, + "workflow": "issue-delivery", + "goal": "Implement the identity fixture.", + "task_class": "identity-fixture", + "acceptance": [{ + "id": "implemented", + "description": "Implementation is complete.", + "evidence_kind": "review", + }], + "checks": [], + "scope": {"read_paths": ["src"], "write_paths": ["src"]}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + }, + "budget": { + "wall_seconds": 5, + "max_worker_invocations": 1, + "max_revisions": 0, + "max_fallbacks_per_step": 0, + }, + "origin": {"surface": "test"}, + } + with patch.dict(os.environ, {"PATH": str(self.root)}): + adapter = freeze_claude_implementer(selected) + return { + "task": task, + "delivery_workspace": { + "path": str(self.workspace), + "baseline_oid": baseline, + "write_scope": task["scope"]["write_paths"], + }, + "routing": { + "roles": { + "implementer": {"selected": selected, "fallbacks": []}, + }, + }, + "implementation_adapter": adapter, + "implementation_adapters": {profile["id"]: adapter}, + } + + def run_document(self, document, *, requested_model=None): + self.output.write_text(json.dumps(document) + "\n") + return run_claude_implementer(self.snapshot(requested_model)) + + def test_exact_native_model_is_observed_but_effective_effort_is_unknown(self): + evidence = self.run_document(self.document()) + attempt = evidence["attempt"] + observed = attempt["observed_identity"] + self.assertEqual(observed["model_id"], self.MODEL) + self.assertEqual(observed["verification"], "verified") + self.assertIsNone(observed["effort"]) + self.assertIsNone(observed["backing_revision"]) + self.assertEqual( + attempt["selected_profile"]["profile"]["effort"]["value"], "high", + ) + self.assertEqual( + attempt["native_ids"]["session_id"], "session-identity-fixture", + ) + self.assertEqual(attempt["usage"]["total_tokens"], 19) + + def test_unexpected_model_cannot_be_verified_as_requested_model(self): + document = self.document() + document["modelUsage"] = {"claude-opus-4-6": self.model_usage()} + with self.assertRaises(ContractError): + self.run_document(document) + + def test_missing_model_usage_cannot_be_verified_from_requested_argv(self): + document = self.document() + del document["modelUsage"] + with self.assertRaises(ContractError): + self.run_document(document) + + def test_empty_or_malformed_model_usage_cannot_verify_identity(self): + values = [ + None, {}, [], "invalid", {"": self.model_usage()}, + {self.MODEL: None}, {self.MODEL: []}, {self.MODEL: "invalid"}, + {self.MODEL: {}}, + ] + for value in values: + with self.subTest(model_usage=value): + document = self.document() + document["modelUsage"] = value + with self.assertRaises(ContractError): + self.run_document(document) + + def test_malformed_native_counters_cannot_verify_identity(self): + for field in ("inputTokens", "outputTokens"): + for invalid in (None, -1, True, "12", 1.5): + with self.subTest(field=field, invalid=invalid): + document = self.document() + document["modelUsage"][self.MODEL][field] = invalid + with self.assertRaises(ContractError): + self.run_document(document) + with self.subTest(missing=field): + document = self.document() + del document["modelUsage"][self.MODEL][field] + with self.assertRaises(ContractError): + self.run_document(document) + + def test_multiple_model_entries_do_not_identify_a_unique_writer(self): + document = self.document() + document["modelUsage"]["claude-haiku-4-5-20251001"] = self.model_usage() + with self.assertRaises(ContractError): + self.run_document(document) + + def stream_records(self): + document = self.document() + document["modelUsage"]["claude-haiku-4-5-20251001"] = self.model_usage() + return [{ + "type": "assistant", "session_id": document["session_id"], + "parent_tool_use_id": None, + "message": {"role": "assistant", "model": self.MODEL}, + }, document] + + def run_stream(self, records, *, requested_model=None): + self.output.write_text("\n".join(json.dumps(item) for item in records) + "\n") + return run_claude_implementer(self.snapshot(requested_model)) + + def test_correlated_stream_identifies_writer_and_retains_auxiliary_usage(self): + evidence = self.run_stream(self.stream_records(), requested_model="sonnet") + attempt = evidence["attempt"] + observed = attempt["observed_identity"] + self.assertEqual(observed["model_id"], self.MODEL) + self.assertEqual(observed["model_source"], "claude.stream.assistant.message.model") + native = observed["native_evidence"] + self.assertEqual(native["schema_version"], 2) + self.assertEqual(native["writer_messages"], { + "session_id": "session-identity-fixture", "models": [self.MODEL], "message_count": 1, + }) + self.assertEqual(len(native["model_usage"]), 2) + self.assertIsNone(observed["effort"]) + self.assertIsNone(observed["backing_revision"]) + + def test_stream_cannot_guess_writer_from_usage_or_requested_settings(self): + import copy + original = self.stream_records() + mutations = [] + for path, value in ( + ((0, "session_id"), "unrelated-session"), + ((0, "parent_tool_use_id"), "delegated-tool"), + ((0, "message", "model"), "claude-other"), + ((0, "message", "role"), "user"), + ((0, "message", "model"), None), + ((1, "modelUsage"), {"claude-haiku-4-5-20251001": self.model_usage()}), + ): + records = copy.deepcopy(original) + target = records + for key in path[:-1]: + target = target[key] + target[path[-1]] = value + mutations.append(records) + contradictory = copy.deepcopy(original[0]) + contradictory["message"]["model"] = "claude-haiku-4-5-20251001" + mutations += [original[1:], [original[0], contradictory, original[1]], + [*original, original[1]], [*original, {"type": "system"}]] + for index, records in enumerate(mutations): + with self.subTest(case=index), self.assertRaises(ContractError): + self.run_stream(records) + + def test_stream_session_model_proof_is_revalidated_on_import(self): + import copy + from devsquad.workflows import validate_implementation_evidence + snapshot = self.snapshot() + evidence = self.run_stream(self.stream_records()) + for field, value in (("session_id", "other"), ("models", [self.MODEL, "claude-haiku"]), + ("message_count", True), ("message_count", 0)): + changed = copy.deepcopy(evidence) + changed["attempt"]["observed_identity"]["native_evidence"]["writer_messages"][field] = value + with self.subTest(field=field, value=value), self.assertRaises(ContractError): + validate_implementation_evidence(changed, snapshot) + + def test_native_first_party_transport_label_is_version_bound_not_a_provider_override(self): + from devsquad.claude_identity import observed_identity + records = self.stream_records() + for entry in records[-1]["modelUsage"].values(): + entry["provider"] = "firstParty" + snapshot = self.snapshot() + observed = self.run_stream(records)["attempt"]["observed_identity"] + self.assertEqual(observed["model_provider"], "anthropic") + self.assertEqual(observed["native_evidence"]["model_usage"][self.MODEL]["provider"], "firstParty") + with self.assertRaisesRegex(ContractError, "provider"): + observed_identity(observed["native_evidence"], + {**snapshot["implementation_adapter"], "harness_version": "unverified-version"}, + snapshot["routing"]["roles"]["implementer"]["selected"]["profile"]) + records[-1]["modelUsage"]["claude-haiku-4-5-20251001"]["provider"] = "other-provider" + with self.assertRaisesRegex(ContractError, "implementation failed"): + self.run_stream(records) + + def test_family_alias_resolves_only_to_reported_concrete_model(self): + evidence = self.run_document(self.document(), requested_model="sonnet") + attempt = evidence["attempt"] + observed = attempt["observed_identity"] + self.assertEqual(observed["model_id"], self.MODEL) + self.assertEqual(observed["verification"], "verified") + self.assertIsNone(observed["effort"]) + self.assertIsNone(observed["backing_revision"]) + self.assertEqual( + attempt["selected_profile"]["profile"]["model_id"], "sonnet", + ) + + def test_alias_cannot_resolve_to_a_different_model_family(self): + document = self.document() + document["modelUsage"] = {"claude-opus-4-6": self.model_usage()} + with self.assertRaises(ContractError): + self.run_document(document, requested_model="sonnet") + + def test_supported_family_aliases_resolve_without_a_baked_in_revision(self): + for alias, concrete in ( + ("sonnet", "claude-sonnet-4-6"), + ("opus", "claude-opus-4-6"), + ("haiku", "claude-haiku-4-5-20251001"), + ): + with self.subTest(alias=alias, concrete=concrete): + document = self.document() + document["modelUsage"] = {concrete: self.model_usage()} + evidence = self.run_document(document, requested_model=alias) + observed = evidence["attempt"]["observed_identity"] + self.assertEqual(observed["model_id"], concrete) + self.assertIsNone(observed["backing_revision"]) + + def test_reported_alias_is_not_concrete_execution_identity(self): + document = self.document() + document["modelUsage"] = {"sonnet": self.model_usage()} + with self.assertRaises(ContractError): + self.run_document(document, requested_model="sonnet") + + def test_pricing_canonical_model_does_not_replace_native_model_key(self): + document = self.document() + document["modelUsage"][self.MODEL]["canonicalModel"] = "pricing-only-model" + evidence = self.run_document(document) + self.assertEqual(evidence["attempt"]["observed_identity"]["model_id"], self.MODEL) + + def test_pricing_canonical_model_cannot_hide_an_unexpected_native_model(self): + document = self.document() + usage = self.model_usage() + usage["canonicalModel"] = self.MODEL + document["modelUsage"] = {"unexpected-serving-model": usage} + with self.assertRaises(ContractError): + self.run_document(document) + + def test_top_level_model_cannot_override_native_model_usage(self): + document = self.document() + document["model"] = self.MODEL + evidence = self.run_document(document) + self.assertEqual(evidence["attempt"]["observed_identity"]["model_id"], self.MODEL) + document["model"] = "claude-opus-4-6" + with self.assertRaises(ContractError): + self.run_document(document) + + def test_only_a_success_result_envelope_can_supply_identity(self): + changes = [ + {"type": "assistant"}, + {"subtype": "error_during_execution"}, + {"is_error": True}, + {"is_error": "false"}, + {"result": ""}, + {"result": None}, + ] + for change in changes: + with self.subTest(change=change): + document = {**self.document(), **change} + with self.assertRaises(ContractError): + self.run_document(document) + for missing in ("type", "subtype", "is_error", "result"): + with self.subTest(missing=missing): + document = self.document() + del document[missing] + with self.assertRaises(ContractError): + self.run_document(document) + + def test_synthetic_auth_failure_is_classified_without_verifying_identity(self): + from devsquad.claude_identity import ClaudeResultError, decode_native_result + records = self.stream_records() + records[0]["message"]["model"] = "" + records[-1].update({"is_error": True, "result": "Not logged in. Please run /login.", "modelUsage": {}}) + payload = "\n".join(json.dumps(item) for item in records) + document, messages = decode_native_result(payload) + self.assertIsNone(messages) + self.assertTrue(document["is_error"]) + original_binary = self.binary.read_text() + for stderr in ("", "rate limit"): + with self.subTest(stderr=stderr): + self.binary.write_text(original_binary + f"printf '%s\\n' {shlex.quote(stderr)} >&2\nexit 1\n") + with self.assertRaises(ClaudeResultError) as raised: + self.run_stream(records) + self.assertEqual(raised.exception.code, "AUTH_ERROR") + self.assertEqual(raised.exception.diagnostics["identity_status"], "unverified") + self.assertEqual(raised.exception.diagnostics["model_usage"], {}) + + def test_missing_or_invalid_native_session_cannot_supply_identity(self): + for session in (None, "", " ", 42, [], "s" * 10000): + with self.subTest(session=session): + document = self.document() + document["session_id"] = session + with self.assertRaises(ContractError): + self.run_document(document) + document = self.document() + del document["session_id"] + with self.assertRaises(ContractError): + self.run_document(document) + + def test_non_object_native_json_fails_as_contract_error(self): + for document in (None, [], "success", 42): + with self.subTest(document=document): + with self.assertRaises(ContractError): + self.run_document(document) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_cli.py b/test/core/test_cli.py new file mode 100644 index 0000000..698d0dd --- /dev/null +++ b/test/core/test_cli.py @@ -0,0 +1,823 @@ +import contextlib +import io +import json +import os +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad import cli +from devsquad.contracts import ContractError +from devsquad.store import ConflictError, SchemaVersionError + + +class CliTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-cli-") + self.root = Path(self.temp.name) + self.runtime = self.root / "runtime" + self.task_file = self.root / "task.json" + self.task_file.write_text('{"schema_version": 1}\n') + + def tearDown(self): + self.temp.cleanup() + + def invoke(self, argv, service=None): + stdout, stderr = io.StringIO(), io.StringIO() + patcher = mock.patch.object(cli, "Service", return_value=service) if service is not None else contextlib.nullcontext() + with patcher, contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): + code = cli.main(argv if "--json" in argv else [*argv, "--json"]) + lines = stdout.getvalue().splitlines() + self.assertEqual(len(lines), 1, stdout.getvalue()) + return code, json.loads(lines[0]), stderr.getvalue() + + def assert_success_envelope(self, payload, data): + self.assertEqual(payload, { + "schema_version": 1, + "ok": True, + "data": data, + "error": None, + }) + + def assert_error_envelope(self, payload, code, message): + self.assertEqual(payload, { + "schema_version": 1, + "ok": False, + "data": None, + "error": { + "code": code, + "message": message, + "retryable": False, + "details": {}, + }, + }) + + def test_start_returns_exact_envelope_and_forwards_inputs(self): + service = mock.Mock() + response = {"run_id": "run-1", "state": "queued", "created": True} + service.start.return_value = response + code, payload, stderr = self.invoke([ + "start", "--task-file", str(self.task_file), + "--idempotency-key", "key-1", "--supersedes-run", "old-run", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual(code, 0) + self.assertEqual(stderr, "") + self.assert_success_envelope(payload, response) + service.start.assert_called_once_with({"schema_version": 1}, "key-1", "old-run") + service.status.assert_not_called() + + def test_trial_dispatch_is_explicit_and_preserves_the_predeclaration(self): + experiment = {"schema_version": 2, "experiment_id": "explicit-fixture"} + path = self.root / "trial-experiment.json" + path.write_text(json.dumps(experiment)) + service = mock.Mock() + response = {"run_id": "trial-run", "state": "queued", "created": True} + service.trial_start.return_value = response + code, payload, stderr = self.invoke([ + "trial", "--experiment", str(path), "--case", "held-out-case", "--arm", "candidate", + "--task-file", str(self.task_file), "--idempotency-key", "explicit-trial-key", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.trial_start.assert_called_once_with(experiment, "held-out-case", "candidate", {"schema_version": 1}, "explicit-trial-key") + service.start.assert_not_called() + service.status.assert_not_called() + + def test_review_dry_run_prepares_a_managed_task_without_starting(self): + identity = { + "harness": "codex", "harness_version": "codex fixture", + "model_id": "gpt-fixture", "model_family": "gpt", + "effort": "low", + } + task = {"managed": "review"} + summary = { + "workflow": "branch-review", "task_sha256": "a" * 64, + "project": str(self.root), "base_oid": "b" * 40, + "target_oid": "c" * 40, + "planned_roles": {"reviewer": {"harness": "codex"}}, + "selection_reason": "bounded fixture", "scope": {}, + "checks": ["candidate-diff-check"], + } + with ( + mock.patch.object(cli, "resolve_repository", return_value=self.root) as resolve, + mock.patch.object(cli, "discover_codex_identity", return_value=identity) as discover, + mock.patch.object(cli, "build_managed_task", return_value=(task, summary)) as build, + mock.patch.object(cli, "Service") as service, + ): + code, payload, stderr = self.invoke([ + "review", "--base", "main", "--target", "HEAD", + "--project-dir", str(self.root), "--model", "gpt-fixture", + "--effort", "low", "--check", "python3 -m unittest", + "--dry-run", "--runtime-dir", str(self.runtime), "--json", + ]) + self.assertEqual((code, stderr), (0, "")) + data = payload["data"] + self.assertTrue(data["dry_run"]) + self.assertEqual(data["state"], "not_started") + self.assertIsNone(data["run_id"]) + self.assertEqual( + data["idempotency_key"], + f"normal-branch-review-{'a' * 64}", + ) + self.assertIn("without --dry-run", data["next_action"]) + resolve.assert_called_once_with(str(self.root)) + discover.assert_called_once_with( + self.root, requested_model="gpt-fixture", requested_effort="low", + runtime=self.runtime, + ) + build.assert_called_once_with( + workflow="branch-review", project_dir=self.root, + base_ref="main", target_ref="HEAD", + goal="Review exact target HEAD against base main.", + codex_identity=identity, write_paths=(), + checks=(("python3", "-m", "unittest"),), check_timeout=600, + review_mode="standard", review_focus=None, + claude_model="sonnet", claude_effort="high", + role_bindings={}, pinned_roles=("reviewer",), + ) + service.assert_not_called() + + def test_fix_starts_managed_delivery_and_reports_run_and_plan(self): + identity = { + "harness": "codex", "harness_version": "codex fixture", + "model_id": "gpt-review", "model_family": "gpt", + "effort": "high", + } + task = {"managed": "fix"} + summary = { + "workflow": "issue-delivery", "task_sha256": "d" * 64, + "project": str(self.root), "base_oid": "e" * 40, + "target_oid": "e" * 40, + "planned_roles": { + "implementer": {"harness": "claude"}, + "reviewer": {"harness": "codex"}, + }, + "selection_reason": "bounded fixture", + "scope": {"write_paths": ["src"]}, + "checks": ["candidate-diff-check", "user-check-1"], + } + service = mock.Mock() + service.start.return_value = { + "run_id": "run-managed", "state": "queued", "created": True, + } + with ( + mock.patch.object(cli, "resolve_repository", return_value=self.root), + mock.patch.object(cli, "discover_codex_identity", return_value=identity) as discover, + mock.patch.object(cli, "build_managed_task", return_value=(task, summary)) as build, + ): + code, payload, stderr = self.invoke([ + "fix", "Correct the parser edge case.", + "--project-dir", str(self.root), "--write-path", "src", + "--check", "bash test/run.sh", "--review-model", "gpt-review", + "--review-effort", "high", "--review-mode", "adversarial", + "--review-focus", "state transitions", + "--implementer-model", "claude-sonnet-exact", + "--implementer-effort", "high", "--idempotency-key", "fix-1", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + data = payload["data"] + self.assertFalse(data["dry_run"]) + self.assertEqual(data["run_id"], "run-managed") + self.assertEqual(data["state"], "queued") + self.assertEqual(data["planned_roles"], summary["planned_roles"]) + self.assertEqual(data["next_action"], "squad status run-managed --json") + service.start.assert_called_once_with(task, "fix-1", None) + discover.assert_called_once_with( + self.root, requested_model="gpt-review", requested_effort="high", + runtime=self.runtime, + ) + build.assert_called_once_with( + workflow="issue-delivery", project_dir=self.root, + base_ref="HEAD", target_ref="HEAD", + goal="Correct the parser edge case.", codex_identity=identity, + write_paths=("src",), checks=(("bash", "test/run.sh"),), + check_timeout=600, review_mode="adversarial", + review_focus="state transitions", + claude_model="claude-sonnet-exact", claude_effort="high", + role_bindings={}, pinned_roles=("reviewer", "implementer"), + ) + + def test_normal_entry_rejects_waiting_for_a_dry_run(self): + code, payload, _ = self.invoke([ + "review", "--dry-run", "--wait", "--json", + ]) + self.assertEqual(code, 64) + self.assertEqual(payload["error"]["code"], "INPUT_INVALID") + self.assertIn("--wait cannot be combined", payload["error"]["message"]) + + def test_status_events_result_cancel_and_resume_operations(self): + cases = [ + (["status", "run-1"], "status", ("run-1",), {"run_id": "run-1", "state": "failed"}), + (["events", "run-1", "--after", "7", "--limit", "9"], "events", ("run-1", 7, 9), {"events": [], "next_cursor": 7}), + (["result", "run-1"], "result", ("run-1",), {"run_id": "run-1", "ready": False}), + (["cancel", "run-1"], "cancel", ("run-1",), {"run_id": "run-1", "state": "cancelling"}), + ] + for arguments, method_name, expected_args, response in cases: + with self.subTest(command=arguments[0]): + service = mock.Mock() + getattr(service, method_name).return_value = response + code, payload, stderr = self.invoke(arguments + ["--runtime-dir", str(self.runtime), "--json"], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + getattr(service, method_name).assert_called_once_with(*expected_args) + + recovery_file = self.root / "recovery.json" + recovery_file.write_text('{"attempt_id":"a-1","disposition":"confirm_dead"}\n') + service = mock.Mock() + response = {"run_id": "run-1", "disposition": "continued", "launched": True} + service.resume.return_value = response + code, payload, stderr = self.invoke([ + "resume", "run-1", "--recovery-file", str(recovery_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.resume.assert_called_once_with("run-1", {"attempt_id": "a-1", "disposition": "confirm_dead"}) + + def test_capacity_observe_dispatches_file(self): + observation = { + "schema_version": 1, + "observation_id": "obs-1", + "pool_id": "pool-a", + } + observation_file = self.root / "capacity.json" + observation_file.write_text(json.dumps(observation)) + service = mock.Mock() + response = { + "record": {"replayed": False}, + "capacity": {"status": "unknown"}, + } + service.capacity_observe.return_value = response + code, payload, stderr = self.invoke([ + "capacity", "observe", "--file", str(observation_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.capacity_observe.assert_called_once_with(observation) + + def test_finish_preserves_the_json_envelope_and_forwards_a_guided_choice(self): + for disposition in ("accept", "reject", "revise"): + with self.subTest(disposition=disposition): + service = mock.Mock() + response = {"run_id": "run-1", "state": "succeeded", "disposition": disposition} + service.finish.return_value = response + code, payload, stderr = self.invoke([ + "finish", "run-1", f"--{disposition}", "--reason", "Exact evidence assessed.", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.finish.assert_called_once_with("run-1", disposition, "Exact evidence assessed.") + service.resolve_run_id.assert_not_called() + + def test_finish_forwards_explicit_council_choice_without_inference(self): + service = mock.Mock() + response = {"run_id": "run-1", "state": "succeeded", "disposition": "accept"} + service.finish.return_value = response + code, payload, stderr = self.invoke([ + "finish", "run-1", "--accept", "--reason", "Exact evidence assessed.", + "--choose", "synthesis", "--supported-claim", "bounded retries", + "--discarded-alternative", "unbounded retries", "--validation", "checks pass", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.finish.assert_called_once_with( + "run-1", "accept", "Exact evidence assessed.", chosen="synthesis", + supported_claims=["bounded retries"], discarded_alternatives=["unbounded retries"], + validation="checks pass", + ) + + def test_normal_doctor_default_is_readable_and_keeps_unknown_proof_unknown(self): + report = {"core_version": "fixture", "ready": False, "adapters": [{ + "adapter": "claude", "status": "supported", "version": "fixture", + "authentication": {"status": "unauthenticated", "next_action": "claude auth login"}, + "operation_verified": None, + }], "supported_workflows": {"issue-delivery": {"supported": True, "ready": False}}} + output = io.StringIO() + with mock.patch.object(cli, "build_doctor_report", return_value=report), contextlib.redirect_stdout(output): + code = cli.main(["doctor", "--project-dir", str(self.root)]) + self.assertEqual(code, 1) + self.assertIn("authentication unauthenticated; operation unknown", output.getvalue()) + self.assertIn("claude auth login", output.getvalue()) + self.assertIn("issue-delivery: needs attention", output.getvalue()) + + def test_outcome_add_and_report_dispatch(self): + outcome = {"schema_version": 1, "outcome_id": "outcome-1"} + outcome_file = self.root / "outcome.json" + outcome_file.write_text(json.dumps(outcome)) + service = mock.Mock() + service.outcome_add.return_value = {"run_id": "run-1", "replayed": False} + code, payload, stderr = self.invoke([ + "outcome", "add", "run-1", "--file", str(outcome_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope( + payload, {"run_id": "run-1", "replayed": False}, + ) + service.outcome_add.assert_called_once_with("run-1", outcome) + + service = mock.Mock() + service.learning_report.return_value = {"sample_size": 3} + code, payload, stderr = self.invoke([ + "report", "--project", str(self.root), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, {"sample_size": 3}) + service.learning_report.assert_called_once_with(str(self.root)) + + def test_policy_evaluate_dispatches_frozen_experiment(self): + experiment = { + "schema_version": 1, + "experiment_id": "experiment-1", + "project_path": str(self.root), + } + experiment_file = self.root / "experiment.json" + experiment_file.write_text(json.dumps(experiment)) + service = mock.Mock() + response = { + "evaluation": { + "verdict": "no_change", + "active_policy_changed": False, + }, + "replayed": False, + } + service.policy_evaluate.return_value = response + code, payload, stderr = self.invoke([ + "policy", "evaluate", "--experiment", str(experiment_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.policy_evaluate.assert_called_once_with(experiment) + + def test_policy_evaluate_dispatches_explicit_review_revision(self): + experiment = {"schema_version": 2, "experiment_id": "experiment-1", "project_path": str(self.root)} + experiment_file = self.root / "experiment.json" + experiment_file.write_text(json.dumps(experiment)) + service = mock.Mock() + service.policy_evaluate.return_value = {"revision_id": "review-1", "replayed": False} + code, payload, stderr = self.invoke([ + "policy", "evaluate", "--experiment", str(experiment_file), "--revision-id", "review-1", + "--previous-evaluation-sha256", "a" * 64, "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, service.policy_evaluate.return_value) + service.policy_evaluate.assert_called_once_with( + experiment, revision_id="review-1", previous_evaluation_sha256="a" * 64, + ) + + def test_learn_propose_dispatches_project(self): + service = mock.Mock() + response = { + "proposal": { + "verdict": "no_change", + "active_policy_changed": False, + }, + "artifacts": {}, + } + service.learning_propose.return_value = response + code, payload, stderr = self.invoke([ + "learn", "propose", "--project", str(self.root), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.learning_propose.assert_called_once_with(str(self.root)) + + def test_profile_lifecycle_commands_dispatch_strict_files(self): + operations = [ + ( + "template-add", "profile_template_add", "profile-template.json", + {"schema_version": 1, "template_id": "template-1"}, + ), + ( + "binding-bootstrap", "profile_binding_bootstrap", "bootstrap.json", + {"template": {}, "profile": {}, "version": 1}, + ), + ( + "qualification-add", "profile_qualification_add", "qualification.json", + {"schema_version": 1, "qualification_id": "qualification-1"}, + ), + ( + "binding-change", "profile_binding_change", "change.json", + {"schema_version": 1, "decision_id": "decision-1"}, + ), + ( + "binding-fallback", "profile_binding_fallback", "fallback.json", + {"schema_version": 1, "decision_id": "decision-fallback"}, + ), + ] + for command, method_name, filename, document in operations: + with self.subTest(command=command): + path = self.root / filename + path.write_text(json.dumps(document)) + service = mock.Mock() + response = {"operation": command} + getattr(service, method_name).return_value = response + code, payload, stderr = self.invoke([ + "profile", command, "--file", str(path), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + getattr(service, method_name).assert_called_once_with(document) + + service = mock.Mock() + response = {"binding": {"alias": "review.deep"}, "decisions": []} + service.profile_binding_status.return_value = response + code, payload, stderr = self.invoke([ + "profile", "binding-show", "review.deep", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.profile_binding_status.assert_called_once_with("review.deep") + + def test_handoff_claim_renew_and_complete_dispatch_parsed_objects(self): + claim_payload = { + "schema_version": 1, + "run_id": "run-1", + "handoff_id": "handoff-1", + "owner": "terminal-a", + "fencing_token": 7, + "expires_at": "2026-09-15T06:00:00+00:00", + "run_version": 11, + } + claim_file = self.root / "claim.json" + claim_file.write_text(json.dumps(claim_payload)) + decision_payload = { + "schema_version": 1, + "submission_id": "submission-1", + "submission_hash": "a" * 64, + "disposition": "accept", + "reason": "accepted", + "evidence_refs": [], + } + decision_file = self.root / "decision.json" + decision_file.write_text(json.dumps(decision_payload)) + + service = mock.Mock() + response = { + "run_id": "run-1", "state": "awaiting_host", "version": 12, + "claim": claim_payload, + } + service.handoff_claim.return_value = response + code, payload, stderr = self.invoke([ + "handoff", "claim", "run-1", "--expected-version", "11", + "--owner", "terminal-a", "--claim-file", str(claim_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.handoff_claim.assert_called_once_with( + "run-1", 11, "terminal-a", claim_payload, + ) + + service = mock.Mock() + response = { + "run_id": "run-1", "state": "awaiting_host", + "phase": "handoff_submitted", "replayed": False, + } + service.handoff_complete.return_value = response + code, payload, stderr = self.invoke([ + "handoff", "complete", "run-1", "--claim-file", str(claim_file), + "--decision-file", str(decision_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.handoff_complete.assert_called_once_with( + "run-1", claim_payload, decision_payload, + ) + + code, payload, _ = self.invoke(["handoff"]) + self.assertEqual(code, 64) + self.assertEqual(payload["error"]["code"], "INPUT_INVALID") + + def test_parser_and_json_file_failures_are_input_errors(self): + service = mock.Mock() + service.resolve_run_id.side_effect = ContractError("No saved runs for this Git project") + code, payload, _ = self.invoke(["status", "--runtime-dir", str(self.runtime)], service) + self.assertEqual(code, 64) + self.assertEqual(payload["error"]["code"], "INPUT_INVALID") + + malformed = self.root / "malformed.json" + malformed.write_text("{") + code, payload, _ = self.invoke([ + "start", "--task-file", str(malformed), "--idempotency-key", "key", + "--runtime-dir", str(self.runtime), "--json", + ]) + self.assertEqual(code, 64) + self.assertEqual(payload["error"]["code"], "INPUT_INVALID") + self.assertIn("cannot read task file", payload["error"]["message"]) + + malformed.write_text('{"schema_version":1,"schema_version":1}') + code, payload, _ = self.invoke([ + "start", "--task-file", str(malformed), + "--idempotency-key", "duplicate-json", + "--runtime-dir", str(self.runtime), "--json", + ]) + self.assertEqual(code, 64) + self.assertIn("duplicate key", payload["error"]["message"]) + + def test_setup_filters_hosts_and_treats_a_valid_dry_run_as_completed(self): + codex = mock.Mock(id="codex") + grok = mock.Mock(id="grok") + manager = mock.Mock() + manager.setup.return_value = { + "id": "codex", + "status": "missing", + "ready": False, + "action": "would_add", + "changed": False, + } + squad = ROOT / "plugin/core/bin/squad" + with ( + mock.patch.object(cli, "load_integrations", return_value=(codex, grok)), + mock.patch.object(cli, "LocalIntegrationManager", return_value=manager) as constructor, + ): + code, payload, stderr = self.invoke([ + "setup", "--host", "codex", "--dry-run", + "--project-dir", str(self.root), + "--squad-executable", str(squad), "--json", + ]) + self.assertEqual((code, stderr), (0, "")) + self.assertTrue(payload["data"]["completed"]) + self.assertFalse(payload["data"]["ready"]) + self.assertTrue(payload["data"]["dry_run"]) + constructor.assert_called_once_with( + project=self.root, squad_executable=squad, + ) + manager.setup.assert_called_once_with(codex, dry_run=True) + + def test_setup_and_doctor_return_one_when_readiness_is_blocked(self): + template = mock.Mock(id="codex") + manager = mock.Mock() + manager.setup.return_value = { + "id": "codex", + "status": "duplicate", + "ready": False, + "action": "blocked_duplicate", + "changed": False, + } + with ( + mock.patch.object(cli, "load_integrations", return_value=(template,)), + mock.patch.object(cli, "LocalIntegrationManager", return_value=manager), + ): + code, payload, stderr = self.invoke([ + "setup", "--project-dir", str(self.root), "--json", + ]) + self.assertEqual((code, stderr), (1, "")) + self.assertFalse(payload["data"]["completed"]) + self.assertEqual(payload["data"]["hosts"][0]["action"], "blocked_duplicate") + + report = { + "core_version": "0.1.0", "ready": False, + "adapters": [], "local_apps": [], + } + with mock.patch.object(cli, "build_doctor_report", return_value=report) as doctor: + code, payload, stderr = self.invoke([ + "doctor", "--project-dir", str(self.root), "--json", + ]) + self.assertEqual((code, stderr), (1, "")) + self.assertEqual(payload["data"], report) + doctor.assert_called_once_with( + project=self.root, squad_executable=None, + ) + + def test_contract_conflict_schema_and_runtime_errors_have_exact_exits(self): + cases = [ + (ContractError("bad input"), 64, "INPUT_INVALID"), + (ConflictError("run version changed"), 75, "CONFLICT"), + (SchemaVersionError("database is newer"), 75, "SCHEMA_UNSUPPORTED"), + (OSError("disk failed"), 1, "INTERNAL_ERROR"), + (RuntimeError("unexpected"), 1, "INTERNAL_ERROR"), + ] + for exception, expected_exit, expected_code in cases: + with self.subTest(exception=type(exception).__name__): + service = mock.Mock() + service.status.side_effect = exception + code, payload, stderr = self.invoke([ + "status", "run-1", "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (expected_exit, "")) + self.assert_error_envelope(payload, expected_code, str(exception)) + + def test_wait_maps_every_terminal_and_paused_state_to_contract_exit(self): + expected = { + "succeeded": 0, + "blocked": 2, + "awaiting_host": 2, + "failed": 3, + "cancelled": 4, + } + for state, expected_exit in expected.items(): + with self.subTest(state=state): + service = mock.Mock() + service.start.return_value = {"run_id": "run-1", "state": "queued", "created": True} + status = {"run_id": "run-1", "state": state, "version": 3} + service.status.return_value = status + code, payload, stderr = self.invoke([ + "start", "--task-file", str(self.task_file), "--idempotency-key", "key", + "--wait", "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (expected_exit, "")) + self.assert_success_envelope(payload, status) + service.cancel.assert_not_called() + + def test_wait_observes_active_states_until_terminal(self): + service = mock.Mock() + service.start.return_value = {"run_id": "run-1", "state": "queued", "created": True} + service.status.side_effect = [ + {"run_id": "run-1", "state": "queued"}, + {"run_id": "run-1", "state": "running"}, + {"run_id": "run-1", "state": "succeeded"}, + ] + with mock.patch.object(cli.time, "sleep") as sleep: + code, payload, _ = self.invoke([ + "start", "--task-file", str(self.task_file), "--idempotency-key", "key", + "--wait", "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual(code, 0) + self.assertEqual(payload["data"]["state"], "succeeded") + self.assertEqual(service.status.call_count, 3) + self.assertEqual(sleep.call_count, 2) + service.cancel.assert_not_called() + + def test_managed_fix_wait_resumes_saved_candidate_review_once(self): + service = mock.Mock() + service.status.side_effect = [ + { + "run_id": "run-fix", "state": "queued", "version": 11, + "next_action": "resume_candidate_review", + }, + {"run_id": "run-fix", "state": "running", "version": 13}, + { + "run_id": "run-fix", "state": "awaiting_host", "version": 17, + "next_action": "claim_handoff", + }, + ] + service.resume.return_value = { + "run_id": "run-fix", "disposition": "continued", "launched": True, + } + with mock.patch.object(cli.time, "sleep") as sleep: + response, code = cli._wait_for_run( + service, + {"run_id": "run-fix", "state": "queued"}, + resume_candidate_review=True, + ) + self.assertEqual(code, 2) + self.assertEqual(response["data"]["state"], "awaiting_host") + service.resume.assert_called_once_with("run-fix") + self.assertEqual(sleep.call_count, 2) + + def test_wait_keyboard_interrupt_stops_observation_without_cancelling(self): + service = mock.Mock() + service.start.return_value = {"run_id": "run-42", "state": "running", "created": True} + service.status.side_effect = KeyboardInterrupt() + code, payload, stderr = self.invoke([ + "start", "--task-file", str(self.task_file), "--idempotency-key", "key", + "--wait", "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual(code, 130) + self.assert_success_envelope(payload, { + "run_id": "run-42", + "state": "running", + "observation_stopped": True, + "cancelled": False, + "next_action": "squad cancel run-42", + }) + self.assertIn("run-42", stderr) + self.assertIn("squad cancel run-42", stderr) + self.assertIn("not cancelled", stderr) + service.cancel.assert_not_called() + + def test_wait_rejects_unknown_service_state_as_internal_error(self): + service = mock.Mock() + service.start.return_value = {"run_id": "run-1", "state": "queued", "created": True} + service.status.return_value = {"run_id": "run-1", "state": "mystery"} + code, payload, _ = self.invoke([ + "start", "--task-file", str(self.task_file), "--idempotency-key", "key", + "--wait", "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual(code, 1) + self.assertEqual(payload["error"]["code"], "INTERNAL_ERROR") + + +class InstalledWheelMigrationTest(unittest.TestCase): + @staticmethod + def build_python(): + candidates = [ + os.environ.get("DEVSQUAD_BUILD_PYTHON"), + sys.executable, + str(Path.home() / ".cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3"), + shutil.which("python3.13"), + shutil.which("python3.12"), + shutil.which("python3.11"), + ] + for candidate in dict.fromkeys(value for value in candidates if value): + try: + result = subprocess.run([ + candidate, "-c", + "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68", + ], text=True, capture_output=True) + except OSError: + continue + if result.returncode == 0: + return candidate + return None + + def test_build_python_skips_missing_first_candidate_for_supported_interpreter(self): + missing = str(Path(tempfile.mkdtemp(prefix="devsquad-missing-")) / "no-such-python") + probed = [] + + def fake_run(argv, **kwargs): + probed.append(argv[0]) + if argv[0] == missing: + raise FileNotFoundError(argv[0]) + return subprocess.CompletedProcess(argv, 0) + + with ( + mock.patch.dict(os.environ, {"DEVSQUAD_BUILD_PYTHON": missing}), + mock.patch("subprocess.run", side_effect=fake_run), + ): + result = self.build_python() + self.assertEqual(probed[0], missing) + self.assertEqual(result, sys.executable) + + def test_installed_wheel_contains_and_applies_current_migrations(self): + build_python = self.build_python() + if build_python is None: + self.skipTest("offline wheel gate requires setuptools>=68 and wheel; set DEVSQUAD_BUILD_PYTHON") + with tempfile.TemporaryDirectory(prefix="devsquad-wheel-") as directory: + root = Path(directory) + source = root / "core" + shutil.copytree(ROOT / "plugin/core", source) + wheels = root / "wheels" + wheels.mkdir() + subprocess.run([ + build_python, "-m", "pip", "wheel", str(source), + "--wheel-dir", str(wheels), "--no-index", "--no-deps", "--no-build-isolation", + ], check=True, text=True, capture_output=True) + wheel = next(wheels.glob("devsquad_core-*.whl")) + environment = os.environ.copy() + environment.pop("PYTHONPATH", None) + venv = root / "venv" + subprocess.run([build_python, "-m", "venv", str(venv)], check=True, env=environment) + python = venv / ("Scripts/python.exe" if os.name == "nt" else "bin/python") + subprocess.run([ + str(python), "-m", "pip", "install", "--no-index", "--no-deps", str(wheel), + ], check=True, text=True, capture_output=True, env=environment) + probe = r''' +from importlib.resources import files +from pathlib import Path +import sqlite3 +import sys +from devsquad.store import Store, SUPPORTED_SCHEMA_VERSION + +root = Path(sys.argv[1]) +root.mkdir(parents=True) +database = root / "state.sqlite3" +connection = sqlite3.connect(database) +migrations = files("devsquad.migrations") +for version, name in ((1, "001_initial.sql"), (2, "002_supervisor.sql"), (3, "003_durable_io.sql")): + connection.executescript(migrations.joinpath(name).read_text()) + connection.execute("INSERT INTO schema_migrations(version, applied_at) VALUES(?, 'fixture')", (version,)) +connection.commit() +connection.close() +store = Store(database, root / "artifacts") +try: + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == SUPPORTED_SCHEMA_VERSION + assert migrations.joinpath("015_experiment_evaluation_revisions.sql").is_file() + assert migrations.joinpath("016_objective_outcome_jobs.sql").is_file() + attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} + assert {"role", "account_pool_id", "profile_id", "profile_index"} <= attempt_columns + columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} + assert {"package_path", "package_digest", "supersedes_run_id"} <= columns + assert store.connection.execute( + "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoffs'" + ).fetchone() +finally: + store.close() +''' + subprocess.run([ + str(python), "-P", "-c", probe, str(root / "probe-runtime"), + ], check=True, text=True, capture_output=True, cwd=root, env=environment) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_comparison.py b/test/core/test_council_comparison.py new file mode 100644 index 0000000..6596071 --- /dev/null +++ b/test/core/test_council_comparison.py @@ -0,0 +1,92 @@ +from __future__ import annotations + +import copy +import json +from pathlib import Path +import sys +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.council_comparison import predeclare, report +from devsquad.contracts import ContractError +from devsquad.store import Store, request_hash, ConflictError +import test_council_runtime as council_fixture_tests + + +class CouncilComparisonTest(unittest.TestCase): + setUp = council_fixture_tests.CouncilRuntimeTest.setUp + git = council_fixture_tests.CouncilRuntimeTest.git + wait = council_fixture_tests.CouncilRuntimeTest.wait + receipt = council_fixture_tests.CouncilRuntimeTest.receipt + + def test_predeclared_actual_matched_heldout_process_receipts_are_inconclusive_not_native_quality(self): + profiles = json.loads((self.repo / "profiles.json").read_text()) + policy = json.loads((self.repo / "policy.json").read_text()) + cases = [] + for identifier, split, question in (("retry-key", "matched", "Compare retry key retention alternatives"), + ("retry-expiry", "heldout", "Compare safe retry after key expiry")): + council = copy.deepcopy(self.task) + council["goal"] = question + council["project"]["base_ref"] = self.source_oid + council["project"]["target_ref"] = self.source_oid + council["lead"]["mode"] = "host" + council["routing"] = {"profiles": profiles, "policy": policy} + control = copy.deepcopy(council) + control["workflow"] = "branch-review" + control.pop("council") + control["routing"]["policy"]["roles"] = {"reviewer": [{"kind": "profile", "id": "critic"}]} + cases.append({"id": identifier, "split": split, "control": control, "council": council}) + declaration = predeclare(self.root / "predeclared-workflow-comparison.json", cases) + copied = copy.deepcopy(cases[0]) + copied["id"], copied["split"] = "renamed-repetition", "heldout" + with self.assertRaises(ContractError): + predeclare(self.root / "invalid-heldout.json", [cases[0], copied]) + with self.assertRaises(FileExistsError): + predeclare(Path(declaration["path"]), cases) + pairs = [] + self.task["lead"]["mode"] = "host" # wait helper does not auto-submit + for case in cases: + control = self.service.start(case["control"], case["id"] + "-control", + _internal_review_fixture={"verdict": "clean", "summary": "Controlled public review", "findings": []}) + waiting = self.wait(control["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + claimed = self.service.handoff_claim(control["run_id"], waiting["version"], "public-fixture-host") + body = {"schema_version": 1, "submission_id": "public-control", "disposition": "accept", "reason": "Controlled mechanical acceptance", + "evidence_refs": [{"artifact_id": item["artifact_id"], "sha256": item["sha256"]} for item in claimed["handoff"]["packet"]["artifacts"]]} + self.service.handoff_complete(control["run_id"], claimed["claim"], {**body, "submission_hash": request_hash(body)}) + council = self.service.start(case["council"], case["id"] + "-council", _internal_council_fixture=self.fixture) + self.assertEqual(self.wait(council["run_id"], {"awaiting_host", "failed"})["state"], "awaiting_host") + self.service.finish_council(council["run_id"], "accept", "Controlled mechanical acceptance", chosen="synthesis", + supported_claims=["No blind retry"], discarded_alternatives=["Blind retry"], validation="Test duplicate effects") + pairs.append({"id": case["id"], "control": control["run_id"], "council": council["run_id"]}) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + self.addCleanup(store.close) + compared = report(store, declaration, pairs) + self.assertTrue(Path(compared["path"]).is_file()) + self.assertEqual(report(store, declaration, pairs), compared) + self.assertEqual(compared["report"]["conclusion"], "inconclusive") + self.assertFalse(compared["report"]["automatic_enabled"]) + self.assertEqual({case["split"] for case in compared["report"]["cases"]}, {"matched", "heldout"}) + for case in compared["report"]["cases"]: + self.assertEqual(case["arms"]["control"]["worker_invocations"], 1) + self.assertEqual(case["arms"]["council"]["worker_invocations"], 3) + for arm in case["arms"].values(): + self.assertIsNone(arm["accepted_quality"]) + self.assertIsNone(arm["escaped_defects"]) + self.assertIsNone(arm["rework"]) + self.assertGreaterEqual(arm["execution_elapsed_ms"], 0) + self.assertTrue(all(arm["actual_prompt_sha256"])) + with self.assertRaises(ContractError): + report(store, declaration, [pairs[0], pairs[0]]) + original = Path(declaration["path"]).read_bytes() + tampered = json.loads(original) + tampered["automatic_enabled"] = True + Path(declaration["path"]).write_text(json.dumps(tampered)) + with self.assertRaises(ConflictError): + report(store, declaration, pairs) + Path(declaration["path"]).write_bytes(original) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_contract.py b/test/core/test_council_contract.py new file mode 100644 index 0000000..726824f --- /dev/null +++ b/test/core/test_council_contract.py @@ -0,0 +1,79 @@ +from __future__ import annotations + +import unittest +import copy +from pathlib import Path +import sys + +sys.path.insert(0, str(Path(__file__).resolve().parents[2] / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.council import digest, label_mapping, validate_spec, validate_critique, PROMPT_VERSION, output_schema, prompt_digest +from devsquad.council_runtime import validate_document + + +class CouncilContractTest(unittest.TestCase): + def spec(self): + return {"schema_version": 1, "enabled": True, "automatic": False, + "reason": "Resolve the competing hypotheses", "min_valid_proposals": 2, + "required_critics": 1, "max_invocations": 4, "seed": "a" * 64, + "evidence": [], "rubric": [{"id": "correctness", "description": "Fits evidence"}]} + + def test_strict_spec_and_automatic_off(self): + validate_spec(self.spec()) + for changes in ({"automatic": True}, {"min_valid_proposals": 1}, + {"required_critics": 0}, {"unknown": 1}): + with self.assertRaises(ContractError): + validate_spec({**self.spec(), **changes}) + + def test_seeded_labels_are_reversible(self): + mapping = label_mapping(self.spec()["seed"], ["proposer_a", "proposer_b"]) + self.assertEqual(set(mapping), {"A", "B"}) + self.assertEqual(set(mapping.values()), {"proposer_a", "proposer_b"}) + self.assertEqual(mapping, label_mapping(self.spec()["seed"], ["proposer_a", "proposer_b"])) + + def test_missing_unknown_duplicate_criterion_or_label_is_invalid(self): + good = {"summary": "Both considered", "assessments": [ + {"label": label, "criterion_id": "correctness", "status": "supported", + "reason": "Matches the packet", "evidence_ids": []} for label in ("A", "B")], + "objections": [{"id": "o1", "label": "B", "reason": "Unmeasured alternative", "evidence_ids": []}]} + validate_critique(good, self.spec()) + for assessments in (good["assessments"][:1], good["assessments"] * 2, + [{**good["assessments"][0], "label": "C"}, good["assessments"][1]]): + with self.assertRaises(ContractError): + validate_critique({**good, "assessments": assessments}, self.spec()) + + def test_native_import_requires_exact_frozen_identity_correlated_ids_and_usage(self): + profile = {"id": "author", "model_id": "actual-model", "effort": {"value": "high"}} + selected = {"profile_id": "author", "profile": profile, "profile_sha256": digest(profile)} + adapter = {"harness": "codex", "harness_version": "codex-cli tested", "model_provider": "openai"} + brief = {"source_files": []} + snapshot = {"routing": {"roles": {"proposer_a": {"selected": selected, "fallbacks": []}}}, + "council_brief": brief, "council_fixture": None, "task": {"council": self.spec()}, + "council_state": {"documents": {}}, + "council_adapters": {"proposer_a": {"author": adapter}}, + "council_boundaries": {"proposer_a": {"author": {"profile_sha256": "b" * 64}}}} + evidence = {"schema_version": 1, "role": "proposer_a", "profile": profile, "profile_sha256": digest(profile), + "prompt_version": PROMPT_VERSION, "prompt_sha256": prompt_digest("proposer_a", {"brief": brief}), + "role_packet_sha256": digest({"brief": brief}), "output_schema_sha256": digest(output_schema("proposer_a")), + "brief_sha256": digest(brief), "identity_scope": "native_verified", "boundary_sha256": "b" * 64, + "observed_identity": {**adapter, "model_id": "actual-model", "effort": "high", "permission_policy": "read_only", "verification": "verified"}, + "native_ids": {"thread_id": "observed-thread", "turn_id": "observed-turn"}, + "usage": {"input_tokens": None, "output_tokens": None, "total_tokens": None, "source": "unavailable"}, + "document": {"summary": "Proposal", "approach": "Approach", "claims": [{"text": "Claim", "evidence_ids": []}], "validation": "Check"}, "checks": []} + attempt = {"role": "proposer_a", "profile_id": "author", "profile_index": 0} + validate_document(evidence, snapshot, attempt) + invalid = [] + for key, value in (("effort", "low"), ("harness_version", "other"), ("model_provider", "other"), ("permission_policy", "write"), ("model_id", "requested-only")): + altered = copy.deepcopy(evidence) + altered["observed_identity"][key] = value + invalid.append(altered) + for field, value in (("native_ids", {"thread_id": "thread"}), ("native_ids", {"thread_id": "thread", "turn_id": ""}), + ("usage", {"input_tokens": 0, "output_tokens": 0, "total_tokens": 0, "source": "unavailable"}), + ("observed_identity", {"verification": "verified", "model_id": "actual-model"})): + altered = copy.deepcopy(evidence) + altered[field] = value + invalid.append(altered) + for altered in invalid: + with self.assertRaises(ContractError): + validate_document(altered, snapshot, attempt) diff --git a/test/core/test_council_entry.py b/test/core/test_council_entry.py new file mode 100644 index 0000000..e23eb39 --- /dev/null +++ b/test/core/test_council_entry.py @@ -0,0 +1,84 @@ +from __future__ import annotations + +from pathlib import Path +import contextlib +import io +import json +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.council_task_entry import build_council_task +from devsquad.contracts import CapabilityUnavailable, ContractError +from devsquad import cli + + +class CouncilEntryTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="council-entry-public-") + self.addCleanup(self.temporary.cleanup) + self.repo = Path(self.temporary.name) + for args in (("init", "-q"), ("config", "user.email", "fixture@example.invalid"), + ("config", "user.name", "Public fixture")): + self.git(*args) + (self.repo / "README").write_text("Frozen public input\n") + self.git("add", "README") + self.git("commit", "-qm", "Frozen public input") + self.identities = [{"harness": "codex", "model_id": "model-" + name, "model_family": name, + "effort": "low", "harness_version": "codex-cli test", "account_pool_id": "subscription-shared"} + for name in ("a", "b", "c")] + + def git(self, *args): + return subprocess.run(["git", "-C", str(self.repo), *args], check=True, capture_output=True, text=True).stdout.strip() + + def test_manual_scope_budget_and_selection_are_explicit_without_launch(self): + task, summary = build_council_task(project_dir=self.repo, goal="Compare alternatives", identities=self.identities, + read_paths=["README"], max_invocations=4) + self.assertEqual(task["scope"], {"read_paths": ["README"], "write_paths": []}) + self.assertEqual(task["budget"]["max_worker_invocations"], 4) + self.assertEqual(task["budget"]["max_revisions"], 0) + self.assertFalse(summary["automatic_enabled"]) + self.assertEqual(set(summary["profiles"]), {"proposer_a", "proposer_b", "critic"}) + self.assertTrue(all(profile["quality_status"] == "trial" for profile in summary["profiles"].values())) + self.assertEqual(self.git("status", "--porcelain"), "") + + def test_catalog_duplicates_missing_identity_and_unsafe_scope_fail_closed(self): + with self.assertRaises(CapabilityUnavailable): + build_council_task(project_dir=self.repo, goal="Question", identities=self.identities[:2] + [self.identities[0]]) + incomplete = [{**identity} for identity in self.identities] + incomplete[1].pop("harness_version") + with self.assertRaises(ContractError): + build_council_task(project_dir=self.repo, goal="Question", identities=incomplete) + for path in ("../peer", "/private/peer"): + with self.assertRaises(ContractError): + build_council_task(project_dir=self.repo, goal="Question", identities=self.identities, read_paths=[path]) + + def test_normal_dry_run_human_and_json_preserve_explicit_scope_and_cap(self): + for json_mode in (False, True): + output = io.StringIO() + with patch.object(cli, "discover_codex_identity", return_value=self.identities), contextlib.redirect_stdout(output): + args = ["council", "Compare retry alternatives", "--project-dir", str(self.repo), + "--read-path", "README", "--lead", "host", "--max-invocations", "3", "--dry-run"] + if json_mode: + args.append("--json") + self.assertEqual(cli.main(args), 0) + if json_mode: + value = json.loads(output.getvalue()) + self.assertEqual(set(value), {"schema_version", "ok", "data", "error"}) + self.assertTrue(value["data"]["dry_run"]) + self.assertIsNone(value["data"]["run_id"]) + self.assertFalse(value["data"]["native_ready"]) + self.assertEqual(value["data"]["scope"]["read_paths"], ["README"]) + else: + self.assertIn("proposer_a: codex model-a", output.getvalue()) + self.assertIn("Read scope: README; write scope: none", output.getvalue()) + self.assertIn("Worker cap: 3; rounds: 1; lead: host; automatic off", output.getvalue()) + self.assertIn("Native readiness: unavailable", output.getvalue()) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_epoch.py b/test/core/test_council_epoch.py new file mode 100644 index 0000000..af57363 --- /dev/null +++ b/test/core/test_council_epoch.py @@ -0,0 +1,58 @@ +"""Contract-epoch fences; immutable installed-old-client proof is separate.""" +from pathlib import Path +import sqlite3 +import subprocess +import sys +import tempfile +import types +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.store import Store, SchemaVersionError, SUPPORTED_SCHEMA_VERSION + + +class CouncilEpochTest(unittest.TestCase): + def test_schema16_active_and_recoverable_deferral_then_upgrade_and_old_connection_fence(self): + # A separately loaded module keeps its connection-schema function at 16 + # after current Store commits 17. This tests existing hot-client triggers, + # not the older Service's permission semantics or installed package proof. + old = types.ModuleType("devsquad._controlled_schema16_store") + old.__package__ = "devsquad" + old.__file__ = str(ROOT / "plugin/core/src/devsquad/store.py") + sys.modules[old.__name__] = old + self.addCleanup(sys.modules.pop, old.__name__, None) + source = Path(old.__file__).read_text().replace(f"SUPPORTED_SCHEMA_VERSION = {SUPPORTED_SCHEMA_VERSION}", "SUPPORTED_SCHEMA_VERSION = 16", 1) + exec(compile(source, old.__file__, "exec"), old.__dict__) + with tempfile.TemporaryDirectory(prefix="council-epoch-public-") as temporary: + root = Path(temporary) + repo = root / "repo" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + database, artifacts = root / "state.sqlite3", root / "artifacts" + previous = old.Store(database, artifacts) + self.addCleanup(previous.close) + before_tables = {r[0] for r in previous.connection.execute("SELECT name FROM sqlite_master WHERE type='table'")} + active = previous.claim_start(repo, "active16", {"task": "legacy"}, "old16") + with self.assertRaisesRegex(SchemaVersionError, "deferred.*active/recoverable"): + Store(database, artifacts) + previous.complete_preparation(active.run_id, active.fencing_token, {"legacy": True}) + with self.assertRaisesRegex(SchemaVersionError, "deferred.*active/recoverable"): + Store(database, artifacts) + self.assertEqual(previous.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 16) + previous.cancel_queued(active.run_id) + upgraded = Store(database, artifacts) + try: + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], SUPPORTED_SCHEMA_VERSION) + self.assertGreaterEqual(SUPPORTED_SCHEMA_VERSION, 17) + self.assertEqual(before_tables, {r[0] for r in upgraded.connection.execute("SELECT name FROM sqlite_master WHERE type='table'")}) + with self.assertRaisesRegex(sqlite3.DatabaseError, "newer than connection supports"): + previous.claim_start(repo, "old16-after17", {"task": "forbidden"}, "old16") + with self.assertRaisesRegex(old.SchemaVersionError, "newer than supported 16"): + old.Store(database, artifacts) + self.assertEqual(upgraded.connection.execute("SELECT COUNT(*) FROM runs").fetchone()[0], 1) + finally: + upgraded.close() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_install_epoch.py b/test/core/test_council_install_epoch.py new file mode 100644 index 0000000..ac22662 --- /dev/null +++ b/test/core/test_council_install_epoch.py @@ -0,0 +1,229 @@ +"""Actual temporary installer gate using exact accepted R5 core provenance.""" +from __future__ import annotations + +from contextlib import closing +import hashlib +import io +import json +from pathlib import Path +import sqlite3 +import subprocess +import tarfile +import time +import unittest + +import test_install_core as installer_tests +import test_council_runtime as council_fixture_tests +from devsquad.store import SUPPORTED_SCHEMA_VERSION + +ROOT = Path(__file__).resolve().parents[2] +OLD_COMMIT = "bf3de0867484552d354e6e8b6ba835f31a93aeb3" +OLD_SOURCE_SHA256 = "90f1e87fb9acf7756dad46780672de9b84fe75e3631322857a310f089280345e" +OLD_PACKAGE_SHA256 = "e03bf3a2e362b3aff6d0a5dd00bd837f15afb216d4c58ac1d9c776c4d8f09c67" + + +class CouncilInstallEpochTest(unittest.TestCase): + setUp = installer_tests.StandaloneInstallerTest.setUp + tearDown = installer_tests.StandaloneInstallerTest.tearDown + install = installer_tests.StandaloneInstallerTest.install + cli_json = installer_tests.StandaloneInstallerTest.cli_json + + def _schema(self, runtime): + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + return connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] + + def _wait(self, launcher, runtime, run_id, states): + deadline = time.monotonic() + 12 + while time.monotonic() < deadline: + status = self.cli_json(launcher, "status", run_id, "--runtime-dir", str(runtime))["data"] + if status["state"] in states: + return status + time.sleep(.05) + self.fail(f"Controlled installed run did not reach {states}: {status}") + + def _python(self, python, script, *arguments): + completed = subprocess.run([str(python), "-P", "-c", script, *map(str, arguments)], + capture_output=True, text=True, check=True, env=self.environment, cwd=self.root, timeout=20) + return json.loads(completed.stdout) + + def test_exact_old16_install_deferral_reconciliation_upgrade_and_old_handoff_mutation_fences(self): + # This archive's entire core digest is exactly the accepted immutable + # installed R5 payload, not a rewritten current Store with a lower label. + archived = subprocess.run(["git", "archive", OLD_COMMIT + ":plugin/core"], + cwd=ROOT, capture_output=True, check=True).stdout + old_source = self.root / "accepted-schema16-core" + old_source.mkdir() + with tarfile.open(fileobj=io.BytesIO(archived)) as archive: + digest = hashlib.sha256() + for member in sorted((member for member in archive.getmembers() if member.isfile()), key=lambda member: member.name): + digest.update(member.name.encode() + b"\0" + archive.extractfile(member).read()) + self.assertEqual(digest.hexdigest(), OLD_SOURCE_SHA256) + archive.extractall(old_source, filter="data") + self.environment["PIP_NO_INDEX"] = "1" + self.environment["DEVSQUAD_RUNTIME_DIR"] = str(self.install_root / "runtime") + accepted_python = Path("/Users/Dikshant/.devsquad/releases/0.1.0-py31214-90f1e87fb9ac-mcp-a26bc88afbef/venv/bin/python") + if accepted_python.is_file(): + self.environment["DEVSQUAD_PYTHON"] = str(accepted_python) + first = self.install(old_source) + old_release = Path(first["current_target"]) + old_python = old_release / "venv/bin/python" + self.assertEqual(json.loads((old_release / "release.json").read_text())["source_digest"], OLD_SOURCE_SHA256) + runtime, launcher = self.install_root / "runtime", self.bin_dir / "squad" + + # Build harmless public input/fixture documents; all execution below uses + # the actual temporary installed packages through their own interpreters. + fixture = council_fixture_tests.CouncilRuntimeTest() + fixture.setUp() + self.addCleanup(fixture.doCleanups) + profiles = json.loads((fixture.repo / "profiles.json").read_text()) + policy = json.loads((fixture.repo / "policy.json").read_text()) + control = json.loads(json.dumps(fixture.task)) + control.pop("council") + control["workflow"] = "branch-review" + control["lead"]["mode"] = "host" + control["routing"] = {"profiles": profiles, "policy": json.loads(json.dumps(policy))} + control["routing"]["policy"]["roles"] = {"reviewer": [{"kind": "profile", "id": "critic"}]} + task_path = self.root / "controlled-old16-task.json" + task_path.write_text(json.dumps(control)) + script = """ +import json,sys +from pathlib import Path +from unittest.mock import patch +from devsquad.service import Service +from devsquad.store import SUPPORTED_SCHEMA_VERSION +s=Service(Path(sys.argv[2])) +task=json.loads(Path(sys.argv[1]).read_text()) +live=s.start(task,'accepted16-active',_internal_fake_delay=30) +with patch.object(s,'_spawn_daemon',return_value=0): + queued=s.start(task,'accepted16-recoverable',_internal_fake_delay=.1) +print(json.dumps({'supported':SUPPORTED_SCHEMA_VERSION,'live':live,'queued':queued})) +""" + admitted = self._python(old_python, script, task_path, runtime) + self.assertEqual(admitted["supported"], 16) + live, queued = admitted["live"]["run_id"], admitted["queued"]["run_id"] + hot = None + council_id = None + try: + self.assertEqual(self._wait(launcher, runtime, live, {"running"})["state"], "running") + hot_script = """ +import json,sys +from pathlib import Path +from devsquad.store import Store,SUPPORTED_SCHEMA_VERSION +from devsquad.service import Service +runtime=Path(sys.argv[1]); s=Store(runtime/'state.sqlite3',runtime/'artifacts') +package=Service(runtime)._freeze_package()[1] +print(json.dumps({'supported':SUPPORTED_SCHEMA_VERSION,'schema':s.connection.execute('SELECT MAX(version) FROM schema_migrations').fetchone()[0],'package_digest':package}),flush=True) +request=json.loads(sys.stdin.readline()) +try: + s.claim_handoff(request['run_id'],request['version'],'old16-long-lived-owner') +except Exception as exc: + print(json.dumps({'rejected':True,'error':str(exc)}),flush=True) +else: + print(json.dumps({'rejected':False}),flush=True) +finally: + s.close() +""" + hot = subprocess.Popen([str(old_python), "-P", "-c", hot_script, str(runtime)], + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + text=True, env=self.environment, cwd=self.root) + ready = json.loads(hot.stdout.readline()) + self.assertEqual(ready, {"supported": 16, "schema": 16, "package_digest": OLD_PACKAGE_SHA256}) + before_selector = (self.install_root / "current").resolve() + before_launcher = launcher.read_bytes() + before_state = (self.install_root / "install-state.json").read_bytes() + + def assert_deferred(): + attempted = subprocess.run(["/bin/bash", str(installer_tests.INSTALLER), "--source-core", str(installer_tests.CORE), "--json"], + capture_output=True, text=True, env=self.environment, cwd=ROOT, timeout=90) + self.assertNotEqual(attempted.returncode, 0) + self.assertIn("active/recoverable", attempted.stderr) + self.assertEqual((self.install_root / "current").resolve(), before_selector) + self.assertEqual(launcher.read_bytes(), before_launcher) + self.assertEqual((self.install_root / "install-state.json").read_bytes(), before_state) + self.assertEqual(self._schema(runtime), 16) + + assert_deferred() + self.cli_json(launcher, "cancel", live, "--runtime-dir", str(runtime)) + self.assertEqual(self._wait(launcher, runtime, live, {"cancelled"})["state"], "cancelled") + # Active work is gone, but a recoverable old run still defers update. + assert_deferred() + self.cli_json(launcher, "resume", queued, "--runtime-dir", str(runtime)) + self.assertEqual(self._wait(launcher, runtime, queued, {"succeeded"})["state"], "succeeded") + self.assertEqual(self._schema(runtime), 16) + updated = self.install() + new_release = Path(updated["current_target"]) + self.assertNotEqual(new_release, old_release) + self.assertTrue(old_release.is_dir()) + self.assertEqual(self._schema(runtime), 16) # selector activation does not migrate + self.assertTrue(self.cli_json(launcher, "result", queued, "--runtime-dir", str(runtime))["data"]["ready"]) + self.assertEqual(self._schema(runtime), SUPPORTED_SCHEMA_VERSION) + + council = json.loads(json.dumps(fixture.task)) + council["lead"]["mode"] = "host" + council["routing"] = {"profiles": profiles, "policy": policy} + payload = self.root / "controlled-council.json" + payload.write_text(json.dumps({"task": council, "fixture": fixture.fixture})) + start_script = """ +import json,sys +from pathlib import Path +from devsquad.service import Service +value=json.loads(Path(sys.argv[1]).read_text()) +print(json.dumps(Service(Path(sys.argv[2])).start(value['task'],'new17-host-council',_internal_council_fixture=value['fixture']))) +""" + council_id = self._python(new_release / "venv/bin/python", start_script, payload, runtime)["run_id"] + waiting = self._wait(launcher, runtime, council_id, {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + output, stderr = hot.communicate(json.dumps({"run_id": council_id, "version": waiting["version"]}) + "\n", timeout=8) + self.assertEqual(hot.returncode, 0, stderr) + hot_result = json.loads(output) + self.assertTrue(hot_result["rejected"]) + self.assertIn("newer than connection supports", hot_result["error"]) + fresh_script = """ +import json,sys +from pathlib import Path +from devsquad.service import Service +try: + Service(Path(sys.argv[1])).handoff_claim(sys.argv[2],int(sys.argv[3]),'old16-fresh-owner') +except Exception as exc: + print(json.dumps({'rejected':True,'type':type(exc).__name__,'error':str(exc)})) +else: + print(json.dumps({'rejected':False})) +""" + fresh = self._python(old_python, fresh_script, runtime, council_id, waiting["version"]) + self.assertTrue(fresh["rejected"]) + self.assertEqual(fresh["type"], "SchemaVersionError") + self.assertIn(f"{SUPPORTED_SCHEMA_VERSION} is newer than supported 16", fresh["error"]) + after = self.cli_json(launcher, "status", council_id, "--runtime-dir", str(runtime))["data"] + self.assertEqual(after["version"], waiting["version"]) + self.assertIsNone(after["handoff"]["claimed_by"]) + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + self.assertEqual(connection.execute("SELECT COUNT(*) FROM claims WHERE run_id=? AND handoff_id IS NOT NULL AND active=1", (council_id,)).fetchone()[0], 0) + self.cli_json(launcher, "cancel", council_id, "--runtime-dir", str(runtime)) + self.assertEqual(self._wait(launcher, runtime, council_id, {"cancelled"})["state"], "cancelled") + from devsquad.supervisor import _live_group_exists + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + groups = [row[0] for row in connection.execute("SELECT pgid FROM attempts WHERE pgid IS NOT NULL")] + self.assertEqual(connection.execute("SELECT COUNT(*) FROM runs WHERE state NOT IN ('succeeded','failed','cancelled')").fetchone()[0], 0) + self.assertTrue(all(not _live_group_exists(group) for group in groups)) + self.epoch_probe_result = {"old_commit": OLD_COMMIT, "old_source_sha256": OLD_SOURCE_SHA256, + "old_package_sha256": OLD_PACKAGE_SHA256, "old_schema": 16, "new_schema": self._schema(runtime), + "active_deferral": True, "recoverable_deferral": True, "selector_and_launcher_preserved": True, + "old_active_cancelled": True, "old_recoverable_reconciled": True, + "long_lived_old_claim_rejected": True, "fresh_old_service_claim_rejected": True, + "council_claim_unchanged": True, "native_generation": False, "mcp_installed_in_temporary_release": False, + "owned_process_groups_gone": True, + "candidate_source_sha256": json.loads((new_release / "release.json").read_text())["source_digest"]} + finally: + if hot is not None: + if hot.poll() is None: + hot.terminate() + hot.communicate(timeout=5) + if (runtime / "state.sqlite3").is_file(): + current = (self.install_root / "current").resolve() + cleanup = "from pathlib import Path;import sys;from devsquad.service import Service;s=Service(Path(sys.argv[1]));[s.cancel(r) for r in sys.argv[2:]]" + subprocess.run([str(current / "venv/bin/python"), "-P", "-c", cleanup, str(runtime), live, queued, *([council_id] if council_id else [])], + capture_output=True, env=self.environment, cwd=self.root, timeout=15) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_isolation.py b/test/core/test_council_isolation.py new file mode 100644 index 0000000..a53762d --- /dev/null +++ b/test/core/test_council_isolation.py @@ -0,0 +1,105 @@ +from __future__ import annotations + +import hashlib +import os +from pathlib import Path +import platform +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.council_isolation import command, freeze_boundary, verify_read_boundary, verify_native_bootstrap, verify_native_network +from devsquad.contracts import CapabilityUnavailable +from devsquad.mcp_server import MCPBridge + + +class CouncilIsolationTest(unittest.TestCase): + def test_bootstrap_uses_shared_cleanup_and_refuses_missing_identity_before_rpc(self): + # No provider execution: test only the exact caller authority contract. + from unittest.mock import Mock + process = Mock() + with tempfile.TemporaryDirectory(prefix="council-bootstrap-contract-") as temporary: + with patch("devsquad.diagnostics._probe_output", return_value=(0, "codex-cli test")), \ + patch("devsquad.council_isolation.command", return_value=["frozen-command"]), \ + patch("devsquad.council_isolation.subprocess.Popen", return_value=process), \ + patch("devsquad.probe_process.capture_probe_identity", return_value=None), \ + patch("devsquad.probe_process.close_probe") as close, \ + patch("devsquad.codex_protocol.JsonLinePeer") as peer: + with self.assertRaisesRegex(CapabilityUnavailable, "ownership identity"): + verify_native_bootstrap({}, executable=Path("/bin/cat"), expected_version="codex-cli test", evidence=Path(temporary)) + peer.assert_not_called() + close.assert_called_once_with(process, start_identity=None) + + def test_native_bootstrap_or_cached_catalog_is_not_network_attestation(self): + with self.assertRaisesRegex(CapabilityUnavailable, "genuine non-generating HTTPS backend response"): + verify_native_network({"probe_status": "passed", "model_list_count": 8}) + + def test_unsupported_platform_has_no_unsandboxed_fallback(self): + with patch("devsquad.council_isolation.platform.system", return_value="Linux"): + with self.assertRaises(CapabilityUnavailable): + freeze_boundary(executable=Path("/bin/cat"), evidence=ROOT) + + @unittest.skipUnless(platform.system() == "Darwin" and Path("/usr/bin/sandbox-exec").is_file(), "macOS Seatbelt required") + def test_actual_same_user_process_denies_peer_ledger_logs_artifacts_and_symlink_children(self): + with tempfile.TemporaryDirectory(prefix="council-isolation-public-") as temporary: + root = Path(temporary).resolve() + own = root / "proposer_a" + own.mkdir() + brief = own / "brief.txt" + brief.write_text("Only own frozen evidence\n") + forbidden = [] + for directory, name in (("proposer_b", "proposal.txt"), ("critic", "critique.txt"), + ("runtime", "state.sqlite3"), ("private-logs", "supervisor.log"), ("artifacts", "raw-provenance.json")): + parent = root / directory + parent.mkdir() + path = parent / name + path.write_text("Private peer/coordinator data\n") + forbidden.append(path) + boundary = freeze_boundary(executable=Path("/bin/cat"), evidence=own) + verify_read_boundary(boundary, own_file=brief, forbidden=tuple(forbidden)) + self.assertEqual(subprocess.run(command(boundary, ["/bin/cat", str(brief)]), capture_output=True).returncode, 0) + for path in forbidden: + result = subprocess.run(command(boundary, ["/bin/cat", str(path)]), capture_output=True) + self.assertNotEqual(result.returncode, 0) + self.assertNotIn(b"Private peer", result.stdout) + link = own / "peer-link" + link.symlink_to(forbidden[0]) + self.assertNotEqual(subprocess.run(command(boundary, ["/bin/cat", str(link)]), capture_output=True).returncode, 0) + # The native jail is stricter (no shell exec). Permit shell only in + # this diagnostic variant to prove the file rules inherit to children. + policy = boundary["profile"] + '(allow process-exec (literal "/bin/sh"))\n(allow file-read* (literal "/bin/sh"))\n' + child = subprocess.run(["/usr/bin/sandbox-exec", "-p", policy, "/bin/sh", "-c", 'exec /bin/cat "$1"', "probe", str(forbidden[0])], capture_output=True) + self.assertNotEqual(child.returncode, 0) + corrupt = {**boundary, "profile": boundary["profile"] + "(allow default)"} + with self.assertRaises(CapabilityUnavailable): + command(corrupt, ["/bin/cat", str(brief)]) + # The native dependency additions must preserve these exact denials. + native = freeze_boundary(executable=Path("/bin/cat"), evidence=own, native_codex=True) + verify_read_boundary(native, own_file=brief, forbidden=tuple(forbidden)) + self.assertNotEqual(subprocess.run(command(native, ["/bin/cat", str(link)]), capture_output=True).returncode, 0) + environment = {"PATH": "/usr/bin:/bin"} # no Council/worker marker + self.assertNotEqual(subprocess.run(command(native, ["/bin/cat", str(forbidden[2])]), + capture_output=True, env=environment).returncode, 0) + # Alternate MCP servers cannot escape by spawning a new interpreter: + # it is not a frozen execution dependency, even under the same uid. + alternate = subprocess.run(["/usr/bin/sandbox-exec", "-p", native["profile"], sys.executable, + "-c", "import pathlib; pathlib.Path(__import__('sys').argv[1]).read_bytes()", str(forbidden[2])], + capture_output=True, env=environment) + self.assertNotEqual(alternate.returncode, 0) + + def test_worker_mcp_saved_reads_cannot_bypass_role_directory(self): + with tempfile.TemporaryDirectory(prefix="council-mcp-denial-") as directory: + bridge = MCPBridge(Path(directory)) + with patch.dict(os.environ, {"DEVSQUAD_WORKER": "1", "DEVSQUAD_DELEGATION_DEPTH": "1", "DEVSQUAD_COUNCIL_ROLE": "proposer_a"}): + for operation in (lambda: bridge.status("peer-run"), lambda: bridge.events("peer-run"), lambda: bridge.result("peer-run")): + result = operation() + self.assertFalse(result["ok"]) + self.assertEqual(result["error"]["code"], "POLICY_DENIED") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_reconciliation.py b/test/core/test_council_reconciliation.py new file mode 100644 index 0000000..1750423 --- /dev/null +++ b/test/core/test_council_reconciliation.py @@ -0,0 +1,229 @@ +"""Public fixture/process Council gates against the shared R6 authority.""" +from __future__ import annotations + +import contextlib +from datetime import datetime, timedelta, timezone +import io +import json +import os +from pathlib import Path +import shlex +import sys +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path[:0] = [str(ROOT / "plugin/core/src"), str(ROOT / "test/core")] +import test_council_runtime as fixtures +from devsquad import cli, detached +from devsquad.contracts import ContractError +from devsquad.service import Service +from devsquad.store import ConflictError, Store, request_hash +from devsquad.supervisor import _live_group_exists + + +class CouncilReconciliationTest(unittest.TestCase): + setUp = fixtures.CouncilRuntimeTest.setUp + git = fixtures.CouncilRuntimeTest.git + wait = fixtures.CouncilRuntimeTest.wait + receipt = fixtures.CouncilRuntimeTest.receipt + + def host_run(self, key): + self.task["lead"]["mode"] = "host" + started = self.service.start(self.task, key, _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + return started["run_id"] + + def invoke(self, argv): + output = io.StringIO() + with contextlib.redirect_stdout(output): + code = cli.main([*argv, "--runtime-dir", str(self.runtime)]) + return code, output.getvalue() + + def finish(self, run): + return self.service.finish(run, "accept", "--frozen 'choice'", chosen="synthesis", + supported_claims=["--no blind retry"], discarded_alternatives=["'blind' retry"], validation="--duplicate test") + + def test_human_omitted_id_explicit_choice_and_exact_retry_round_trip(self): + run = self.host_run("normal-council-finish") + code, output = self.invoke(["status", "--project-dir", str(self.repo)]) + self.assertEqual(code, 0, output) + self.assertIn("Proposal A:", output) + self.assertIn("Proposal B:", output) + self.assertIn("Dissent retention", output) + self.assertIn("explicit disposition, --choose", output) + self.assertNotIn("--accept --reason=", output) # no automatic choice + code, output = self.invoke(["finish", "--project-dir", str(self.repo), "--accept", "--reason", "no choice"]) + self.assertEqual(code, 64, output) + before = self.service.status(run)["version"] + with patch.object(self.service, "handoff_complete", side_effect=RuntimeError("claim committed")): + with self.assertRaises(RuntimeError): + self.finish(run) + version = self.service.status(run)["version"] + self.assertEqual(version, before + 1) + code, output = self.invoke(["status", run]) + self.assertEqual(code, 0, output) + command = shlex.split(next(line[6:] for line in output.splitlines() if line.startswith("Next: "))) + self.assertIn("--choose=synthesis", command) + self.assertIn("--supported-claim=--no blind retry", command) + code, output = self.invoke(["status", run, "--json"]) + self.assertNotIn("handoff_view", json.loads(output)["data"]) + code, output = self.invoke(command[1:]) + self.assertEqual(code, 0, output) + self.assertEqual(self.receipt(run)["lead"]["choice"]["validation"], "--duplicate test") + code, output = self.invoke(["result", "--project-dir", str(self.repo)]) + self.assertEqual(code, 0, output) + self.assertIn("receipt.json:", output) + + def test_host_expiry_rejection_is_audited_then_recovers_exact_intent(self): + run = self.host_run("host-expiry-at-submission") + captured = {} + original_complete = self.service.handoff_complete + def expire(run_id, claim, decision): + captured.update(claim=claim, decision=decision) + captured["later"] = datetime.fromisoformat(claim["expires_at"]) + timedelta(seconds=1) + with patch("devsquad.store._authoritative_now", return_value=captured["later"]): + return original_complete(run_id, claim, decision) + with patch.object(self.service, "handoff_complete", side_effect=expire): + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.finish(run) + store = self.service._store() + try: + rejected = dict(store.connection.execute("SELECT * FROM handoff_submissions WHERE handoff_id=?", (captured["claim"]["handoff_id"],)).fetchone()) + marker = store.terminal_finish_decision(run, captured["claim"]["handoff_id"]) + self.assertEqual(marker, captured["decision"]) + finally: + store.close() + with patch("devsquad.store._authoritative_now", return_value=captured["later"]): + with self.assertRaisesRegex(ConflictError, "different terminal finish intent"): + self.service.finish(run, "reject", "changed intent", chosen="A", validation="check") + self.assertEqual(Service(self.runtime).resume(run)["state"], "succeeded") + store = self.service._store() + try: + events = store.events_for_run(run) + recovered = [e for e in events if e["type"] == "handoff.completion_recovered"] + self.assertEqual(len(recovered), 1) + self.assertEqual(recovered[0]["payload"]["rejected_submission"], rejected) + self.assertEqual(recovered[0]["payload"]["rejected_submission_sha256"], request_hash(rejected)) + self.assertEqual(recovered[0]["payload"]["fencing_token"], captured["claim"]["fencing_token"] + 1) + self.assertEqual(len(store.outcomes_for_run(run)), 1) + finally: + store.close() + + def test_guided_finish_never_adopts_app_claim_even_same_owner_expired(self): + run = self.host_run("app-claim-no-terminal-authority") + claim = self.service.handoff_claim(run, self.service.status(run)["version"], "terminal-operator")["claim"] + for now in (datetime.now(timezone.utc), datetime.fromisoformat(claim["expires_at"]) + timedelta(seconds=1)): + version = self.service.status(run)["version"] + with patch("devsquad.store._authoritative_now", return_value=now): + with self.assertRaisesRegex(ConflictError, "already has a host claim"): + self.finish(run) + with self.assertRaisesRegex(ConflictError, "no exact current"): + Service(self.runtime).resume(run) + self.assertEqual(self.service.status(run)["version"], version) + self.service.cancel(run) + + def controlled_stage(self, run_id): + # Scheduling seam only: public start + real durable subprocess workers; + # coordinator auto-resume is suppressed at the exact durable barrier. + store = self.service._store() + try: + run = store.run(run_id) + finally: + store.close() + with patch.dict(os.environ, {"PYTHONPATH": run["package_path"]}), patch.object(Service, "resume", return_value={}): + self.assertEqual(detached.main(["--database", str(self.service.database), "--artifacts", str(self.service.artifacts), + "--run-id", run_id, "--expected-version", str(run["version"]), "--package-digest", run["package_digest"]]), 0) + + def headless_at_imported_lead(self, key): + self.task["lead"]["mode"] = "headless" + with patch.object(Service, "_spawn_daemon", return_value=0): + started = self.service.start(self.task, key, _internal_council_fixture=self.fixture) + for _ in range(3): + self.controlled_stage(started["run_id"]) + self.assertEqual(self.service.resume(started["run_id"])["state"], "queued") + self.controlled_stage(started["run_id"]) + return started["run_id"] + + def test_headless_claim_and_submitted_crashes_recover_without_extra_workers(self): + for boundary in ("record_handoff_submission", "complete_handoff_terminal"): + for expired in (False, True): + with self.subTest(boundary=boundary, expired=expired): + run = self.headless_at_imported_lead(f"headless-{boundary}-{expired}") + with patch.object(Store, boundary, side_effect=RuntimeError("after durable boundary")): + with self.assertRaises(RuntimeError): + self.service.resume(run) + store = self.service._store() + try: + handoff = store.handoff_snapshot(run) + old = dict(store.connection.execute("SELECT * FROM claims WHERE run_id=?", (run,)).fetchone()) + submitted = store.recorded_handoff_submission(run, handoff.handoff_id) + finally: + store.close() + future = datetime.fromisoformat(old["lease_expires_at"]) + timedelta(seconds=1) + class FutureDateTime(datetime): + @classmethod + def now(cls, tz=None): + return future + with contextlib.ExitStack() as stack: + if expired: + stack.enter_context(patch("devsquad.council_runtime.datetime", FutureDateTime)) + stack.enter_context(patch("devsquad.store._authoritative_now", return_value=future)) + result = Service(self.runtime).resume(run) + self.assertEqual(result["state"], "succeeded") + receipt = self.receipt(run) + self.assertEqual(receipt["worker_invocations"], 4) + self.assertEqual(receipt["identity_scope"], "all_fixture") + store = self.service._store() + try: + attempts = store.attempts_for_run(run) + self.assertTrue(all(not _live_group_exists(a["pgid"]) for a in attempts)) + latest = dict(store.connection.execute("SELECT * FROM claims WHERE run_id=?", (run,)).fetchone()) + # Recorded submissions need no new lease or fence. + self.assertEqual(latest["fencing_token"], old["fencing_token"] + int(expired and submitted is None)) + self.assertFalse(latest["active"]) + self.assertEqual(len(store.outcomes_for_run(run)), 1) + finally: + store.close() + + def test_headless_expired_submission_is_retained_and_new_fence_completes_saved_lead(self): + run = self.headless_at_imported_lead("headless-expiry-at-submission") + with self.assertRaisesRegex(ConflictError, "does not accept a host claim"): + self.service.handoff_claim(run, self.service.status(run)["version"], "app-owner") + captured = {} + original_record = Store.record_handoff_submission + def expire(store, run_id, claim, decision, **kwargs): + captured.update(claim=claim, decision=decision) + captured["later"] = datetime.fromisoformat(claim.expires_at) + timedelta(seconds=1) + return original_record(store, run_id, claim, decision, now=captured["later"]) + with patch.object(Store, "record_handoff_submission", new=expire): + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.service.resume(run) + future = captured["later"] + class FutureDateTime(datetime): + @classmethod + def now(cls, tz=None): + return future + with patch("devsquad.council_runtime.datetime", FutureDateTime), patch("devsquad.store._authoritative_now", return_value=future): + self.assertEqual(Service(self.runtime).resume(run)["state"], "succeeded") + store = self.service._store() + try: + submissions = [dict(row) for row in store.connection.execute("SELECT * FROM handoff_submissions WHERE handoff_id=?", (captured["claim"].handoff_id,))] + self.assertEqual(len(submissions), 2) + rejected = next(row for row in submissions if row["outcome"] == "rejected") + recorded = next(row for row in submissions if row["outcome"] == "recorded") + self.assertEqual(rejected["rejection_code"], "expired_claim") + self.assertEqual(recorded["fencing_token"], rejected["fencing_token"] + 1) + self.assertNotEqual(recorded["submission_id"], rejected["submission_id"]) + self.assertEqual(json.loads(recorded["decision_json"])["council_choice"], captured["decision"]["council_choice"]) + self.assertEqual(len(store.attempts_for_run(run)), 4) + self.assertEqual(len(store.outcomes_for_run(run)), 1) + self.assertTrue(all(not _live_group_exists(a["pgid"]) for a in store.attempts_for_run(run))) + with self.assertRaisesRegex(ConflictError, "expired_claim"): + store.record_handoff_submission(run, captured["claim"], captured["decision"]) + finally: + store.close() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_runtime.py b/test/core/test_council_runtime.py new file mode 100644 index 0000000..14254d6 --- /dev/null +++ b/test/core/test_council_runtime.py @@ -0,0 +1,314 @@ +from __future__ import annotations + +import copy +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import time +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.service import Service +from devsquad.store import Store, request_hash, ConflictError +from devsquad.contracts import ContractError +from devsquad.council_worker import role_packet +from devsquad.contracts import ExecutionIdentity, LaunchSpec +from devsquad.supervisor import Supervisor, _live_group_exists + + +class CouncilRuntimeTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-council-") + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + self.repo.mkdir() + self.git("init", "-q") + self.git("config", "user.email", "fixture@example.invalid") + self.git("config", "user.name", "Public controlled fixture") + (self.repo / "README").write_text("A retry duplicates a non-idempotent request.\n") + profiles = [] + for role, model in (("proposer_a", "model-a"), ("proposer_b", "model-b"), ("critic", "model-c")): + profiles.append({"id": role, "harness": "fixture", "model_family": model, + "model_id": model, "effort": {"value": "low", "transport": "native"}, + "required_tools": [], "permission_policy": "read_only", "account_pool_id": "shared", + "billing_mode": "subscription", "quality_status": "proven", "evidence_refs": ["public-fixture"]}) + (self.repo / "profiles.json").write_text(json.dumps({"schema_version": 1, "profiles": profiles, "bindings": {}})) + policy = {"schema_version": 1, "id": "controlled-council", "version": 1, + "roles": {role: [{"kind": "profile", "id": role}] for role in ("proposer_a", "proposer_b", "critic")}, + "task_classes": {"fixture-council": "proven"}, "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, "account_pools": {"shared": { + "allowed_billing_modes": ["subscription"], "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded"}}, "experiment_budget": {}} + policy["roles"]["lead"] = [{"kind": "profile", "id": "critic"}] + (self.repo / "policy.json").write_text(json.dumps(policy)) + self.git("add", ".") + self.git("commit", "-qm", "public frozen input") + self.source_oid = self.git("rev-parse", "HEAD") + self.task = {"schema_version": 1, "project": {"repo_path": str(self.repo), "base_ref": "HEAD", "target_ref": "HEAD"}, + "workflow": "council-decision", "goal": "Choose a safe retry policy", "task_class": "fixture-council", + "acceptance": [{"id": "safe-retry", "description": "Avoid duplicate side effects", "evidence_kind": "review"}], + "checks": [{"id": "public-check", "argv": ["git", "diff", "--check"], "cwd": ".", "timeout_seconds": 5, "required_to_pass": True}], + "scope": {"read_paths": ["README"], "write_paths": []}, "lead": {"mode": "headless"}, + "routing": {"profiles_file": "profiles.json", "policy_file": "policy.json"}, + "budget": {"wall_seconds": 60, "max_worker_invocations": 4, "max_revisions": 0, "max_fallbacks_per_step": 0}, + "origin": {"surface": "cli"}, "council": {"schema_version": 1, "enabled": True, "automatic": False, + "reason": "Compare independent retry alternatives", "min_valid_proposals": 2, "required_critics": 1, + "max_invocations": 4, "seed": "a" * 64, "evidence": [], + "rubric": [{"id": "safety", "description": "Does not duplicate side effects"}]}} + proposal = {"summary": "Use an idempotency key", "approach": "Bounded retries with a stable key", + "claims": [{"text": "A stable key avoids duplicate effects", "evidence_ids": []}], + "validation": "Exercise a repeated request"} + self.fixture = {"proposer_a": {"document": proposal}, "proposer_b": {"document": dict(proposal, summary="Do not retry blindly")}, + "critic": {"document": {"summary": "Both avoid a blind retry", "assessments": [ + {"label": label, "criterion_id": "safety", "status": "supported", "reason": "Requires duplicate protection", "evidence_ids": []} for label in ("A", "B")], + "objections": [{"id": "retention", "label": "A", "reason": "Key retention must cover retries", "evidence_ids": []}]}}, + "lead": {"document": {"disposition": "accept", "chosen": "synthesis", "reason": "Retain both safeguards", + "supported_claims": ["Never retry a non-idempotent request blindly"], "discarded_alternatives": ["Blind retry"], + "unresolved_objections": ["retention"], "validation": "Test duplicates and retention expiry"}}} + self.service = Service(self.runtime) + + def git(self, *args): + return subprocess.run(["git", "-C", str(self.repo), *args], check=True, capture_output=True, text=True).stdout.strip() + + def wait(self, run_id, states, timeout=12): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status = self.service.status(run_id) + if status["state"] in states: + return status + if status["state"] == "awaiting_host" and self.task["lead"]["mode"] == "headless": + try: + self.service.resume(run_id) + except ConflictError: + pass # Another detached coordinator may win this exact fence. + time.sleep(.05) + self.fail(f"Council did not reach {states}: {self.service.status(run_id)}") + + def receipt(self, run_id): + result = self.service.result(run_id) + artifact = next(a for a in result["artifacts"] if a["name"] == "receipt.json") + return json.loads(Path(artifact["path"]).read_bytes()) + + def test_public_process_flow_seals_proposals_and_preserves_dissent_and_unknown_usage(self): + started = self.service.start(self.task, "public-council", _internal_council_fixture=self.fixture) + self.assertEqual(self.wait(started["run_id"], {"succeeded", "failed"})["state"], "succeeded") + receipt = self.receipt(started["run_id"]) + self.assertEqual([a["role"] for a in receipt["attempts"]], ["proposer_a", "proposer_b", "critic", "lead"]) + self.assertEqual(receipt["worker_invocations"], 4) + self.assertEqual(receipt["identity_scope"], "all_fixture") + self.assertTrue(all(a["usage"] is None and a["observed_identity"] is None for a in receipt["attempts"])) + self.assertEqual(receipt["dissent"][0]["id"], "retention") + self.assertFalse(receipt["automatic_enabled"]) + self.assertEqual(self.git("rev-parse", "HEAD"), self.source_oid) + self.assertEqual(self.git("status", "--porcelain"), "") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + self.addCleanup(store.close) + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + outcomes = store.outcomes_for_run(started["run_id"]) + self.assertEqual(len(outcomes), 1) + outcome = outcomes[0]["outcome"] + self.assertEqual(outcome["kind"], "final") + self.assertEqual({item["role"] for item in outcome["contributions"]}, {"proposer_a", "proposer_b", "critic", "lead"}) + self.assertTrue(all(item["status"] == "passed" for item in outcome["criteria"])) + self.service.result(started["run_id"]) + self.service.learning_report(str(self.repo)) + self.assertEqual(len(store.outcomes_for_run(started["run_id"])), 1) + for role in ("proposer_a", "proposer_b"): + packet = role_packet(snapshot, role) + self.assertEqual(set(packet), {"brief"}) + self.assertNotIn("profile", json.dumps(packet)) + critic = role_packet(snapshot, "critic") + self.assertEqual(set(critic["proposals"]), {"A", "B"}) + self.assertNotIn("model-a", json.dumps(critic)) + self.assertNotIn("model-b", json.dumps(critic)) + + def test_invalid_critic_cannot_become_consensus(self): + fixture = copy.deepcopy(self.fixture) + fixture["critic"]["document"]["assessments"].pop() + started = self.service.start(self.task, "invalid-critic", _internal_council_fixture=fixture) + self.assertEqual(self.wait(started["run_id"], {"failed"})["state"], "failed") + receipt = self.receipt(started["run_id"]) + self.assertEqual(receipt["worker_invocations"], 3) + self.assertIsNone(receipt["lead"]["disposition"]) + + def test_cancel_owned_gated_launcher_never_publishes_terminal_before_cleanup(self): + with patch.object(self.service, "_spawn_daemon", return_value=0): + started = self.service.start(self.task, "gated-owned-cancel", _internal_council_fixture=self.fixture) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + self.addCleanup(store.close) + run = store.run(started["run_id"]) + snapshot = json.loads(run["mutable_snapshot"]) + selected = snapshot["routing"]["roles"]["proposer_a"]["selected"] + identity = ExecutionIdentity("fixture", "controlled", None, selected["profile"]["model_family"], + selected["profile"]["model_id"], "low", permissions="read_only", account_pool="shared") + spec = LaunchSpec(1, "fixture", "cli_exec", (sys.executable, "-P", "-m", "devsquad.fake_step", "--delay", "30"), + run["worktree_path"], None, 5, identity, + {"DEVSQUAD_WORKER": "1", "DEVSQUAD_COUNCIL_ROLE": "proposer_a"}) + original_release = Supervisor._release_runner_gate + owned = [] + + def cancel_before_gate_release(descriptor): + attempt = store.attempt(started["run_id"]) + owned.append(attempt["pgid"]) + self.assertTrue(_live_group_exists(attempt["pgid"])) + cancelled = self.service.cancel(started["run_id"]) + self.assertEqual(cancelled["state"], "cancelling") + self.assertIsNone(store.artifact_named(started["run_id"], "receipt.json")) + original_release(descriptor) + + supervisor = Supervisor(store) + with patch.dict(os.environ, {"PYTHONPATH": str(Path(run["package_path"]))}), patch.object(Supervisor, "_release_runner_gate", side_effect=cancel_before_gate_release): + handle = supervisor.launch_durable(started["run_id"], run["version"], spec, "controlled-gated-owner", run["package_digest"], + role="proposer_a", profile_id=selected["profile_id"], profile_index=0) + supervisor.wait_durable(handle, 5) + self.assertEqual(self.service.status(started["run_id"])["state"], "cancelled") + self.assertTrue(all(not _live_group_exists(group) for group in owned)) + self.assertIsNone(self.receipt(started["run_id"])["lead"]["disposition"]) + + def test_failed_mandatory_check_cannot_be_overridden_by_lead(self): + self.task["checks"][0]["argv"] = ["false"] + started = self.service.start(self.task, "failed-check", _internal_council_fixture=self.fixture) + self.assertEqual(self.wait(started["run_id"], {"failed"})["state"], "failed") + receipt = self.receipt(started["run_id"]) + self.assertIsNone(receipt["lead"]["disposition"]) + + def test_host_exact_claim_choice_and_terminal_replay(self): + self.task["lead"]["mode"] = "host" + self.task["budget"]["max_worker_invocations"] = 3 + started = self.service.start(self.task, "host", _internal_council_fixture=self.fixture) + waiting = self.wait(started["run_id"], {"awaiting_host"}) + acquired = self.service.handoff_claim(started["run_id"], waiting["version"], "fixture-host") + packet = acquired["handoff"]["packet"] + choice = self.fixture["lead"]["document"] + body = {"schema_version": 1, "submission_id": "host-choice", "disposition": "accept", "reason": choice["reason"], + "evidence_refs": [{"artifact_id": a["artifact_id"], "sha256": a["sha256"]} for a in packet["artifacts"]], "council_choice": choice} + decision = {**body, "submission_hash": request_hash(body)} + result = self.service.handoff_complete(started["run_id"], acquired["claim"], decision) + self.assertEqual(result["state"], "succeeded") + self.assertTrue(self.service.handoff_complete(started["run_id"], acquired["claim"], decision)["replayed"]) + + def test_guided_host_finish_has_explicit_choice_and_no_json(self): + self.task["lead"]["mode"] = "host" + started = self.service.start(self.task, "guided-host", _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + view = self.service.council_handoff_view(started["run_id"]) + self.assertIn("Council handoff", view["report"]) + result = self.service.finish_council(started["run_id"], "accept", "retain safeguards", chosen="synthesis", + supported_claims=["No blind retries"], discarded_alternatives=["Blind retry"], validation="Test retention expiry") + self.assertEqual(result["state"], "succeeded") + with self.assertRaises(ConflictError): + self.service.finish_council(started["run_id"], "accept", "replay", chosen="A", + supported_claims=["claim"], discarded_alternatives=[], validation="check") + + def test_missing_proposer_empty_output_and_missing_critic_never_form_quorum(self): + for index, (role, empty) in enumerate((("proposer_a", False), ("proposer_b", True), ("critic", False))): + fixture = copy.deepcopy(self.fixture) + if empty: + fixture[role] = {"document": {}} + else: + fixture.pop(role) + started = self.service.start(self.task, f"missing-{index}", _internal_council_fixture=fixture) + self.assertEqual(self.wait(started["run_id"], {"failed"})["state"], "failed") + receipt = self.receipt(started["run_id"]) + self.assertIsNone(receipt["lead"]["disposition"]) + self.assertLess(receipt["worker_invocations"], 4) + + def test_cancel_active_role_and_saved_host_wait_never_accept(self): + fixture = copy.deepcopy(self.fixture) + fixture["proposer_a"]["delay_seconds"] = 4 + started = self.service.start(self.task, "active-cancel", _internal_council_fixture=fixture) + self.wait(started["run_id"], {"running"}) + self.service.cancel(started["run_id"]) + self.assertEqual(self.wait(started["run_id"], {"cancelled"})["state"], "cancelled") + self.assertIsNone(self.receipt(started["run_id"])["lead"]["disposition"]) + self.task["lead"]["mode"] = "host" + started = self.service.start(self.task, "waiting-cancel", _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + self.assertEqual(self.service.cancel(started["run_id"])["state"], "cancelled") + self.assertEqual(self.receipt(started["run_id"])["worker_invocations"], 3) + + def test_queued_cancel_and_restart_use_saved_exact_snapshot(self): + with patch.object(self.service, "_spawn_daemon"): + started = self.service.start(self.task, "queued-cancel", _internal_council_fixture=self.fixture) + self.assertEqual(self.service.cancel(started["run_id"])["state"], "cancelled") + self.assertEqual(self.receipt(started["run_id"])["worker_invocations"], 0) + with patch.object(self.service, "_spawn_daemon"): + started = self.service.start(self.task, "restart", _internal_council_fixture=self.fixture) + self.service = Service(self.runtime) + self.service.resume(started["run_id"]) + self.assertEqual(self.wait(started["run_id"], {"succeeded", "failed"})["state"], "succeeded") + + def test_guided_claim_crash_and_submitted_crash_resume_exact_decision(self): + self.task["lead"]["mode"] = "host" + for boundary in ("record_handoff_submission", "complete_handoff_terminal"): + started = self.service.start(self.task, f"crash-{boundary}", _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + with patch.object(Store, boundary, side_effect=RuntimeError("injected crash")): + with self.assertRaises(RuntimeError): + self.service.finish_council(started["run_id"], "accept", "frozen crash choice", chosen="A", + supported_claims=["No blind retry"], discarded_alternatives=["Blind retry"], validation="Duplicate request test") + self.service = Service(self.runtime) + result = self.service.resume(started["run_id"]) + self.assertEqual(result["state"], "succeeded") + self.assertEqual(self.receipt(started["run_id"])["lead"]["choice"]["reason"], "frozen crash choice") + + def test_expired_guided_claim_reacquires_only_its_frozen_intent(self): + self.task["lead"]["mode"] = "host" + started = self.service.start(self.task, "expired-guided", _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + with patch.object(Store, "record_handoff_submission", side_effect=RuntimeError("crash after claim")): + with self.assertRaises(RuntimeError): + self.service.finish_council(started["run_id"], "accept", "intent before expiry", chosen="B", + supported_claims=["claim"], discarded_alternatives=[], validation="check") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + from datetime import datetime, timedelta, timezone + expires = store.handoff_snapshot(started["run_id"]).claim.expires_at + store.close() + future = datetime.fromisoformat(expires) + timedelta(seconds=1) + class FutureDateTime(datetime): + @classmethod + def now(cls, tz=None): + return future + with patch("devsquad.council_runtime.datetime", FutureDateTime), patch("devsquad.store._authoritative_now", return_value=future): + self.assertEqual(Service(self.runtime).resume(started["run_id"])["state"], "succeeded") + + def test_same_owner_low_level_takeover_does_not_recover_old_guided_intent(self): + self.task["lead"]["mode"] = "host" + started = self.service.start(self.task, "intent-race", _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + with patch.object(Store, "record_handoff_submission", side_effect=RuntimeError("crash")): + with self.assertRaises(RuntimeError): + self.service.finish_council(started["run_id"], "accept", "old intent", chosen="A", + supported_claims=["claim"], discarded_alternatives=[], validation="check") + from datetime import datetime, timedelta + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + future = datetime.fromisoformat(store.handoff_snapshot(started["run_id"]).claim.expires_at) + timedelta(seconds=1) + store.close() + with patch("devsquad.store._authoritative_now", return_value=future): + version = self.service.status(started["run_id"])["version"] + self.service.handoff_claim(started["run_id"], version, "terminal-operator") + with self.assertRaises(ConflictError): + self.service.resume(started["run_id"]) + self.assertEqual(self.service.status(started["run_id"])["state"], "awaiting_host") + + def test_preflight_quota_exhaustion_is_zero_launches_not_consensus(self): + from datetime import datetime, timedelta, timezone + now = datetime.now(timezone.utc) + self.service.capacity_observe({"schema_version": 1, "observation_id": "quota-zero", "pool_id": "shared", "window_id": "subscription", + "applies_to": {"harnesses": [], "model_families": [], "model_ids": []}, + "source": "native_reported", "observed_at": now.isoformat(), "expires_at": (now + timedelta(minutes=5)).isoformat(), + "used": 100, "limit": 100, "unit": "percent", "resets_at": (now + timedelta(minutes=5)).isoformat(), "confidence": "confirmed"}) + started = self.service.start(self.task, "quota", _internal_council_fixture=self.fixture) + self.assertEqual(started["state"], "failed") + self.assertEqual(self.receipt(started["run_id"])["worker_invocations"], 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_decision_helper.py b/test/core/test_decision_helper.py new file mode 100644 index 0000000..d7dd987 --- /dev/null +++ b/test/core/test_decision_helper.py @@ -0,0 +1,297 @@ +import copy +import hashlib +import json +from pathlib import Path +import sys +import unittest + + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +sys.path.insert(0, str(ROOT / "test/core/fakes")) + +from decision_adapter import FakeDecisionAdapter +from devsquad.contracts import ContractError +from devsquad.decision import ( + apply_decision_response, + build_decision_request, + decision_cache_key, + decision_fallback, + validate_decision_policy, + validate_decision_response, +) +from devsquad.router import resolve_routing +from devsquad.validation import validate_policy + + +def profile(profile_id, *, permission="read_only", quality="proven"): + return { + "id": profile_id, + "harness": "fixture", + "model_family": "fixture-family", + "model_id": f"model-{profile_id}", + "effort": {"value": "high", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": permission, + "account_pool_id": "fixture-pool", + "billing_mode": "subscription", + "quality_status": quality, + "evidence_refs": [f"evidence-{profile_id}"], + } + + +def decision_config(mode="shadow"): + return { + "schema_version": 1, + "mode": mode, + "purpose": { + "id": "profile-ranking", + "version": 1, + "question_sha256": "a" * 64, + "rubric_sha256": "b" * 64, + }, + "adapter": { + "id": "fixture", + "model": "fixture-v1", + "runtime_revision": "fixture-runtime-1", + "calibration_version": None, + }, + "language": "en", + "min_confidence": 0.7, + "gate_evidence_sha256": "c" * 64 if mode == "advisory" else None, + "budget": { + "max_calls": 1, + "max_input_bytes": 4096, + "wall_seconds": 2, + "max_cost_usd": 0.01, + }, + } + + +class DecisionHelperTest(unittest.TestCase): + def setUp(self): + self.task = { + "schema_version": 1, + "project": { + "repo_path": "/tmp/decision-fixture", + "base_ref": "base", + "target_ref": "target", + }, + "workflow": "branch-review", + "goal": "Review the bounded change.", + "task_class": "fixture-review", + "acceptance": [{ + "id": "review", "description": "Return a bound review.", + "evidence_kind": "review", + }], + "checks": [], + "scope": {"read_paths": ["src"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "profiles.json", "policy_file": "policy.json", + }, + "budget": { + "wall_seconds": 60, "max_worker_invocations": 2, + "max_revisions": 0, "max_fallbacks_per_step": 1, + }, + "origin": {"surface": "test"}, + } + self.registry = { + "schema_version": 1, + "profiles": [ + profile("review-a"), profile("review-b"), + profile("writer", permission="workspace_write"), + profile("trial", quality="trial"), + ], + "bindings": {}, + } + self.policy = { + "schema_version": 1, + "id": "fixture-policy", + "version": 1, + "roles": {"reviewer": [ + {"kind": "profile", "id": "review-a"}, + {"kind": "profile", "id": "writer"}, + {"kind": "profile", "id": "trial"}, + {"kind": "profile", "id": "review-b"}, + ]}, + "task_classes": {"fixture-review": "proven"}, + "require_different_model_for_review": False, + "prefer_different_harness_for_review": False, + "account_pools": {"fixture-pool": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 2, + "unknown_capacity_policy": "allow_bounded", + }}, + "experiment_budget": {}, + } + self.routing = resolve_routing(self.task, self.registry, self.policy) + + def request(self, mode="shadow", payload="bounded evidence"): + policy = {**self.policy, "decision_helper": decision_config(mode)} + return policy, build_decision_request( + self.task, self.routing, policy, payload, + ) + + def test_default_off_is_exactly_equivalent_and_does_not_build_a_request(self): + validate_policy(self.policy) + self.assertEqual( + validate_decision_policy(None), {"schema_version": 1, "mode": "off"}, + ) + self.assertIsNone(build_decision_request( + self.task, self.routing, self.policy, "ignored", + )) + off = apply_decision_response( + self.routing, + {}, + {}, + {"schema_version": 1, "mode": "off"}, + ) + self.assertEqual(off, self.routing) + self.assertNotIn("decision_helper", off) + + def test_shadow_records_a_valid_suggestion_without_changing_routes(self): + policy, request = self.request("shadow") + adapter = FakeDecisionAdapter({"reviewer": ["review-b", "review-a"]}) + response = adapter.decide(request) + routed = apply_decision_response( + self.routing, request, response, policy["decision_helper"], + ) + self.assertEqual(adapter.calls, 1) + self.assertEqual(routed["roles"], self.routing["roles"]) + self.assertEqual( + routed["decision_helper"]["role_status"], + {"reviewer": "shadow_mode"}, + ) + self.assertEqual(routed["decision_helper"]["applied_roles"], []) + + def test_advisory_can_only_reorder_the_eligible_set(self): + policy, request = self.request("advisory") + self.assertEqual(request["candidates"], { + "reviewer": ["review-a", "review-b"], + }) + response = FakeDecisionAdapter({ + "reviewer": ["review-b", "review-a"], + }).decide(request) + routed = apply_decision_response( + self.routing, request, response, policy["decision_helper"], + ) + reviewer = routed["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "review-b") + self.assertEqual( + [item["profile_id"] for item in reviewer["fallbacks"]], ["review-a"], + ) + self.assertEqual( + {item["profile_id"] for item in reviewer["excluded"]}, + {"writer", "trial"}, + ) + + attacked = copy.deepcopy(response) + attacked["recommendations"]["reviewer"]["ranking"][0] = "writer" + with self.assertRaisesRegex(ContractError, "eligible set"): + validate_decision_response( + request, attacked, policy["decision_helper"], + ) + + def test_pin_unknown_ids_nan_and_identity_drift_fail_closed(self): + self.task["routing"]["overrides"] = { + "reviewer": {"profile_id": "review-a", "fallback": "policy"}, + } + pinned_routing = resolve_routing(self.task, self.registry, self.policy) + policy = {**self.policy, "decision_helper": decision_config("advisory")} + request = build_decision_request( + self.task, pinned_routing, policy, "bounded evidence", + ) + response = FakeDecisionAdapter().decide(request) + routed = apply_decision_response( + pinned_routing, request, response, policy["decision_helper"], + ) + self.assertEqual( + routed["roles"]["reviewer"]["selected"]["profile_id"], "review-a", + ) + self.assertEqual( + routed["decision_helper"]["role_status"]["reviewer"], "pinned_route", + ) + + malformed = copy.deepcopy(response) + malformed["recommendations"]["reviewer"]["probabilities"]["review-a"] = float("nan") + with self.assertRaisesRegex(ContractError, "finite probability"): + validate_decision_response(request, malformed, policy["decision_helper"]) + drifted = copy.deepcopy(response) + drifted["adapter"]["model"] = "fixture-latest" + with self.assertRaisesRegex(ContractError, "identity drifted"): + validate_decision_response(request, drifted, policy["decision_helper"]) + + def test_input_drift_changes_cache_key_and_unusable_outputs_preserve_order(self): + policy, request_a = self.request("advisory", "evidence A") + _, request_b = self.request("advisory", "evidence B") + self.assertNotEqual(decision_cache_key(request_a), decision_cache_key(request_b)) + response = FakeDecisionAdapter( + {"reviewer": ["review-b", "review-a"]}, confidence=0.4, + ).decide(request_a) + routed = apply_decision_response( + self.routing, request_a, response, policy["decision_helper"], + ) + self.assertEqual(routed["roles"], self.routing["roles"]) + self.assertEqual( + routed["decision_helper"]["role_status"]["reviewer"], + "below_confidence_gate", + ) + fallback = decision_fallback( + self.routing, "advisory", "invalid_response", + hashlib.sha256(json.dumps(request_a, sort_keys=True).encode()).hexdigest(), + ) + self.assertEqual(fallback["roles"], self.routing["roles"]) + self.assertEqual(fallback["decision_helper"]["applied_roles"], []) + + def test_advisory_requires_gate_and_budget_or_truncation_cannot_be_hidden(self): + invalid = decision_config("advisory") + invalid["gate_evidence_sha256"] = None + with self.assertRaisesRegex(ContractError, "reviewed gate"): + validate_decision_policy(invalid) + policy = {**self.policy, "decision_helper": decision_config("shadow")} + with self.assertRaisesRegex(ContractError, "byte budget"): + build_decision_request( + self.task, self.routing, policy, "x" * 4097, + ) + request = build_decision_request( + self.task, self.routing, policy, "x" * 4097, truncated=True, + ) + response = FakeDecisionAdapter().decide(request) + routed = apply_decision_response( + self.routing, request, response, policy["decision_helper"], + ) + self.assertEqual(routed["roles"], self.routing["roles"]) + self.assertEqual( + routed["decision_helper"]["role_status"]["reviewer"], "shadow_mode", + ) + + def test_frozen_baseline_uses_only_the_synthetic_labeled_corpus(self): + experiment_dir = ROOT / "docs/plans/engineering-team/experiments" + baseline = json.loads( + (experiment_dir / "decision-helper-baseline-v1.json").read_text(), + ) + self.assertEqual(set(baseline), { + "schema_version", "experiment_id", "status", "purpose", "corpus", + "split", "modes", "resource_ceiling", "acceptance", "adoption", + }) + corpus_path = experiment_dir / baseline["corpus"]["path"] + self.assertEqual( + hashlib.sha256(corpus_path.read_bytes()).hexdigest(), + baseline["corpus"]["sha256"], + ) + corpus = json.loads(corpus_path.read_text()) + case_ids = [case["id"] for case in corpus["cases"]] + self.assertEqual(case_ids, baseline["corpus"]["case_ids"]) + self.assertEqual( + set(baseline["split"]["mechanics"] + baseline["split"]["held_out"]), + set(case_ids), + ) + self.assertFalse(baseline["corpus"]["contains_private_content"]) + self.assertEqual(baseline["resource_ceiling"]["network_calls_in_baseline"], 0) + self.assertFalse(baseline["adoption"]["advisory_authorized_by_baseline"]) + self.assertEqual(baseline["adoption"]["runtime_default"], "off") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_decision_store.py b/test/core/test_decision_store.py new file mode 100644 index 0000000..dcea3ca --- /dev/null +++ b/test/core/test_decision_store.py @@ -0,0 +1,205 @@ +import copy +from datetime import datetime, timezone +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +sys.path.insert(0, str(ROOT / "test/core/fakes")) + +from decision_adapter import FakeDecisionAdapter +from devsquad.contracts import ContractError +from devsquad.decision import build_decision_request +from devsquad.store import ConflictError, Store + + +NOW = datetime(2026, 9, 29, 3, 0, tzinfo=timezone.utc) + + +def config(mode="shadow"): + return { + "schema_version": 1, + "mode": mode, + "purpose": { + "id": "profile-ranking", + "version": 1, + "question_sha256": "a" * 64, + "rubric_sha256": "b" * 64, + }, + "adapter": { + "id": "fixture", + "model": "fixture-v1", + "runtime_revision": "fixture-runtime-1", + "calibration_version": None, + }, + "language": "en", + "min_confidence": 0.7, + "gate_evidence_sha256": None, + "budget": { + "max_calls": 1, + "max_input_bytes": 4096, + "wall_seconds": 2, + "max_cost_usd": 0.01, + }, + } + + +class DecisionStoreTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-decision-store-") + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], + check=True, + ) + (self.repo / "README").write_text("fixture\n") + subprocess.run(["git", "-C", str(self.repo), "add", "README"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.store = Store(self.root / "state.sqlite3", self.root / "artifacts") + self.task = {"scope": {"read_paths": ["README"], "write_paths": []}} + self.routing = { + "profile_registry": {"sha256": "d" * 64}, + "roles": {"reviewer": { + "source": "automatic", + "selected": {"profile_id": "profile-a"}, + "fallbacks": [{"profile_id": "profile-b"}], + }}, + } + self.policy = {"decision_helper": config()} + self.request = build_decision_request( + self.task, self.routing, self.policy, "redacted fixture evidence", + ) + + def tearDown(self): + self.store.close() + self.temp.cleanup() + + def run_id(self, key): + return self.store.claim_start( + self.repo, key, {"fixture": key}, f"owner-{key}", + ).run_id + + def test_completed_call_is_cached_across_resume_and_runs(self): + run_one = self.run_id("run-one") + claimed = self.store.claim_decision_observation( + run_one, "shadow", self.request, "decision-owner-1", now=NOW, + ) + self.assertEqual((claimed["action"], claimed["billable_calls"]), ("claimed", 0)) + launched = self.store.launch_decision_call( + claimed["cache_key"], "decision-owner-1", now=NOW, + ) + self.assertEqual(launched["billable_calls"], 1) + adapter = FakeDecisionAdapter({"reviewer": ["profile-b", "profile-a"]}) + completed = self.store.complete_decision_call( + claimed["cache_key"], "decision-owner-1", + adapter.decide(self.request), config(), now=NOW, + ) + self.assertEqual((completed["status"], adapter.calls), ("succeeded", 1)) + + resumed = self.store.claim_decision_observation( + run_one, "shadow", self.request, "decision-owner-2", now=NOW, + ) + self.assertEqual((resumed["action"], resumed["billable_calls"]), ("cached", 1)) + run_two = self.run_id("run-two") + shared = self.store.claim_decision_observation( + run_two, "shadow", self.request, "decision-owner-3", now=NOW, + ) + self.assertEqual((shared["action"], shared["billable_calls"]), ("cached", 1)) + self.assertEqual(shared["response"], completed["response"]) + + effect = { + "mode": "shadow", "applied_roles": [], + "role_status": {"reviewer": "shadow_mode"}, + } + recorded = self.store.record_decision_effect( + run_two, "profile-ranking", shared["cache_key"], effect, now=NOW, + ) + self.assertFalse(recorded["applied"]) + self.assertTrue(self.store.record_decision_effect( + run_two, "profile-ranking", shared["cache_key"], effect, now=NOW, + )["replayed"]) + + def test_launched_unknown_outcome_abstains_without_duplicate_call(self): + run_id = self.run_id("run-crash") + claimed = self.store.claim_decision_observation( + run_id, "shadow", self.request, "decision-owner-old", now=NOW, + ) + self.store.launch_decision_call( + claimed["cache_key"], "decision-owner-old", now=NOW, + ) + recovered = self.store.claim_decision_observation( + run_id, "shadow", self.request, "decision-owner-new", now=NOW, + ) + self.assertEqual(recovered["status"], "indeterminate") + self.assertEqual(recovered["action"], "abstain") + self.assertEqual(recovered["billable_calls"], 1) + with self.assertRaisesRegex(ConflictError, "not launchable"): + self.store.launch_decision_call( + claimed["cache_key"], "decision-owner-new", now=NOW, + ) + + def test_cancellation_before_launch_records_zero_calls(self): + run_id = self.run_id("run-cancel") + claimed = self.store.claim_decision_observation( + run_id, "shadow", self.request, "decision-owner", now=NOW, + ) + cancelled = self.store.cancel_run_decision_observations( + run_id, now=NOW, + )[0] + self.assertEqual((cancelled["status"], cancelled["billable_calls"]), ("cancelled", 0)) + replay = self.store.claim_decision_observation( + run_id, "shadow", self.request, "decision-owner-new", now=NOW, + ) + self.assertEqual((replay["action"], replay["billable_calls"]), ("abstain", 0)) + + def test_invalid_response_is_redacted_and_cannot_be_retried(self): + run_id = self.run_id("run-invalid") + claimed = self.store.claim_decision_observation( + run_id, "shadow", self.request, "decision-owner", now=NOW, + ) + self.store.launch_decision_call( + claimed["cache_key"], "decision-owner", now=NOW, + ) + invalid = FakeDecisionAdapter().decide(self.request) + invalid["recommendations"]["reviewer"]["probabilities"]["profile-a"] = float("nan") + completed = self.store.complete_decision_call( + claimed["cache_key"], "decision-owner", invalid, config(), now=NOW, + ) + self.assertEqual(completed["status"], "invalid") + self.assertIsNone(completed["response"]) + self.assertEqual(completed["billable_calls"], 1) + saved = self.store.decision_observation(run_id, "profile-ranking") + self.assertEqual(saved["status"], "invalid") + self.assertIsNone(saved["response"]) + with self.assertRaisesRegex(ConflictError, "not launchable"): + self.store.launch_decision_call( + claimed["cache_key"], "decision-owner", now=NOW, + ) + + def test_schema_thirteen_contains_decision_cache_and_run_links(self): + version = self.store.connection.execute( + "SELECT MAX(version) FROM schema_migrations", + ).fetchone()[0] + from devsquad.store import SUPPORTED_SCHEMA_VERSION + self.assertEqual(version, SUPPORTED_SCHEMA_VERSION) + tables = { + row[0] for row in self.store.connection.execute( + "SELECT name FROM sqlite_master WHERE type='table'", + ) + } + self.assertTrue({"decision_cache", "run_decision_observations"} <= tables) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_delivery_experiment_integrity.py b/test/core/test_delivery_experiment_integrity.py new file mode 100644 index 0000000..5fff207 --- /dev/null +++ b/test/core/test_delivery_experiment_integrity.py @@ -0,0 +1,111 @@ +"""Delivery exposure requires semantically valid, durably imported evidence.""" + +from contextlib import contextmanager +import hashlib +import json +from pathlib import Path +import tempfile +import unittest + +from experiment_runtime_fixture import ExperimentRuntimeFixture +from devsquad.contracts import ContractError +from devsquad.store import canonical_json + + +class DeliveryExperimentIntegrityTest(unittest.TestCase): + def setUp(self): + temporary = tempfile.TemporaryDirectory(prefix="devsquad-delivery-evidence-") + self.addCleanup(temporary.cleanup) + self.fixture = ExperimentRuntimeFixture(Path(temporary.name), workflow="issue-delivery") + self.addCleanup(self.fixture.close) + self.run_id = self.fixture.run_arm("eval-1", "candidate") + self.store = self.fixture.store() + self.addCleanup(self.store.close) + + @contextmanager + def changed_artifact(self, row, content, *, attempt=None): + path = Path(row["path"]) + original = path.read_bytes() + digest = hashlib.sha256(content).hexdigest() + path.write_bytes(content) + self.store.connection.execute("UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (digest, len(content), row["id"])) + if attempt is not None: + metadata = json.loads(attempt["output_metadata"]) + metadata["stdout"].update(captured_sha256=digest, captured_bytes=len(content), + full_sha256=digest, total_bytes=len(content)) + self.store.connection.execute("UPDATE attempts SET output_metadata=? WHERE id=?", + (canonical_json(metadata), attempt["id"])) + try: + yield + finally: + path.write_bytes(original) + self.store.connection.execute("UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (row["sha256"], row["byte_size"], row["id"])) + if attempt is not None: + self.store.connection.execute("UPDATE attempts SET output_metadata=? WHERE id=?", + (attempt["output_metadata"], attempt["id"])) + + def rejected(self): + with self.assertRaises(ContractError): + self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM experiments").fetchone()[0], 0) + + def test_hash_consistent_malformed_delivery_streams_cannot_count_as_exposure(self): + for role in ("implementer", "reviewer"): + attempt = dict(self.store.connection.execute( + "SELECT * FROM attempts WHERE run_id=? AND role=?", (self.run_id, role)).fetchone()) + row = dict(self.store.connection.execute( + "SELECT * FROM artifacts WHERE id=?", (attempt["stdout_artifact_id"],)).fetchone()) + with self.subTest(role=role), self.changed_artifact(row, b'{"not":"worker evidence"}', attempt=attempt): + self.rejected() + + def test_successful_delivery_requires_both_imports(self): + for prefix in ("implementation", "review"): + row = self.store.connection.execute( + "SELECT id,name FROM artifacts WHERE run_id=? AND name LIKE ?", + (self.run_id, f"{prefix}-attempt-%.json")).fetchone() + with self.subTest(prefix=prefix): + self.store.connection.execute("UPDATE artifacts SET name='hidden-import.json' WHERE id=?", (row["id"],)) + try: + self.rejected() + finally: + self.store.connection.execute("UPDATE artifacts SET name=? WHERE id=?", (row["name"], row["id"])) + + def test_saved_candidate_and_imported_usage_are_bound_to_the_stream(self): + for name in ("candidate-1.json", "implementation-attempt-%", "review-attempt-%", "checks-%"): + row = dict(self.store.connection.execute( + "SELECT * FROM artifacts WHERE run_id=? AND name LIKE ?", (self.run_id, name)).fetchone()) + changed = json.loads(Path(row["path"]).read_bytes()) + if name.startswith("candidate"): + changed["commit_oid"] = "0" * 40 + elif name.startswith("checks"): + changed["results"][0]["target_oid"] = "0" * 40 + else: + attempt = changed.get("attempt", changed) + attempt["usage"] = {"input_tokens": 1, "output_tokens": 1, + "total_tokens": 2, "source": "native_reported"} + with self.subTest(name=name), self.changed_artifact(row, canonical_json(changed).encode()): + self.rejected() + + def test_real_failed_writer_requires_an_exact_terminal_receipt(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "failed-writer", workflow="issue-delivery", fail_candidate=True) + self.addCleanup(fixture.close) + fixture.run_all() + run_id = fixture.runs[("eval-1", "candidate")] + store = fixture.store() + self.addCleanup(store.close) + self.assertEqual(fixture.service.status(run_id)["state"], "failed") + evaluated = fixture.service.policy_evaluate(fixture.spec) + self.assertEqual(evaluated["evaluation"]["cases"][0]["candidate_verdict"], "failed") + row = dict(store.connection.execute("SELECT * FROM artifacts WHERE run_id=? AND name='result-receipt.json'", (run_id,)).fetchone()) + content = json.loads(Path(row["path"]).read_bytes()) + content["attempts"][0]["status"] = "succeeded" + original_store = self.store + self.store = store + try: + with self.changed_artifact(row, canonical_json(content).encode()): + with self.assertRaisesRegex(ContractError, "failed implementer receipt|stale"): + fixture.service.policy_evaluate(fixture.spec) + finally: + self.store = original_store diff --git a/test/core/test_delivery_identity_runtime.py b/test/core/test_delivery_identity_runtime.py new file mode 100644 index 0000000..d979e12 --- /dev/null +++ b/test/core/test_delivery_identity_runtime.py @@ -0,0 +1,462 @@ +"""Native execution identity gates through public durable delivery operations.""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import shlex +import sys +import time +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.reports import TERMINAL_REPORT_NAMES +from devsquad.service import Service +from devsquad.store import Store, canonical_json +import test_delivery_workflow as delivery_fixtures + + +class PublicDeliveryIdentityTest(unittest.TestCase): + MODEL = "claude-sonnet-4-6" + FALLBACK_MODEL = "claude-opus-4-6" + + def setUp(self): + # Compose the existing fixture instead of inheriting all its test cases. + self.fixture = delivery_fixtures.DeliveryWorkspaceTest(methodName="runTest") + self.fixture.setUp() + self.addCleanup(self.fixture.doCleanups) + self.root, self.repo = self.fixture.root, self.fixture.repo + self.service = Service(self.fixture.runtime) + self.run_ids = [] + self.fallbacks = False + self.addCleanup(self.cancel_unfinished_runs) + self.output = self.root / "native-result.json" + fake_bin = self.root / "fake-bin" + fake_bin.mkdir() + self.fake_claude = fake_bin / "claude" + self.fake_claude.write_text( + "#!/bin/sh\n" + "if [ \"$1\" = \"--version\" ]; then\n" + " printf '%s\\n' '2.1.220 (Claude Code)'\n" + " exit 0\n" + "fi\n" + "printf '%s\\n' \"VALUE = 'fixed'\" > src/app.py\n" + f"/bin/cat {shlex.quote(str(self.output))}\n" + ) + self.fake_claude.chmod(0o700) + (fake_bin / "codex").symlink_to(ROOT / "test/core/fakes/codex_review_cli.py") + fake_auth = self.root / "native-auth.json" + fake_auth.write_text("{}\n") + fake_auth.chmod(0o600) + self.environment = {"PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}"} + # Only preflight's file lookup is patched. Both worker binaries and all + # durable service/import/review paths execute normally in child processes. + self.auth_patch = patch( + "devsquad.codex_review_worker._subscription_auth_file", + return_value=fake_auth, + ) + self.auth_patch.start() + self.addCleanup(self.auth_patch.stop) + + def cancel_unfinished_runs(self): + for run_id in self.run_ids: + if self.service.status(run_id)["state"] not in {"succeeded", "failed", "cancelled"}: + self.service.cancel(run_id) + + @staticmethod + def counters(): + return {"inputTokens": 12, "outputTokens": 7} + + def native_result(self): + return { + "type": "result", "subtype": "success", "is_error": False, + "result": "Applied the scoped native fixture implementation.", + "session_id": "native-delivery-session", + "usage": {"input_tokens": 12, "output_tokens": 7}, + "modelUsage": {self.MODEL: self.counters()}, + } + + def configure(self, *, requested_model=None, reviewer_model="gpt-fake-review", fallback=False): + profiles, policy = self.fixture.delivery_routing_documents() + self.fallbacks = fallback + profiles["profiles"] = [ + profile for profile in profiles["profiles"] + if fallback or profile["id"] != "fixture-implementer-fallback" + ] + by_id = {profile["id"]: profile for profile in profiles["profiles"]} + implementer = by_id["fixture-implementer"] + reviewer = by_id["fixture-reviewer"] + lead = by_id["fixture-lead"] + implementer.update({ + "harness": "claude", "model_family": "writer-family-label", + "model_id": requested_model or self.MODEL, + "effort": {"value": "high", "transport": "native"}, + }) + if fallback: + by_id["fixture-implementer-fallback"].update({ + "harness": "claude", "model_family": "fallback-family-label", + "model_id": self.FALLBACK_MODEL, + "effort": {"value": "high", "transport": "native"}, + }) + reviewer.update({ + "harness": "codex", "model_family": "distinct-reviewer-label", + "model_id": reviewer_model, + }) + lead.update({"harness": "codex", "model_id": "gpt-fake-lead"}) + if not fallback: + policy["roles"]["implementer"] = policy["roles"]["implementer"][:1] + for name, document in (("profiles", profiles), ("policy", policy)): + (self.repo / "devsquad" / f"{name}.json").write_text(json.dumps(document)) + self.fixture.git(self.repo, "add", "devsquad") + self.fixture.git(self.repo, "commit", "-qm", "native identity fixture") + self.fixture.baseline = self.fixture.git(self.repo, "rev-parse", "HEAD").strip() + self.fixture.source_refs = self.fixture.git(self.repo, "show-ref") + self.fixture.source_status = self.fixture.git(self.repo, "status", "--porcelain") + + def task(self, mode="host"): + task = self.fixture.delivery_task() + task["lead"] = {"mode": mode} + task["budget"]["wall_seconds"] = 60 + task["budget"]["max_fallbacks_per_step"] = int(self.fallbacks) + return task + + def start(self, document, key, *, mode="host"): + self.output.write_text(json.dumps(document)) + with patch.dict(os.environ, self.environment): + started = self.service.start(self.task(mode), key) + self.run_ids.append(started["run_id"]) + self.assertEqual(started["state"], "queued", started) + return started["run_id"] + + def wait(self, run_id, *, candidate=False): + deadline = time.monotonic() + 20 + while time.monotonic() < deadline: + status = self.service.status(run_id) + if status["state"] in {"succeeded", "failed", "cancelled", "awaiting_host"}: + return status + if (candidate and status["state"] == "queued" + and status.get("next_action") == "resume_candidate_review"): + return status + time.sleep(0.05) + self.fail(f"identity run did not reach its next gate: {self.service.status(run_id)}") + + def snapshot(self, run_id): + store = Store(self.service.database, self.service.artifacts) + try: + return json.loads(store.run(run_id)["mutable_snapshot"]) + finally: + store.close() + + def result_documents(self, run_id): + result = self.service.result(run_id) + artifacts = {item["name"]: item for item in result["artifacts"]} + self.assertTrue(TERMINAL_REPORT_NAMES <= set(artifacts)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + return receipt, artifacts + + def finish_valid_delivery(self, requested_model): + self.configure(requested_model=requested_model) + run_id = self.start(self.native_result(), f"valid-{requested_model}") + self.assertEqual(self.wait(run_id, candidate=True)["state"], "queued") + snapshot = self.snapshot(run_id) + attempt = snapshot["delivery_iterations"][0]["implementation"]["attempt"] + self.assertEqual(attempt["selected_profile"]["profile"]["model_id"], requested_model) + self.assertEqual(attempt["selected_profile"]["profile"]["effort"]["value"], "high") + self.assertEqual(attempt["observed_identity"]["model_id"], self.MODEL) + self.assertIsNone(attempt["observed_identity"]["effort"]) + self.assertIsNone(attempt["observed_identity"]["backing_revision"]) + self.assertEqual(attempt["usage"]["total_tokens"], 19) + self.assertEqual(attempt["native_ids"]["session_id"], "native-delivery-session") + self.assertTrue(self.service.resume(run_id)["launched"]) + status = self.wait(run_id) + self.assertEqual(status["state"], "awaiting_host", status) + claimed = self.service.handoff_claim(run_id, status["version"], "identity-host") + packet = claimed["handoff"]["packet"] + completed = self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + packet, "identity-accept", "accept", "Independent native review passed.", + ), + ) + self.assertEqual(completed["state"], "succeeded") + receipt, artifacts = self.result_documents(run_id) + self.assertEqual([item["role"] for item in receipt["attempts"]], ["implementer", "reviewer"]) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertIn("candidate-1.patch", artifacts) + self.fixture.assert_source_unchanged() + + def test_exact_native_identity_completes_public_delivery(self): + self.finish_valid_delivery(self.MODEL) + + def test_alias_resolves_to_observed_identity_through_public_delivery(self): + self.finish_valid_delivery("sonnet") + + def test_invalid_identity_never_publishes_candidate_and_retains_failed_evidence(self): + self.configure() + documents = {} + documents["missing"] = self.native_result() + del documents["missing"]["modelUsage"] + documents["multiple"] = self.native_result() + documents["multiple"]["modelUsage"]["claude-haiku-4-5-20251001"] = self.counters() + documents["mismatch"] = self.native_result() + documents["mismatch"]["modelUsage"] = {"claude-opus-4-6": self.counters()} + for kind, document in documents.items(): + with self.subTest(kind=kind): + run_id = self.start(document, f"invalid-{kind}") + status = self.wait(run_id, candidate=True) + self.assertEqual(status["state"], "failed", status) + self.assertIsNone(status["handoff"]) + receipt, artifacts = self.result_documents(run_id) + self.assertNotIn("candidate-1.json", artifacts) + self.assertNotIn("candidate-1.patch", artifacts) + self.assertEqual(receipt["state"], "failed") + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertEqual(len(receipt["attempts"]), 1) + failed_attempt = receipt["attempts"][0] + self.assertEqual(failed_attempt["role"], "implementer") + diagnostics = failed_attempt["native_diagnostics"] + self.assertEqual(diagnostics["schema_version"], 1) + self.assertEqual(diagnostics["identity_status"], "unverified") + self.assertTrue(diagnostics["reason"]) + self.assertEqual(diagnostics["usage"]["total_tokens"], 19) + self.assertEqual(diagnostics["session_id"], "native-delivery-session") + self.assertEqual( + diagnostics["output_sha256"], + hashlib.sha256(self.output.read_bytes()).hexdigest(), + ) + self.assertEqual(diagnostics["output_bytes"], self.output.stat().st_size) + # Bound native diagnostics must survive in the public receipt, + # not only in an unreferenced private stderr traceback. + for reported_model in document.get("modelUsage", {}): + self.assertIn(reported_model, diagnostics["model_usage"]) + captures = [name for name in artifacts if name.endswith((".stdout", ".stderr"))] + self.assertEqual(len(captures), 2) + reopened = Service(self.fixture.runtime) + self.assertEqual(reopened.result(run_id), self.service.result(run_id)) + replayed = reopened.start(self.task(), f"invalid-{kind}") + self.assertFalse(replayed["created"]) + self.assertEqual(replayed["run_id"], run_id) + self.assertEqual(replayed["state"], "failed") + store = Store(reopened.database, reopened.artifacts) + try: + self.assertEqual(store.worker_invocations(run_id), 1) + finally: + store.close() + self.fixture.assert_source_unchanged() + + def test_same_reported_model_cannot_pass_host_independence_gate(self): + self.assert_independence_blocked("host") + + def test_same_reported_model_cannot_pass_headless_independence_gate(self): + self.assert_independence_blocked("headless") + + def fallback_output(self, *, delay=False): + """Install bounded fake native outputs selected by frozen profile model.""" + fallback_result = self.native_result() + fallback_result["modelUsage"] = {self.FALLBACK_MODEL: self.counters()} + fallback_result["session_id"] = "native-fallback-session" + output_path = self.root / "fallback-result.json" + output_path.write_text(json.dumps(fallback_result)) + self.fallback_marker = self.root / "fallback-entered" + pause = "/bin/sleep 5\n" if delay else "" + self.fake_claude.write_text( + "#!/bin/sh\n" + "if [ \"$1\" = \"--version\" ]; then\n" + " printf '%s\\n' '2.1.220 (Claude Code)'\n" + " exit 0\n" + "fi\n" + "previous=''\n" + "for argument in \"$@\"; do\n" + f" if [ \"$previous\" = '--model' ] && [ \"$argument\" = '{self.FALLBACK_MODEL}' ]; then\n" + f" printf '%s\\n' 'started' > {shlex.quote(str(self.fallback_marker))}\n" + f" {pause}" + " printf '%s\\n' \"VALUE = 'fixed'\" > src/app.py\n" + f" /bin/cat {shlex.quote(str(output_path))}\n" + " exit 0\n" + " fi\n" + " previous=$argument\n" + "done\n" + f"/bin/cat {shlex.quote(str(self.output))}\n" + ) + failed_result = self.native_result() + failed_result["modelUsage"] = {"claude-haiku-4-5-20251001": self.counters()} + return failed_result + + def assert_failed_native_attempt_preserved(self, receipt): + failed = receipt["attempts"][0] + self.assertEqual(failed["role"], "implementer") + self.assertEqual(failed["status"], "failed") + self.assertEqual(failed["selected_profile"]["profile_id"], "fixture-implementer") + self.assertIsNone(failed["observed_identity"]) + self.assertEqual(failed["usage"]["total_tokens"], 19) + self.assertEqual(failed["usage"]["source"], "native_reported") + diagnostic = failed["native_diagnostics"] + self.assertEqual(diagnostic, failed["error"]["native_diagnostics"]) + self.assertEqual(diagnostic["identity_status"], "unverified") + self.assertEqual(diagnostic["session_id"], "native-delivery-session") + self.assertEqual(diagnostic["usage"], failed["usage"]) + self.assertIn("claude-haiku-4-5-20251001", diagnostic["model_usage"]) + self.assertEqual(diagnostic["output_sha256"], hashlib.sha256(self.output.read_bytes()).hexdigest()) + self.assertEqual(receipt["accounting"]["attempt_usage"][0], failed["usage"]) + + def ready_handoff(self, run_id): + self.assertEqual(self.wait(run_id, candidate=True)["state"], "queued") + self.assertTrue(self.service.resume(run_id)["launched"]) + status = self.wait(run_id) + self.assertEqual(status["state"], "awaiting_host", status) + return self.service.handoff_claim(run_id, status["version"], "identity-host") + + def test_native_fallback_keeps_failed_identity_and_usage_in_success_receipt(self): + self.configure(fallback=True) + run_id = self.start(self.fallback_output(), "native-identity-fallback") + claimed = self.ready_handoff(run_id) + self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + claimed["handoff"]["packet"], "accept-native-fallback", "accept", + "Verified fallback received independent native review.", + ), + ) + receipt, _ = self.result_documents(run_id) + self.assertEqual(receipt["state"], "succeeded") + self.assert_failed_native_attempt_preserved(receipt) + implementers = [item for item in receipt["attempts"] if item["role"] == "implementer"] + self.assertEqual(len(implementers), 2) + self.assertEqual(implementers[1]["selected_profile"]["profile_id"], "fixture-implementer-fallback") + self.assertEqual(implementers[1]["observed_identity"]["model_id"], self.FALLBACK_MODEL) + self.assertEqual( + {item["selected_profile"]["profile"]["permission_policy"] for item in implementers}, + {"workspace_write"}, + ) + self.assertEqual(receipt["accounting"]["worker_invocations"], 3) + self.fixture.assert_source_unchanged() + + def test_cancellation_keeps_prior_failed_native_identity_and_usage(self): + self.configure(fallback=True) + run_id = self.start(self.fallback_output(delay=True), "cancel-native-fallback") + deadline = time.monotonic() + 15 + while time.monotonic() < deadline and not self.fallback_marker.exists(): + self.assertNotIn(self.service.status(run_id)["state"], {"failed", "cancelled", "succeeded"}) + time.sleep(0.05) + self.assertTrue(self.fallback_marker.exists(), "fallback never reached its bounded native call") + self.service.cancel(run_id) + self.assertEqual(self.wait(run_id)["state"], "cancelled") + receipt, _ = self.result_documents(run_id) + self.assert_failed_native_attempt_preserved(receipt) + self.assertEqual(len(receipt["attempts"]), 2) + self.assertEqual(receipt["attempts"][1]["status"], "cancelled") + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual(receipt["lead"]["status"], "not_reached") + self.fixture.assert_source_unchanged() + + def downgrade_saved_identity_for_legacy_fixture(self, run_id): + """Simulate pre-v2 persisted identity; never use as qualification proof.""" + store = Store(self.service.database, self.service.artifacts) + try: + snapshot = json.loads(store.run(run_id)["mutable_snapshot"]) + implementation = snapshot["delivery_iterations"][0]["implementation"] + implementation["schema_version"] = 1 + attempt = implementation["attempt"] + old_fields = { + "harness", "harness_version", "model_provider", "model_id", + "effort", "permission_policy", "verification", + } + attempt["observed_identity"] = { + key: value for key, value in attempt["observed_identity"].items() + if key in old_fields + } + attempt["observed_identity"]["effort"] = attempt["selected_profile"]["profile"]["effort"]["value"] + store.connection.execute( + "UPDATE runs SET mutable_snapshot=? WHERE id=?", (canonical_json(snapshot), run_id), + ) + finally: + store.close() + + def test_legacy_identity_cannot_authorize_new_acceptance_but_rejection_replays(self): + self.configure() + run_id = self.start(self.native_result(), "legacy-new-acceptance") + claimed = self.ready_handoff(run_id) + packet = claimed["handoff"]["packet"] + self.downgrade_saved_identity_for_legacy_fixture(run_id) + self.service = Service(self.fixture.runtime) + with self.assertRaisesRegex(ContractError, "requires v2"): + self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + packet, "legacy-new-accept", "accept", "Try old copied identity evidence.", + ), + ) + decision = self.fixture.decision(packet, "legacy-reject", "reject", "New verification required.") + self.service.handoff_complete(run_id, claimed["claim"], decision) + receipt, _ = self.result_documents(run_id) + self.assertEqual(receipt["state"], "failed") + self.assertEqual(receipt["delivery_iterations"][0]["implementation"]["schema_version"], 1) + replay = Service(self.fixture.runtime).handoff_complete(run_id, claimed["claim"], decision) + self.assertTrue(replay["replayed"]) + self.assertEqual(replay["state"], "failed") + self.assertEqual(self.result_documents(run_id)[0], receipt) + self.fixture.assert_source_unchanged() + + def test_historical_successful_legacy_receipt_reads_and_exact_acceptance_replays(self): + self.configure() + run_id = self.start(self.native_result(), "historical-legacy-acceptance") + claimed = self.ready_handoff(run_id) + self.downgrade_saved_identity_for_legacy_fixture(run_id) + decision = self.fixture.decision( + claimed["handoff"]["packet"], "historical-accept", "accept", "Historical fixture acceptance.", + ) + # Construct the historical fixture under pre-v2 semantics, then remove + # the bypass before any assertions. This is compatibility data only. + with patch("devsquad.workflows.require_independent_delivery_review", return_value=None): + self.service.handoff_complete(run_id, claimed["claim"], decision) + self.service = Service(self.fixture.runtime) + receipt, _ = self.result_documents(run_id) + self.assertEqual(receipt["state"], "succeeded") + self.assertEqual(receipt["delivery_iterations"][0]["implementation"]["schema_version"], 1) + replay = self.service.handoff_complete(run_id, claimed["claim"], decision) + self.assertTrue(replay["replayed"]) + self.assertEqual(replay["state"], "succeeded") + self.assertEqual(self.result_documents(run_id)[0], receipt) + with self.assertRaises(ContractError): + self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + claimed["handoff"]["packet"], "different-new-accept", "accept", "Not an exact replay.", + ), + ) + self.fixture.assert_source_unchanged() + + def assert_independence_blocked(self, mode): + self.configure(requested_model="sonnet", reviewer_model=self.MODEL) + run_id = self.start(self.native_result(), f"same-model-{mode}", mode=mode) + status = self.wait(run_id, candidate=True) + if status["state"] == "queued": + resumed = self.service.resume(run_id) + self.assertTrue(resumed["launched"], resumed) + status = self.wait(run_id) + if status["state"] == "awaiting_host": + claimed = self.service.handoff_claim(run_id, status["version"], "same-model-host") + packet = claimed["handoff"]["packet"] + with self.assertRaises(ContractError): + self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + packet, "same-model-accept", "accept", "Try the same model despite new labels.", + ), + ) + self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + packet, "same-model-reject", "reject", "Independent review was not established.", + ), + ) + status = self.service.status(run_id) + self.assertEqual(status["state"], "failed", status) + receipt, _ = self.result_documents(run_id) + self.assertNotEqual(receipt["lead"].get("disposition"), "accept") + self.assertIn("independen", json.dumps(receipt).lower()) + self.fixture.assert_source_unchanged() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_delivery_workflow.py b/test/core/test_delivery_workflow.py new file mode 100644 index 0000000..7723f6c --- /dev/null +++ b/test/core/test_delivery_workflow.py @@ -0,0 +1,1145 @@ +from __future__ import annotations + +import hashlib +import json +import os +import signal +import subprocess +import sys +import tempfile +import time +import unittest +from pathlib import Path +from unittest.mock import patch + +CORE = Path(__file__).resolve().parents[2] / "plugin" / "core" +sys.path.insert(0, str(CORE / "src")) + +from devsquad.contracts import ContractError +from devsquad.claude_delivery_worker import ( + freeze_claude_implementer, + run as run_claude_implementer, +) +from devsquad.service import Service +from devsquad.store import ConflictError, Store, request_hash +from devsquad.supervisor import inspect_process +from devsquad.workspaces import ( + freeze_delivery_candidate, + prepare_delivery_workspace, + resolve_commit, +) +from devsquad.workflows import validate_branch_review_evidence + + +class DeliveryWorkspaceTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + self.remote = self.root / "remote.git" + self.repo.mkdir() + self.git(self.repo, "init", "-q") + self.git(self.repo, "config", "user.name", "Fixture") + self.git(self.repo, "config", "user.email", "fixture@example.test") + (self.repo / "src").mkdir() + (self.repo / "tests").mkdir() + (self.repo / "src/app.py").write_text("VALUE = 'base'\n") + (self.repo / "tests/test_app.py").write_text("# base test\n") + (self.repo / "README.md").write_text("fixture\n") + (self.repo / "devsquad").mkdir() + profiles, policy = self.delivery_routing_documents() + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) + self.git(self.repo, "add", ".") + self.git(self.repo, "commit", "-qm", "base") + self.baseline = resolve_commit(self.repo, "HEAD") + subprocess.run( + ["git", "init", "--bare", "-q", str(self.remote)], check=True, + ) + self.git(self.repo, "remote", "add", "origin", str(self.remote)) + self.git(self.repo, "push", "-q", "origin", "HEAD:refs/heads/main") + self.source_status = self.git(self.repo, "status", "--porcelain") + self.source_refs = self.git(self.repo, "show-ref") + self.remote_refs = self.git(self.remote, "show-ref") + + @staticmethod + def delivery_routing_documents(): + def profile( + profile_id: str, + *, + family: str, + model: str, + permission: str, + ) -> dict[str, object]: + return { + "id": profile_id, + "harness": "fixture", + "model_family": family, + "model_id": model, + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read", "write"] if permission == "workspace_write" else ["read"], + "permission_policy": permission, + "account_pool_id": f"{profile_id}-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["tracked-fixture"], + } + + profiles = { + "schema_version": 1, + "profiles": [ + profile( + "fixture-implementer", + family="fixture-family-a", + model="fixture-write-model", + permission="workspace_write", + ), + profile( + "fixture-implementer-fallback", + family="fixture-family-a2", + model="fixture-write-model-fallback", + permission="workspace_write", + ), + profile( + "fixture-reviewer", + family="fixture-family-b", + model="fixture-review-model", + permission="read_only", + ), + profile( + "fixture-lead", + family="fixture-family-c", + model="fixture-lead-model", + permission="read_only", + ), + ], + "bindings": {}, + } + policy = { + "schema_version": 1, + "id": "delivery-fixture-policy", + "version": 1, + "roles": { + "implementer": [ + {"kind": "profile", "id": "fixture-implementer"}, + {"kind": "profile", "id": "fixture-implementer-fallback"}, + ], + "reviewer": [ + {"kind": "profile", "id": "fixture-reviewer"} + ], + "lead": [{"kind": "profile", "id": "fixture-lead"}], + }, + "task_classes": {"fixture-delivery-small": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "fixture-implementer-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + "fixture-implementer-fallback-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + "fixture-reviewer-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + "fixture-lead-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + return profiles, policy + + @staticmethod + def git(repo: Path, *args: str) -> str: + return subprocess.run( + ["git", "-C", str(repo), *args], + check=True, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + ).stdout + + def prepare(self) -> dict[str, object]: + return prepare_delivery_workspace( + self.repo, + self.runtime, + "project-1", + "run-1", + self.baseline, + ("src", "tests"), + ("src/app.py", "tests"), + ) + + def delivery_task(self) -> dict[str, object]: + return { + "schema_version": 1, + "project": { + "repo_path": str(self.repo), + "base_ref": self.baseline, + "target_ref": self.baseline, + }, + "workflow": "issue-delivery", + "goal": "Fix the fixture value and add one regression test.", + "task_class": "fixture-delivery-small", + "acceptance": [ + { + "id": "value-fixed", + "description": "The fixture value is fixed.", + "evidence_kind": "check", + }, + { + "id": "independent-review", + "description": "A different model reviews the candidate.", + "evidence_kind": "review", + }, + ], + "checks": [ + { + "id": "fixture-check", + "argv": ["python3", "-c", "print('ok')"], + "cwd": ".", + "timeout_seconds": 10, + "required_to_pass": True, + } + ], + "scope": { + "read_paths": ["src", "tests"], + "write_paths": ["src/app.py", "tests"], + }, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + }, + "budget": { + "wall_seconds": 30, + "max_worker_invocations": 5, + "max_revisions": 1, + "max_fallbacks_per_step": 0, + }, + "origin": {"surface": "test"}, + } + + @staticmethod + def implementation_fixture(delay: float = 0) -> dict[str, object]: + return { + "writes": [ + {"path": "src/app.py", "content": "VALUE = 'fixed'\n"}, + { + "path": "tests/test regression.py", + "content": "def test_regression():\n assert True\n", + }, + ], + "delay_seconds": delay, + } + + @staticmethod + def repair_fixtures() -> dict[str, object]: + return { + "iterations": [ + { + "writes": [ + {"path": "src/app.py", "content": "VALUE = 'wrong'\n"}, + ], + "delay_seconds": 0, + }, + { + "writes": [ + {"path": "src/app.py", "content": "VALUE = 'fixed'\n"}, + ], + "delay_seconds": 0, + }, + ], + } + + @staticmethod + def clean_review_fixture() -> dict[str, object]: + return { + "verdict": "clean", + "summary": "The exact frozen candidate satisfies the task.", + "findings": [], + } + + @staticmethod + def decision( + packet: dict[str, object], + submission_id: str, + disposition: str, + reason: str, + ) -> dict[str, object]: + body = { + "schema_version": 1, + "submission_id": submission_id, + "disposition": disposition, + "reason": reason, + "evidence_refs": [ + { + "artifact_id": reference["artifact_id"], + "sha256": reference["sha256"], + } + for reference in packet["artifacts"] + ], + } + return {**body, "submission_hash": request_hash(body)} + + def wait_for_candidate(self, service: Service, run_id: str) -> dict[str, object]: + deadline = time.monotonic() + 15 + while time.monotonic() < deadline: + status = service.status(run_id) + if status["state"] == "queued" and status["next_action"] == "resume_candidate_review": + return status + if status["state"] in {"failed", "cancelled"}: + self.fail(f"delivery run terminalized early: {status}") + time.sleep(0.05) + self.fail(f"delivery candidate did not become ready: {service.status(run_id)}") + + def wait_for_handoff(self, service: Service, run_id: str) -> dict[str, object]: + deadline = time.monotonic() + 15 + while time.monotonic() < deadline: + status = service.status(run_id) + if status["state"] == "awaiting_host": + return status + if status["state"] in {"failed", "cancelled"}: + self.fail(f"delivery review terminalized early: {status}") + time.sleep(0.05) + self.fail(f"delivery handoff did not become ready: {service.status(run_id)}") + + def wait_for_terminal(self, service: Service, run_id: str) -> dict[str, object]: + deadline = time.monotonic() + 15 + while time.monotonic() < deadline: + status = service.status(run_id) + if status["state"] in {"succeeded", "failed", "cancelled"}: + return status + time.sleep(0.05) + self.fail(f"delivery run did not terminalize: {service.status(run_id)}") + + def assert_source_unchanged(self) -> None: + self.assertEqual(resolve_commit(self.repo, "HEAD"), self.baseline) + self.assertEqual(self.git(self.repo, "status", "--porcelain"), self.source_status) + self.assertEqual(self.git(self.repo, "show-ref"), self.source_refs) + self.assertEqual(self.git(self.remote, "show-ref"), self.remote_refs) + self.assertEqual((self.repo / "src/app.py").read_text(), "VALUE = 'base'\n") + + def test_scoped_candidate_commit_patch_and_replay_preserve_source_and_remote(self): + prepared = self.prepare() + workspace = Path(prepared["path"]) + self.assertEqual(self.git(workspace, "rev-parse", "--abbrev-ref", "HEAD").strip(), "HEAD") + (workspace / "src/app.py").write_text("VALUE = 'fixed'\n") + (workspace / "tests/test regression.py").write_text( + "def test_regression():\n assert True\n" + ) + + candidate, patch = freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assertEqual(candidate["baseline_oid"], self.baseline) + self.assertEqual(candidate["patch_sha256"], hashlib.sha256(patch).hexdigest()) + self.assertEqual( + candidate["captured_untracked_paths"], ["tests/test regression.py"], + ) + self.assertEqual( + candidate["changed_paths"], ["src/app.py", "tests/test regression.py"], + ) + self.assertIn(b"VALUE = 'fixed'", patch) + self.assertEqual( + self.git(workspace, "rev-parse", "HEAD^").strip(), self.baseline, + ) + self.assertEqual(self.git(workspace, "status", "--porcelain"), "") + replayed, replay_patch = freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assertEqual(replayed, candidate) + self.assertEqual(replay_patch, patch) + self.assert_source_unchanged() + + def test_frozen_claude_worker_edits_only_the_delivery_workspace(self): + prepared = self.prepare() + binary = self.root / "claude" + binary.write_text( + "#!/bin/sh\n" + "if [ \"$1\" = \"--version\" ]; then\n" + " printf '%s\\n' '2.1.220 (Claude Code)'\n" + " exit 0\n" + "fi\n" + "separator=0\n" + "for argument do\n" + " if [ \"$separator\" = 1 ]; then break; fi\n" + " if [ \"$argument\" = -- ]; then separator=1; fi\n" + "done\n" + "[ \"$separator\" = 1 ] || exit 9\n" + "printf '%s\\n' \"VALUE = 'fixed'\" > src/app.py\n" + "printf '%s\\n' '{\"type\":\"result\",\"subtype\":\"success\",\"is_error\":false," + "\"result\":\"Applied the bounded fix.\"," + "\"session_id\":\"session-fixture\"," + "\"modelUsage\":{\"claude-sonnet-fixture\":{\"inputTokens\":12,\"outputTokens\":7}}," + "\"usage\":{\"input_tokens\":12,\"output_tokens\":7}}'\n" + ) + binary.chmod(0o700) + profile = { + "id": "claude-implementer", + "harness": "claude", + "model_family": "claude-sonnet", + "model_id": "claude-sonnet-fixture", + "effort": {"value": "high", "transport": "native"}, + "required_tools": ["read", "write"], + "permission_policy": "workspace_write", + "account_pool_id": "claude-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["fixture"], + } + selected = { + "reference": {"kind": "profile", "id": profile["id"]}, + "binding": None, + "profile_id": profile["id"], + "profile_sha256": hashlib.sha256( + json.dumps(profile, sort_keys=True, separators=(",", ":")).encode() + ).hexdigest(), + "profile": profile, + } + with patch.dict(os.environ, {"PATH": str(self.root)}): + adapter = freeze_claude_implementer(selected) + snapshot = { + "task": self.delivery_task(), + "delivery_workspace": prepared, + "routing": { + "roles": { + "implementer": {"selected": selected, "fallbacks": []}, + }, + }, + "implementation_adapter": adapter, + "implementation_adapters": {profile["id"]: adapter}, + } + evidence = run_claude_implementer(snapshot) + self.assertEqual( + (Path(prepared["path"]) / "src/app.py").read_text(), + "VALUE = 'fixed'\n", + ) + self.assertEqual(evidence["attempt"]["observed_identity"]["harness"], "claude") + self.assertEqual(evidence["attempt"]["native_ids"]["session_id"], "session-fixture") + self.assertEqual(evidence["attempt"]["usage"]["total_tokens"], 19) + self.assert_source_unchanged() + + def test_later_mutation_cannot_replay_a_frozen_candidate(self): + workspace = Path(self.prepare()["path"]) + (workspace / "src/app.py").write_text("VALUE = 'candidate'\n") + freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + (workspace / "src/app.py").write_text("VALUE = 'stale'\n") + with self.assertRaisesRegex(ContractError, "later workspace changes"): + freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assert_source_unchanged() + + def test_out_of_scope_change_is_rejected_before_commit(self): + workspace = Path(self.prepare()["path"]) + (workspace / "README.md").write_text("unauthorized\n") + with self.assertRaisesRegex(ContractError, "outside write scope"): + freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assertEqual(resolve_commit(workspace, "HEAD"), self.baseline) + self.assert_source_unchanged() + + def test_new_symlink_cannot_escape_the_delivery_workspace(self): + workspace = Path(self.prepare()["path"]) + outside = self.root / "outside-secret" + outside.write_text("private\n") + (workspace / "tests/leak").symlink_to(outside) + with self.assertRaisesRegex(ContractError, "escapes its workspace"): + freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assertEqual(resolve_commit(workspace, "HEAD"), self.baseline) + self.assert_source_unchanged() + + def test_candidate_requires_a_change(self): + workspace = Path(self.prepare()["path"]) + with self.assertRaisesRegex(ContractError, "contains no changes"): + freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assert_source_unchanged() + + def test_durable_implementer_publishes_candidate_artifacts_once(self): + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "durable-delivery", + _internal_implementation_fixture=self.implementation_fixture(), + ) + status = self.wait_for_candidate(service, started["run_id"]) + self.assertEqual(status["phase"], None) + + store = Store(service.database, service.artifacts) + self.addCleanup(store.close) + run = store.run(started["run_id"]) + snapshot = json.loads(run["mutable_snapshot"]) + candidate = snapshot["candidate"] + attempts = store.attempts_for_run(started["run_id"]) + artifacts = {item["name"]: item for item in store.artifacts_for_run(started["run_id"])} + self.assertEqual(len(attempts), 1) + self.assertEqual(attempts[0]["role"], "implementer") + self.assertEqual(attempts[0]["status"], "finished") + self.assertEqual(store.worker_invocations(started["run_id"]), 1) + self.assertIn("candidate-1.json", artifacts) + self.assertIn("candidate-1.patch", artifacts) + self.assertIn( + f"implementation-attempt-{attempts[0]['id']}.json", artifacts, + ) + self.assertEqual( + artifacts["candidate-1.patch"]["sha256"], candidate["patch_sha256"], + ) + self.assertEqual( + resolve_commit(Path(snapshot["workspace"]["path"]), "HEAD"), + candidate["commit_oid"], + ) + self.assertEqual( + resolve_commit(Path(snapshot["check_workspace"]["path"]), "HEAD"), + candidate["commit_oid"], + ) + event_types = [ + event["type"] for event in store.events_for_run(started["run_id"]) + ] + self.assertEqual(event_types.count("delivery.candidate_ready"), 1) + self.assert_source_unchanged() + + def test_exact_candidate_is_reviewed_checked_and_published_for_lead(self): + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "reviewed-delivery", + _internal_implementation_fixture=self.implementation_fixture(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + resumed = service.resume(started["run_id"]) + self.assertTrue(resumed["launched"]) + status = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], status["version"], "fixture-host", + ) + packet = claimed["handoff"]["packet"] + + store = Store(service.database, service.artifacts) + self.addCleanup(store.close) + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual([item["role"] for item in attempts], ["implementer", "reviewer"]) + self.assertEqual(packet["workflow"], "issue-delivery") + self.assertEqual( + packet["candidate_sha256"], snapshot["candidate"]["candidate_sha256"], + ) + self.assertTrue(packet["evaluation"]["accept_allowed"]) + self.assertEqual(packet["checks"][0]["status"], "passed") + self.assertEqual( + {item["name"] for item in packet["artifacts"]}, + { + f"review-{attempts[1]['id']}.json", + f"checks-{attempts[1]['id']}.json", + f"evaluation-{attempts[1]['id']}.json", + f"review-attempt-{attempts[1]['id']}.json", + }, + ) + self.assert_source_unchanged() + + def test_host_accept_publishes_complete_delivery_receipt(self): + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "accepted-delivery", + _internal_implementation_fixture=self.implementation_fixture(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + status = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], status["version"], "fixture-host", + ) + packet = claimed["handoff"]["packet"] + completed = service.handoff_complete( + started["run_id"], + claimed["claim"], + self.decision(packet, "accept-delivery", "accept", "Candidate accepted."), + ) + self.assertEqual(completed["state"], "succeeded") + self.assertEqual(completed["continuation"]["action"], "terminal") + result = service.result(started["run_id"]) + receipt_artifact = next( + item for item in result["artifacts"] if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["workflow"], "issue-delivery") + self.assertEqual(receipt["candidate"]["sha256"], packet["candidate_sha256"]) + self.assertEqual( + [attempt["role"] for attempt in receipt["attempts"]], + ["implementer", "reviewer"], + ) + self.assertEqual(receipt["delivery_iterations"][0]["iteration"], 1) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual(receipt["lead"]["disposition"], "accept") + self.assert_source_unchanged() + + def test_required_failure_blocks_delivery_accept_and_allows_reject(self): + task = self.delivery_task() + task["checks"][0] = { + **task["checks"][0], + "argv": ["python3", "-c", "raise SystemExit(1)"], + } + service = Service(self.runtime) + started = service.start( + task, + "rejected-delivery", + _internal_implementation_fixture=self.implementation_fixture(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + status = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], status["version"], "fixture-host", + ) + packet = claimed["handoff"]["packet"] + with self.assertRaisesRegex(ContractError, "blocked by required evidence"): + service.handoff_complete( + started["run_id"], + claimed["claim"], + self.decision(packet, "blocked-accept", "accept", "Accept anyway."), + ) + completed = service.handoff_complete( + started["run_id"], + claimed["claim"], + self.decision(packet, "reject-delivery", "reject", "Required check failed."), + ) + self.assertEqual(completed["state"], "failed") + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "REVIEW_REJECTED") + self.assertEqual(receipt["evaluation"]["required_failures"], ["fixture-check"]) + self.assert_source_unchanged() + + def test_revision_returns_to_implementer_and_replaces_candidate_evidence(self): + task = self.delivery_task() + task["checks"][0] = { + **task["checks"][0], + "argv": [ + "python3", "-c", + "from pathlib import Path; " + "assert Path('src/app.py').read_text() == \"VALUE = 'fixed'\\n\"", + ], + } + service = Service(self.runtime) + started = service.start( + task, + "repaired-delivery", + _internal_implementation_fixture=self.repair_fixtures(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + first_wait = self.wait_for_handoff(service, started["run_id"]) + first_claim = service.handoff_claim( + started["run_id"], first_wait["version"], "fixture-host-one", + ) + first_packet = first_claim["handoff"]["packet"] + self.assertFalse(first_packet["evaluation"]["accept_allowed"]) + revised = service.handoff_complete( + started["run_id"], first_claim["claim"], + self.decision( + first_packet, "repair-delivery", "revise", "Fix the required value.", + ), + ) + self.assertEqual(revised["continuation"]["action"], "requeued") + self.assertTrue(revised["launched"]) + self.wait_for_candidate(service, started["run_id"]) + + store = Store(service.database, service.artifacts) + try: + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + self.assertEqual(len(snapshot["delivery_iterations"]), 2) + candidates = [ + item["candidate"] for item in snapshot["delivery_iterations"] + ] + self.assertNotEqual( + candidates[0]["candidate_sha256"], candidates[1]["candidate_sha256"], + ) + self.assertNotEqual(candidates[0]["commit_oid"], candidates[1]["commit_oid"]) + self.assertEqual( + candidates[1]["baseline_oid"], candidates[0]["baseline_oid"], + ) + self.assertEqual( + snapshot["delivery_iterations"][1]["revision_request"][ + "previous_candidate_sha256" + ], + candidates[0]["candidate_sha256"], + ) + self.assertEqual( + [item["role"] for item in store.attempts_for_run(started["run_id"])], + ["implementer", "reviewer", "implementer"], + ) + stale_evidence = { + field: first_packet[field] + for field in ( + "schema_version", "workflow", "candidate_sha256", "base_oid", + "target_oid", "review", "checks", "evaluation", "attempt", + ) + } + with self.assertRaisesRegex( + ContractError, "changes frozen candidate_sha256", + ): + validate_branch_review_evidence(stale_evidence, snapshot) + finally: + store.close() + + service.resume(started["run_id"]) + second_wait = self.wait_for_handoff(service, started["run_id"]) + second_claim = service.handoff_claim( + started["run_id"], second_wait["version"], "fixture-host-two", + ) + second_packet = second_claim["handoff"]["packet"] + self.assertNotEqual( + first_packet["candidate_sha256"], second_packet["candidate_sha256"], + ) + self.assertTrue(second_packet["evaluation"]["accept_allowed"]) + completed = service.handoff_complete( + started["run_id"], second_claim["claim"], + self.decision( + second_packet, "accept-repair", "accept", "Repair accepted.", + ), + ) + self.assertEqual(completed["state"], "succeeded") + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual( + [item["role"] for item in receipt["attempts"]], + ["implementer", "implementer", "reviewer", "reviewer"], + ) + self.assertEqual( + [item["disposition"] for item in receipt["dispositions"]], + ["revise", "accept"], + ) + self.assertEqual(receipt["revisions"]["executed"], 1) + with self.assertRaises(ConflictError): + service.handoff_complete( + started["run_id"], first_claim["claim"], + self.decision( + first_packet, "stale-reject", "reject", "Stale evidence.", + ), + ) + self.assert_source_unchanged() + + def test_delivery_revision_and_invocation_budgets_fail_before_new_writer(self): + for suffix, max_revisions, max_invocations in ( + ("revision", 0, 5), + ("invocation", 1, 3), + ): + with self.subTest(budget=suffix): + task = self.delivery_task() + task["budget"]["max_revisions"] = max_revisions + task["budget"]["max_worker_invocations"] = max_invocations + service = Service(self.runtime / suffix) + started = service.start( + task, + f"exhausted-{suffix}", + _internal_implementation_fixture=self.implementation_fixture(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + waiting = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], waiting["version"], f"host-{suffix}", + ) + packet = claimed["handoff"]["packet"] + completed = service.handoff_complete( + started["run_id"], claimed["claim"], + self.decision( + packet, f"revise-{suffix}", "revise", "Request repair.", + ), + ) + self.assertEqual(completed["state"], "failed") + self.assertFalse(completed["launched"]) + store = Store(service.database, service.artifacts) + try: + self.assertEqual( + [item["role"] for item in store.attempts_for_run( + started["run_id"] + )], + ["implementer", "reviewer"], + ) + finally: + store.close() + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "BUDGET_EXHAUSTED") + self.assertEqual(receipt["lead"]["disposition"], "revise") + + def test_headless_delivery_lead_accepts_the_candidate(self): + task = self.delivery_task() + task["lead"] = {"mode": "headless"} + service = Service(self.runtime) + started = service.start( + task, + "headless-delivery", + _internal_implementation_fixture=self.implementation_fixture(), + _internal_review_fixture=self.clean_review_fixture(), + _internal_lead_fixture={ + "disposition": "accept", + "reason": "All required evidence passes.", + }, + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + terminal = self.wait_for_terminal(service, started["run_id"]) + self.assertEqual(terminal["state"], "succeeded") + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["lead"]["mode"], "headless") + self.assertEqual(receipt["lead"]["disposition"], "accept") + self.assertEqual( + [item["role"] for item in receipt["attempts"]], + ["implementer", "reviewer"], + ) + self.assertEqual(len(receipt["lead"]["attempts"]), 1) + self.assertEqual(receipt["accounting"]["worker_invocations"], 3) + + def test_rate_limited_implementer_uses_frozen_same_permission_fallback(self): + task = self.delivery_task() + task["budget"]["max_fallbacks_per_step"] = 1 + fixture = { + **self.implementation_fixture(), + "fail_profile_ids": ["fixture-implementer"], + } + service = Service(self.runtime) + started = service.start( + task, + "implementation-fallback", + _internal_implementation_fixture=fixture, + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + waiting = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], waiting["version"], "fallback-host", + ) + packet = claimed["handoff"]["packet"] + service.handoff_complete( + started["run_id"], claimed["claim"], + self.decision(packet, "accept-fallback", "accept", "Fallback accepted."), + ) + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + implementers = [ + item for item in receipt["attempts"] if item["role"] == "implementer" + ] + self.assertEqual(len(implementers), 2) + self.assertEqual(implementers[0]["status"], "failed") + self.assertEqual(implementers[0]["error"]["error"], "RATE_LIMITED") + self.assertEqual( + [item["selected_profile"]["profile_id"] for item in implementers], + ["fixture-implementer", "fixture-implementer-fallback"], + ) + self.assertEqual( + {item["selected_profile"]["profile"]["permission_policy"] + for item in implementers}, + {"workspace_write"}, + ) + + def test_cancelled_repair_retains_prior_candidate_attempts_and_disposition(self): + fixtures = self.repair_fixtures() + fixtures["iterations"][1]["delay_seconds"] = 5 + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "cancelled-repair", + _internal_implementation_fixture=fixtures, + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + waiting = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], waiting["version"], "cancel-repair-host", + ) + packet = claimed["handoff"]["packet"] + service.handoff_complete( + started["run_id"], claimed["claim"], + self.decision(packet, "cancel-repair", "revise", "Repair then cancel."), + ) + deadline = time.monotonic() + 10 + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + if status["state"] == "running": + break + time.sleep(0.02) + else: + self.fail("repair implementer never entered running state") + service.cancel(started["run_id"]) + terminal = self.wait_for_terminal(service, started["run_id"]) + self.assertEqual(terminal["state"], "cancelled") + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["workflow"], "issue-delivery") + self.assertEqual( + [item["disposition"] for item in receipt["dispositions"]], ["revise"], + ) + self.assertEqual( + [item["role"] for item in receipt["attempts"]], + ["implementer", "reviewer", "implementer"], + ) + self.assertEqual(receipt["attempts"][-1]["status"], "cancelled") + self.assertEqual(receipt["candidate"]["sha256"], packet["candidate_sha256"]) + + def test_killed_repair_supervisor_never_launches_a_duplicate_writer(self): + fixtures = self.repair_fixtures() + fixtures["iterations"][1]["delay_seconds"] = 3 + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "killed-repair-supervisor", + _internal_implementation_fixture=fixtures, + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + waiting = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], waiting["version"], "kill-repair-host", + ) + packet = claimed["handoff"]["packet"] + service.handoff_complete( + started["run_id"], claimed["claim"], + self.decision(packet, "kill-repair", "revise", "Repair candidate."), + ) + deadline = time.monotonic() + 10 + attempt = None + while time.monotonic() < deadline: + store = Store(service.database, service.artifacts) + try: + current = store.attempt(started["run_id"]) + if (current and current["status"] == "running" + and current["role"] == "implementer" + and Path(current["child_record"]).is_file()): + attempt = current + break + finally: + store.close() + time.sleep(0.02) + if attempt is None: + self.fail("repair writer did not publish its child identity") + # The attempt pid is the runner, not the still-live detached coordinator. + os.kill(attempt["pid"], signal.SIGKILL) + deadline = time.monotonic() + 5 + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + store = Store(service.database, service.artifacts) + try: + current = store.attempt(started["run_id"]) + for field in ("id", "attempt_token", "pid", "pgid", "process_start_id"): + self.assertEqual(current[field], attempt[field]) + finally: + store.close() + if (inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]) == "dead" + and status["state"] == "blocked" + and status["phase"] == "recovery_required" + and current["status"] == "ownership_ambiguous"): + break + time.sleep(0.02) + else: + self.fail("killed repair runner did not reach blocked ownership recovery") + self.assertFalse(Path(attempt["exit_record"]).exists()) + recovered = Service(self.runtime).resume( + started["run_id"], + {"attempt_id": attempt["id"], "disposition": "retain_ownership"}, + ) + self.assertFalse(recovered["launched"]) + self.assertEqual(recovered["disposition"], "retain_ownership") + store = Store(service.database, service.artifacts) + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual( + [item["role"] for item in attempts], + ["implementer", "reviewer", "implementer"], + ) + self.assertEqual(attempts[-1]["id"], attempt["id"]) + self.assertEqual(attempts[-1]["attempt_token"], attempt["attempt_token"]) + self.assertEqual(store.worker_invocations(started["run_id"]), 3) + self.assertEqual(store.run(started["run_id"])["state"], "blocked") + finally: + store.close() + cancelled = service.cancel(started["run_id"]) + self.assertEqual(cancelled["state"], "cancelled") + self.assert_source_unchanged() + + def test_live_implementer_cannot_be_resumed_into_a_second_writer(self): + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "one-writer-delivery", + _internal_implementation_fixture=self.implementation_fixture(0.5), + ) + deadline = time.monotonic() + 10 + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + if status["state"] == "running": + break + time.sleep(0.02) + else: + self.fail("implementer did not enter running state") + resumed = service.resume(started["run_id"]) + self.assertEqual(resumed["disposition"], "live") + self.assertFalse(resumed["launched"]) + self.wait_for_candidate(service, started["run_id"]) + store = Store(service.database, service.artifacts) + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual(len(attempts), 1) + self.assertEqual(store.worker_invocations(started["run_id"]), 1) + finally: + store.close() + self.assert_source_unchanged() + + def test_durable_out_of_scope_implementation_cannot_publish_a_candidate(self): + service = Service(self.runtime) + fixture = { + "writes": [{"path": "README.md", "content": "unauthorized\n"}], + "delay_seconds": 0, + } + started = service.start( + self.delivery_task(), + "out-of-scope-delivery", + _internal_implementation_fixture=fixture, + ) + deadline = time.monotonic() + 15 + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + if status["state"] == "failed": + break + time.sleep(0.05) + else: + self.fail("out-of-scope delivery did not fail") + result = service.result(started["run_id"]) + self.assertTrue(result["ready"]) + self.assertEqual(result["state"], "failed") + self.assertNotIn( + "candidate-1.json", {item["name"] for item in result["artifacts"]}, + ) + self.assertTrue({ + "receipt.json", "receipt.md", "events.jsonl", + "artifact-manifest.json", "result-receipt.json", + } <= {item["name"] for item in result["artifacts"]}) + self.assert_source_unchanged() + + def test_revision_resume_after_prelaunch_crash_creates_one_repair_writer(self): + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "revision-prelaunch-crash", + _internal_implementation_fixture=self.repair_fixtures(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + waiting = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], waiting["version"], "fixture-crash-host", + ) + packet = claimed["handoff"]["packet"] + original_spawn = service._spawn_daemon + service._spawn_daemon = lambda *args, **kwargs: 0 + try: + saved = service.handoff_complete( + started["run_id"], claimed["claim"], + self.decision( + packet, "repair-after-crash", "revise", "Repair candidate.", + ), + ) + finally: + service._spawn_daemon = original_spawn + self.assertTrue(saved["launched"]) + self.assertEqual(service.status(started["run_id"])["state"], "queued") + + resumed = Service(self.runtime).resume(started["run_id"]) + self.assertTrue(resumed["launched"]) + self.wait_for_candidate(service, started["run_id"]) + store = Store(service.database, service.artifacts) + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual( + [item["role"] for item in attempts], + ["implementer", "reviewer", "implementer"], + ) + self.assertEqual(len({item["id"] for item in attempts}), 3) + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + self.assertEqual(len(snapshot["delivery_iterations"]), 2) + finally: + store.close() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_diagnostics.py b/test/core/test_diagnostics.py new file mode 100644 index 0000000..9c01af3 --- /dev/null +++ b/test/core/test_diagnostics.py @@ -0,0 +1,911 @@ +import contextlib +import io +import json +import os +from pathlib import Path +import signal +import shutil +import sys +import tempfile +import time +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad import cli, diagnostics, mcp_server, probe_process +from devsquad.contracts import ContractError +from devsquad.supervisor import _live_group_exists, process_start_identity + + +class DoctorReadinessTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-doctor-") + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.home = self.root / "home" + self.home.mkdir() + self.project = self.root / "project" + self.project.mkdir() + self.core = self.root / "core" + self.log = self.root / "probes.jsonl" + self.manager = mock.Mock( + mcp_sdk_available=True, mcp_sdk_supported=True, + mcp_sdk_version="2.2.0", squad_executable=ROOT / "plugin/core/bin/squad", + launcher_error=None, + ) + self.manager.inspect.side_effect = lambda template: { + "id": template.id, "installed": True, "ready": True, "status": "matching", + } + self.core_patch = mock.patch.object(diagnostics, "CORE_ROOT", self.core) + self.core_patch.start() + self.addCleanup(self.core_patch.stop) + + def install(self, name, *, version=None, response=None, returncode=0, raw=None, delay=0, + version_exit_delay=0, auth_exit_delay=0): + versions = { + "codex": "codex-cli 0.159.2", "claude": "2.1.220 (Claude Code)", + "grok": "1.0.46", "antigravity": "1.2.14", + } + version = version or versions[name] + binary = self.root / (name + "-fixture") + configuration = { + "name": name, "version": version, "response": response, + "returncode": returncode, "raw": raw, "log": str(self.log), "delay": delay, + "version_exit_delay": version_exit_delay, "auth_exit_delay": auth_exit_delay, + } + binary.write_text("#!" + sys.executable + "\n" + """ +import json, os, sys, time +configuration = CONFIGURATION +def record(value): + with open(configuration['log'], 'a') as output: + output.write(json.dumps(value) + '\\n') +record({'argv': sys.argv[1:], 'environment': dict(os.environ)}) +if sys.argv[1:] == ['--version']: + print(configuration['version'], flush=True) + if configuration['version_exit_delay']: + os.close(1) + time.sleep(configuration['version_exit_delay']) + os._exit(0) +elif sys.argv[1:] == ['auth', 'status', '--json']: + time.sleep(configuration['delay']) + print(configuration['raw'] if configuration['raw'] is not None else json.dumps(configuration['response']), flush=True) + print('private-stderr-token', file=sys.stderr) + if configuration['auth_exit_delay']: + os.close(1) + time.sleep(configuration['auth_exit_delay']) + os._exit(configuration['returncode']) + sys.exit(configuration['returncode']) +elif sys.argv[1:] == ['app-server', '--listen', 'stdio://']: + for line in sys.stdin: + message = json.loads(line) + record({'request': message}) + if message['method'] == 'initialize': + print(json.dumps({'id': message['id'], 'result': {}}), flush=True) + elif message['method'] == 'account/read': + time.sleep(configuration['delay']) + if configuration['raw'] is not None: + print(configuration['raw'], flush=True) + else: + print(json.dumps({'id': message['id'], 'result': configuration['response']}), flush=True) +else: + sys.exit(99) +""".replace("CONFIGURATION", repr(configuration))) + binary.chmod(0o755) + directory = self.core / "adapters" / name + directory.mkdir(parents=True, exist_ok=True) + (directory / "adapter.json").write_text(json.dumps({ + "schema_version": 1, "name": name, + "transport": "native_protocol" if name == "codex" else "cli_exec", + "binary_candidates": [str(binary)], "capabilities": {}, + "verified_harness_versions": ( + [versions[name]] if name in {"codex", "claude"} else [] + ), + })) + return binary + + def report(self): + return diagnostics.build_doctor_report( + project=self.project, home=self.home, manager=self.manager, + ) + + def row(self, report, name): + return next(row for row in report["adapters"] if row["adapter"] == name) + + def probes(self): + return [json.loads(line) for line in self.log.read_text().splitlines()] + + def codex(self, response=None): + return self.install("codex", response=response or { + "account": {"type": "chatgpt", "email": "private@example.test", + "accountId": "private-account", "accessToken": "private-token"}, + "requiresOpenaiAuth": True, + }) + + def test_logged_out_claude_is_visible_and_delivery_is_unavailable(self): + self.codex() + self.install("claude", response={"loggedIn": False, "authMethod": "none"}, returncode=1) + report = self.report() + claude = self.row(report, "claude") + self.assertTrue(claude["installed"]) + self.assertTrue(claude["supported"]) + self.assertIs(claude["authenticated"], False) + self.assertFalse(claude["ready"]) + self.assertEqual(claude["authentication"]["next_action"], "claude auth login") + self.assertTrue(report["supported_workflows"]["branch-review"]["ready"]) + self.assertFalse(report["supported_workflows"]["issue-delivery"]["ready"]) + self.assertTrue(report["ready"]) + + def test_unverified_cli_never_passes_adapter_readiness_or_auth_probe(self): + self.install("codex", version="codex-cli 99.0.0") + report = self.report() + row = self.row(report, "codex") + self.assertEqual(row["status"], "unverified") + self.assertTrue(row["installed"]) + self.assertFalse(row["supported"]) + self.assertIsNone(row["authenticated"]) + self.assertFalse(report["adapter_ready"]) + self.assertFalse(report["ready"]) + self.assertEqual([row["argv"] for row in self.probes()], [["--version"]]) + + def test_unsupported_authentication_is_unknown_even_with_four_registrations(self): + self.install("grok") + self.install("antigravity") + report = self.report() + self.assertEqual(len(report["local_apps"]), 4) + self.assertTrue(report["local_app_access"]["ready"]) + self.assertTrue(all(row["registered"] for row in report["local_apps"])) + self.assertTrue(all(row["operation_verified"] is None for row in report["local_apps"])) + self.assertFalse(report["ready"]) + for row in report["adapters"]: + self.assertIsNone(row["authenticated"]) + self.assertIsNone(row["operation_verified"]) + self.assertFalse(row["supported"]) + self.assertTrue(all(probe["argv"] == ["--version"] for probe in self.probes())) + + def test_authentication_and_registration_do_not_invent_operation_proof(self): + self.codex() + self.install("claude", response={ + "loggedIn": True, "authMethod": "claude.ai", "email": "private@example.test", + "organizationId": "private-org", "apiKey": "private-key", + }) + report = self.report() + self.assertTrue(report["ready"]) + self.assertTrue(report["supported_workflows"]["issue-delivery"]["ready"]) + self.assertFalse(report["supported_workflows"]["council"]["ready"]) + self.assertTrue(report["supported_workflows"]["council"]["implemented_partial"]) + self.assertFalse(report["supported_workflows"]["council"]["native_ready"]) + self.assertEqual(report["supported_workflows"]["council"]["reason"], "native_network_attestation_unavailable") + for row in report["adapters"]: + self.assertIs(row["authenticated"], True) + self.assertIsNone(row["operation_verified"]) + output = json.dumps(report) + for secret in ("private@example.test", "private-account", "private-token", + "private-org", "private-key", "private-stderr-token"): + self.assertNotIn(secret, output) + + def test_claude_malformed_banner_missing_or_non_boolean_auth_stays_unknown(self): + cases = [ + "Welcome private-token\n" + json.dumps({"loggedIn": True, "authMethod": "claude.ai"}), + "not-json private-token", "[]", "{}", + json.dumps({"loggedIn": "true", "authMethod": "claude.ai"}), + json.dumps({"loggedIn": 1, "authMethod": "claude.ai"}), + json.dumps({"loggedIn": True, "authMethod": "private-token"}), + json.dumps({"loggedIn": True, "authMethod": None}), + json.dumps({"loggedIn": True, "authMethod": []}), + json.dumps({"loggedIn": True, "authMethod": "none"}), + '{"loggedIn":false,"loggedIn":true,"authMethod":"claude.ai"}', + ] + for raw in cases: + with self.subTest(raw=raw): + self.install("claude", raw=raw) + report = self.report() + self.assertIsNone(self.row(report, "claude")["authenticated"]) + self.assertFalse(report["ready"]) + self.assertNotIn("private-token", json.dumps(report)) + + def test_api_key_auth_does_not_make_subscription_workflows_available(self): + self.install("claude", response={"loggedIn": True, "authMethod": "api_key"}) + self.codex({"account": {"type": "apiKey", "apiKey": "private-key"}, + "requiresOpenaiAuth": True}) + report = self.report() + self.assertFalse(report["ready"]) + self.assertFalse(report["adapter_ready"]) + for row in report["adapters"]: + self.assertTrue(row["authenticated"]) + self.assertFalse(row["authentication"]["subscription_supported"]) + self.assertFalse(row["ready"]) + self.assertNotIn("private-key", json.dumps(report)) + + def test_codex_null_account_and_unknown_provider_state_are_not_ready(self): + for response, authenticated in ( + ({"account": None, "requiresOpenaiAuth": True}, False), + ({"account": None, "requiresOpenaiAuth": False}, None), + ({"account": {"type": "unknown", "token": "private-token"}, "requiresOpenaiAuth": True}, None), + ({"account": {"type": []}, "requiresOpenaiAuth": True}, None), + ({"account": {"type": "chatgpt"}, "requiresOpenaiAuth": "true"}, None), + ({"requiresOpenaiAuth": True}, None), + ): + with self.subTest(response=response): + self.install("codex", response=response) + report = self.report() + self.assertIs(self.row(report, "codex")["authenticated"], authenticated) + self.assertFalse(report["ready"]) + self.assertNotIn("private-token", json.dumps(report)) + + def test_codex_protocol_is_nongenerating_and_environment_drops_credentials(self): + self.codex() + with mock.patch.dict(os.environ, { + "USER": "fixture-user", "OPENAI_API_KEY": "private-key", + "ANTHROPIC_API_KEY": "private-key", "CODEX_HOME": "/private/override", + "CLAUDE_CODE_OAUTH_TOKEN": "private-token", "PRIVATE_PROVIDER_OVERRIDE": "secret", + }): + report = self.report() + self.assertTrue(report["ready"]) + probes = self.probes() + requests = [probe["request"] for probe in probes if "request" in probe] + self.assertEqual([request["method"] for request in requests], + ["initialize", "initialized", "account/read"]) + self.assertEqual(requests[-1]["params"], {"refreshToken": False}) + for probe in (probe for probe in probes if "environment" in probe): + environment = probe["environment"] + self.assertEqual(environment["HOME"], str(self.home)) + self.assertEqual(environment["USER"], "fixture-user") + self.assertIn("PATH", environment) + # Python and macOS may add these locale values at process startup. + self.assertLessEqual(set(environment), { + "HOME", "USER", "PATH", "LC_CTYPE", "__CF_USER_TEXT_ENCODING", + }) + + def test_oversized_output_and_version_banners_are_redacted_before_parse(self): + for name in ("claude", "codex"): + with self.subTest(name=name): + self.install(name, raw="private-token" + ("x" * diagnostics.MAX_PROBE_BYTES)) + report = self.report() + self.assertIsNone(self.row(report, name)["authenticated"]) + self.assertNotIn("private-token", json.dumps(report)) + for name in ("claude", "codex"): + with self.subTest(name=name, depth="excessive"): + self.install(name, raw="[" * 1500 + "0" + "]" * 1500) + self.assertIsNone(self.row(self.report(), name)["authenticated"]) + self.install("claude", version="login failed private-token") + report = self.report() + self.assertIsNone(self.row(report, "claude")["version"]) + self.assertNotIn("private-token", json.dumps(report)) + + def test_codex_malformed_banner_error_or_uncorrelated_reply_stays_unknown(self): + for raw in ( + "Welcome private-token", "[]", "{}", + json.dumps({"id": 2, "result": None}), + json.dumps({"id": 99, "result": {"account": {"type": "chatgpt"}, "requiresOpenaiAuth": True}}), + json.dumps({"id": 2, "error": {"message": "private-token"}}), + ): + with self.subTest(raw=raw): + self.install("codex", raw=raw) + report = self.report() + self.assertIsNone(self.row(report, "codex")["authenticated"]) + self.assertFalse(report["ready"]) + self.assertNotIn("private-token", json.dumps(report)) + + def test_auth_checks_time_out_and_reap_the_owned_process(self): + start_process = diagnostics.subprocess.Popen + for name in ("claude", "codex"): + with self.subTest(name=name): + self.install(name, delay=60) + processes = [] + def capture_process(*args, **kwargs): + process = start_process(*args, **kwargs) + processes.append(process) + return process + with ( + mock.patch.object(diagnostics, "PROBE_TIMEOUT_SECONDS", 0.5), + mock.patch.object(diagnostics, "AUTH_TIMEOUT_SECONDS", 0.5), + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=capture_process) as spawned, + ): + report = self.report() + self.assertIsNone(self.row(report, name)["authenticated"]) + self.assertFalse(self.row(report, name)["ready"]) + for call in spawned.call_args_list: + self.assertEqual(set(call.kwargs["env"]), {"HOME", "USER", "PATH"}) + for process in processes: + self.assertIsNotNone(process.poll()) + self.assertTrue(process.stdout.closed) + + def test_exited_probe_parent_cannot_leave_term_ignoring_descendant_alive(self): + child_record = self.root / "probe-child.json" + code = """ +import json, os, signal, sys, time +from pathlib import Path +sys.path.insert(0, SOURCE) +from devsquad.supervisor import process_start_identity +read_ready, write_ready = os.pipe() +if os.fork() == 0: + os.close(read_ready) + signal.signal(signal.SIGTERM, signal.SIG_IGN) + Path(RECORD).write_text(json.dumps({'pid': os.getpid(), 'start': process_start_identity(os.getpid())})) + os.write(write_ready, b'r') + os.close(write_ready) + time.sleep(60) + os._exit(0) +os.close(write_ready) +os.read(read_ready, 1) +os.close(read_ready) +print('codex-cli 0.159.2', flush=True) +os._exit(0) +""".replace("SOURCE", repr(str(ROOT / "plugin/core/src"))).replace("RECORD", repr(str(child_record))) + processes = [] + start_process = diagnostics.subprocess.Popen + def capture_process(*args, **kwargs): + process = start_process(*args, **kwargs) + processes.append(process) + return process + started = time.monotonic() + try: + with ( + mock.patch.object(diagnostics, "PROBE_TIMEOUT_SECONDS", 0.5), + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=capture_process), + self.assertRaises(TimeoutError), + ): + diagnostics._probe_output( + [sys.executable, "-B", "-c", code], project=self.project, + environment=diagnostics._environment(self.home), + ) + self.assertLess(time.monotonic() - started, 3) + self.assertTrue(child_record.exists(), "descendant readiness barrier was not reached") + process = processes[0] + self.assertEqual(process.returncode, 0) + self.assertFalse(_live_group_exists(process.pid), "owned descendant survived probe cleanup") + self.assertTrue(process.stdout.closed) + finally: + if processes: + process = processes[0] + child = json.loads(child_record.read_text()) if child_record.exists() else None + if child and process_start_identity(child["pid"]) == child["start"]: + try: + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + process.wait(timeout=2) + deadline = time.monotonic() + 2 + while _live_group_exists(process.pid) and time.monotonic() < deadline: + time.sleep(0.02) + self.assertFalse(_live_group_exists(process.pid), "controlled fixture cleanup failed") + + def test_reused_probe_identity_is_not_safe_to_signal(self): + process = mock.Mock(pid=987654, returncode=0, stderr=None) + with ( + mock.patch.object(probe_process, "_probe_group_exists", return_value=True), + mock.patch.object(probe_process, "process_start_identity", return_value="new-start"), + mock.patch.object(probe_process.os, "killpg") as signal_group, + self.assertRaisesRegex(ContractError, "identity changed"), + ): + probe_process.close_probe(process, start_identity="original-start") + signal_group.assert_not_called() + process.stdout.close.assert_called_once() + + def test_missing_capture_and_observation_never_authorize_group_signals(self): + process = mock.Mock(pid=987654, returncode=None, stderr=None) + with ( + mock.patch.object(probe_process, "PROBE_CLEANUP_SECONDS", 0.05), + mock.patch.object(probe_process, "PROBE_TERM_GRACE_SECONDS", 0.01), + mock.patch.object(probe_process, "_probe_group_exists", return_value=True), + mock.patch.object(probe_process, "process_start_identity", return_value=None), + mock.patch.object(probe_process.os, "killpg") as signal_group, + self.assertRaises(ContractError), + ): + probe_process.close_probe(process, start_identity=None) + signal_group.assert_not_called() + process.wait.assert_not_called() + process.stdout.close.assert_called_once() + + def test_known_capture_missing_observation_has_no_unproven_group_authority(self): + for returncode in (None, 0): + with self.subTest(returncode=returncode): + process = mock.Mock(pid=987654, returncode=returncode, stderr=None) + with ( + mock.patch.object(probe_process, "_probe_group_exists", return_value=True), + mock.patch.object(probe_process, "process_start_identity", return_value=None), + mock.patch.object(probe_process.os, "killpg") as signal_group, + mock.patch.object(probe_process.os, "kill") as signal_child, + self.assertRaises(ContractError), + ): + probe_process.close_probe(process, start_identity="captured-start") + signal_group.assert_not_called() + signal_child.assert_not_called() + process.wait.assert_not_called() + process.stdout.close.assert_called_once() + + def test_version_eof_preserves_delayed_natural_success_and_retained_anchor(self): + binary = self.install("codex", version_exit_delay=0.15) + processes = [] + real_spawn, real_close = diagnostics.subprocess.Popen, diagnostics._close_probe + def spawn(*args, **kwargs): + process = real_spawn(*args, **kwargs) + if process.args == [str(binary), "--version"]: + processes.append(process) + return process + def close(process, **kwargs): + if process in processes: + self.assertIsNone(process.returncode, "natural-exit observation reaped the ownership anchor") + real_close(process, **kwargs) + started = time.monotonic() + with ( + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=spawn), + mock.patch.object(diagnostics, "_close_probe", side_effect=close), + ): + code, output = diagnostics._probe_output([str(binary), "--version"], + project=self.project, environment=diagnostics._environment(self.home)) + self.assertEqual((code, output.strip()), (0, "codex-cli 0.159.2")) + self.assertGreaterEqual(time.monotonic() - started, 0.15) + self.assertEqual(processes[0].returncode, 0) + self.assertFalse(_live_group_exists(processes[0].pid)) + self.assertTrue(processes[0].stdout.closed) + + def test_claude_auth_eof_preserves_delayed_natural_logged_in_and_logged_out_codes(self): + for logged_in, returncode in ((True, 0), (False, 1)): + with self.subTest(logged_in=logged_in): + method = "claude.ai" if logged_in else "none" + binary = self.install("claude", response={"loggedIn": logged_in, "authMethod": method}, + returncode=returncode, auth_exit_delay=0.15) + processes = [] + real_spawn, real_close = diagnostics.subprocess.Popen, diagnostics._close_probe + def spawn(*args, **kwargs): + process = real_spawn(*args, **kwargs) + if process.args == [str(binary), "auth", "status", "--json"]: + processes.append(process) + return process + def close(process, **kwargs): + if process in processes: + self.assertIsNone(process.returncode, "natural-exit observation reaped the ownership anchor") + real_close(process, **kwargs) + with ( + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=spawn), + mock.patch.object(diagnostics, "_close_probe", side_effect=close), + ): + report = self.report() + row = self.row(report, "claude") + self.assertTrue(row["supported"]) + self.assertIs(row["authenticated"], logged_in) + self.assertEqual(row["ready"], logged_in) + self.assertEqual(processes[0].returncode, returncode) + self.assertFalse(_live_group_exists(processes[0].pid)) + self.assertTrue(processes[0].stdout.closed) + + def test_hung_after_auth_eof_times_out_without_false_success_and_cleans_child(self): + binary = self.install("claude", response={"loggedIn": True, "authMethod": "claude.ai"}, + auth_exit_delay=60) + processes = [] + real_spawn = diagnostics.subprocess.Popen + def spawn(*args, **kwargs): + process = real_spawn(*args, **kwargs) + if process.args == [str(binary), "auth", "status", "--json"]: + processes.append(process) + return process + started = time.monotonic() + with ( + mock.patch.object(diagnostics, "PROBE_TIMEOUT_SECONDS", 0.2), + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=spawn), + ): + authentication = diagnostics._claude_auth(str(binary), project=self.project, + environment=diagnostics._environment(self.home)) + self.assertIsNone(authentication["authenticated"]) + self.assertFalse(authentication["subscription_supported"]) + self.assertGreaterEqual(time.monotonic() - started, 0.2) + self.assertLess(time.monotonic() - started, 2.5) + self.assertIsNotNone(processes[0].returncode) + self.assertFalse(_live_group_exists(processes[0].pid)) + self.assertTrue(processes[0].stdout.closed) + + def test_controlled_python_import_does_not_write_payload_bytecode_with_minimal_env(self): + source = self.root / "private-source" + shutil.copytree(ROOT / "plugin/core/src/devsquad", source / "devsquad", + ignore=shutil.ignore_patterns("__pycache__", "*.pyc")) + before = {str(path.relative_to(source)): path.read_bytes() + for path in source.rglob("*") if path.is_file()} + code = ( + "import os,sys\n" + "assert sys.dont_write_bytecode\n" + "assert 'PYTHONDONTWRITEBYTECODE' not in os.environ\n" + f"sys.path.insert(0,{str(source)!r})\n" + "from devsquad.supervisor import process_start_identity\n" + "assert process_start_identity(os.getpid()) is not None\n" + "print('codex-cli 0.159.2',flush=True)\n" + ) + environment = diagnostics._environment(self.home) + self.assertEqual(set(environment), {"HOME", "USER", "PATH"}) + result = diagnostics._probe_output([sys.executable, "-B", "-c", code], + project=self.project, environment=environment) + self.assertEqual(result, (0, "codex-cli 0.159.2\n")) + after = {str(path.relative_to(source)): path.read_bytes() + for path in source.rglob("*") if path.is_file()} + self.assertEqual(before, after) + self.assertFalse(list(source.rglob("__pycache__"))) + self.assertFalse(list(source.rglob("*.pyc"))) + + def test_probe_does_not_read_output_without_captured_identity(self): + process = mock.Mock(pid=987654, returncode=None, stderr=None) + process.wait.return_value = 0 + with ( + mock.patch.object(diagnostics.subprocess, "Popen", return_value=process), + mock.patch.object(diagnostics, "capture_probe_identity", return_value=None), + mock.patch.object(diagnostics, "_close_probe") as close, + mock.patch.object(diagnostics.selectors, "DefaultSelector") as selector, + mock.patch.object(diagnostics.os, "read", return_value=b"") as read, + self.assertRaisesRegex(ContractError, "ownership is unavailable"), + ): + diagnostics._probe_output(["/fixture/claude", "auth", "status", "--json"], + project=self.project, environment=diagnostics._environment(self.home)) + selector.assert_not_called() + read.assert_not_called() + process.wait.assert_not_called() + close.assert_called_once_with(process, start_identity=None) + + def test_codex_auth_does_not_construct_or_send_protocol_without_identity(self): + process = mock.Mock(pid=987654, returncode=None, stderr=None) + with ( + mock.patch.object(diagnostics.subprocess, "Popen", return_value=process), + mock.patch.object(diagnostics, "capture_probe_identity", return_value=None), + mock.patch.object(diagnostics, "_close_probe") as close, + mock.patch.object(diagnostics, "JsonLinePeer") as peer, + mock.patch.object(diagnostics, "receive_response", side_effect=[ + {"result": {}}, {"result": {"account": {"type": "chatgpt"}, "requiresOpenaiAuth": True}}, + ]) as receive, + ): + authentication = diagnostics._codex_auth("/fixture/codex", project=self.project, + environment=diagnostics._environment(self.home)) + self.assertIsNone(authentication["authenticated"]) + self.assertFalse(authentication["subscription_supported"]) + peer.assert_not_called() + receive.assert_not_called() + close.assert_called_once_with(process, start_identity=None) + + def test_unidentified_live_child_is_reaped_without_group_signals(self): + ready = self.root / "unidentified-ready" + code = ( + "import signal,time\nfrom pathlib import Path\n" + "signal.signal(signal.SIGTERM, signal.SIG_IGN)\n" + f"Path({str(ready)!r}).touch()\n" + "time.sleep(60)\n" + ) + process = diagnostics.subprocess.Popen( + [sys.executable, "-B", "-c", code], stdin=diagnostics.subprocess.DEVNULL, + stdout=diagnostics.subprocess.PIPE, stderr=diagnostics.subprocess.DEVNULL, + start_new_session=True, + ) + try: + deadline = time.monotonic() + 2 + while not ready.exists() and time.monotonic() < deadline: + time.sleep(0.01) + self.assertTrue(ready.exists()) + with ( + mock.patch.object(probe_process, "process_start_identity", return_value=None), + mock.patch.object(probe_process.os, "killpg", wraps=os.killpg) as signal_group, + ): + probe_process.close_probe(process, start_identity=None) + signal_group.assert_not_called() + self.assertIsNotNone(process.returncode) + self.assertFalse(_live_group_exists(process.pid)) + self.assertTrue(process.stdout.closed) + finally: + if process.poll() is None: + process.kill() + process.wait(timeout=2) + process.stdout.close() + + def test_missing_kernel_child_authority_denies_all_signals(self): + process = diagnostics.subprocess.Popen( + [sys.executable, "-B", "-c", "import time; time.sleep(60)"], + stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, + stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, + ) + try: + with ( + mock.patch.object(probe_process.os, "waitpid", side_effect=ChildProcessError), + mock.patch.object(probe_process.os, "killpg") as signal_group, + mock.patch.object(probe_process.os, "kill") as signal_child, + self.assertRaisesRegex(ContractError, "ownership is unavailable"), + ): + probe_process.close_probe(process, start_identity=None) + signal_group.assert_not_called() + signal_child.assert_not_called() + self.assertIsNone(process.returncode) + self.assertTrue(process.stdout.closed) + finally: + process.kill() + process.wait(timeout=2) + process.stdout.close() + self.assertFalse(_live_group_exists(process.pid)) + + def test_known_capture_without_waitid_falls_back_to_verified_child_only(self): + process = diagnostics.subprocess.Popen( + [sys.executable, "-B", "-c", "import time; time.sleep(60)"], + stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, + stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, + ) + captured = process_start_identity(process.pid) + self.assertIsNotNone(captured) + try: + with ( + mock.patch.object(probe_process.os, "waitid", None, create=True), + mock.patch.object(probe_process, "process_start_identity", return_value=None), + mock.patch.object(probe_process.os, "killpg", wraps=os.killpg) as signal_group, + ): + probe_process.close_probe(process, start_identity=captured) + signal_group.assert_not_called() + self.assertIsNotNone(process.returncode) + self.assertFalse(_live_group_exists(process.pid)) + self.assertTrue(process.stdout.closed) + finally: + if process.poll() is None: + process.kill() + process.wait(timeout=2) + process.stdout.close() + + def test_ps_zombie_anchor_rejects_reparented_wrong_group_and_malformed_rows(self): + process = diagnostics.subprocess.Popen( + [sys.executable, "-B", "-c", "import time; time.sleep(60)"], + stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, + stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, + ) + pid, parent = process.pid, os.getpid() + cases = [ + (0, f"{pid} {parent + 1} {pid} Z\n"), + (0, f"{pid} {parent} {pid + 1} Z\n"), + (0, f"{pid + 1} {parent} {pid} Z\n"), + (0, f"{pid} {parent} {pid} S\n"), + (0, f"{pid} {parent} {pid}\n"), + (0, f"{pid} {parent} {pid} Z\n{pid} {parent} {pid} Z\n"), + (0, "private-banner-token\n"), + (0, "x" * 1025), + (1, f"{pid} {parent} {pid} Z\n"), + ] + try: + for code, output in cases: + with self.subTest(code=code, output=output[:100]): + with ( + mock.patch.object(probe_process.os, "waitid", None, create=True), + mock.patch.object(probe_process.subprocess, "run", return_value=mock.Mock(returncode=code, stdout=output)) as inventory, + mock.patch.object(probe_process.os, "killpg") as signal_group, + ): + self.assertFalse(probe_process._retained_child_anchor(process, deadline=time.monotonic() + 1)) + signal_group.assert_not_called() + self.assertLessEqual(inventory.call_args.kwargs["timeout"], 0.5) + self.assertEqual(set(inventory.call_args.kwargs["env"]), {"HOME", "USER", "PATH"}) + for error in (OSError, diagnostics.subprocess.TimeoutExpired("/bin/ps", 0.5)): + with self.subTest(error=type(error).__name__): + with ( + mock.patch.object(probe_process.os, "waitid", None, create=True), + mock.patch.object(probe_process.subprocess, "run", side_effect=error), + ): + self.assertFalse(probe_process._retained_child_anchor(process, deadline=time.monotonic() + 1)) + finally: + process.kill() + process.wait(timeout=2) + process.stdout.close() + + def test_cleanup_reap_lock_contention_cannot_exceed_deadline_or_signal(self): + process = diagnostics.subprocess.Popen( + [sys.executable, "-B", "-c", "import time; time.sleep(60)"], + stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, + stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, + ) + process._waitpid_lock.acquire() + started = time.monotonic() + try: + with ( + mock.patch.object(probe_process, "PROBE_CLEANUP_SECONDS", 0.05), + mock.patch.object(probe_process.os, "killpg") as signal_group, + mock.patch.object(probe_process.os, "kill") as signal_child, + self.assertRaisesRegex(ContractError, "deadline expired"), + ): + probe_process.close_probe(process, start_identity=None) + signal_group.assert_not_called() + signal_child.assert_not_called() + self.assertLess(time.monotonic() - started, 0.5) + self.assertTrue(process.stdout.closed) + finally: + process._waitpid_lock.release() + process.kill() + process.wait(timeout=2) + process.stdout.close() + + def test_probe_eof_retains_exited_parent_until_owned_descendant_cleanup(self): + record = self.root / "eof-child.json" + code = ( + "import json,os,signal,time\nfrom pathlib import Path\n" + f"import sys\nsys.path.insert(0, {str(ROOT / 'plugin/core/src')!r})\n" + "from devsquad.supervisor import process_start_identity\n" + "read_ready,write_ready=os.pipe()\n" + "if os.fork()==0:\n" + " os.close(read_ready)\n" + " os.dup2(os.open(os.devnull,os.O_WRONLY),1)\n" + " signal.signal(signal.SIGTERM, signal.SIG_IGN)\n" + f" Path({str(record)!r}).write_text(json.dumps({{'pid':os.getpid(),'start':process_start_identity(os.getpid())}}))\n" + " os.write(write_ready,b'r')\n os.close(write_ready)\n time.sleep(60)\n os._exit(0)\n" + "os.close(write_ready)\nos.read(read_ready,1)\nos.close(read_ready)\n" + "print('codex-cli 0.159.2',flush=True)\nos._exit(0)\n" + ) + captured_processes = [] + real_spawn, real_close = diagnostics.subprocess.Popen, diagnostics._close_probe + def spawn(*args, **kwargs): + process = real_spawn(*args, **kwargs) + captured_processes.append(process) + return process + def close(process, **kwargs): + self.assertIsNone(process.returncode, "probe reaped its ownership anchor before cleanup") + real_close(process, **kwargs) + try: + with ( + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=spawn), + mock.patch.object(diagnostics, "_close_probe", side_effect=close), + ): + code, output = diagnostics._probe_output([sys.executable, "-B", "-c", code], + project=self.project, environment=diagnostics._environment(self.home)) + self.assertEqual((code, output.strip()), (0, "codex-cli 0.159.2")) + self.assertFalse(_live_group_exists(captured_processes[0].pid)) + finally: + child = json.loads(record.read_text()) if record.exists() else None + if child and process_start_identity(child["pid"]) == child["start"]: + try: + os.kill(child["pid"], signal.SIGKILL) + except ProcessLookupError: + pass + if captured_processes: + process = captured_processes[0] + if process.poll() is None: + process.kill() + process.wait(timeout=2) + process.stdout.close() + + def test_unidentified_descendant_cleanup_is_unconfirmed_without_group_signals(self): + record = self.root / "unidentified-child.json" + code = ( + "import json,os,signal,time\nfrom pathlib import Path\n" + f"import sys\nsys.path.insert(0, {str(ROOT / 'plugin/core/src')!r})\n" + "from devsquad.supervisor import process_start_identity\n" + "read_ready,write_ready=os.pipe()\n" + "if os.fork()==0:\n" + " os.close(read_ready)\n" + " signal.signal(signal.SIGTERM, signal.SIG_IGN)\n" + f" Path({str(record)!r}).write_text(json.dumps({{'pid':os.getpid(),'start':process_start_identity(os.getpid())}}))\n" + " os.write(write_ready,b'r')\n os.close(write_ready)\n time.sleep(60)\n os._exit(0)\n" + "os.close(write_ready)\nos.read(read_ready,1)\nos.close(read_ready)\ntime.sleep(60)\n" + ) + process = diagnostics.subprocess.Popen( + [sys.executable, "-B", "-c", code], stdin=diagnostics.subprocess.DEVNULL, + stdout=diagnostics.subprocess.PIPE, stderr=diagnostics.subprocess.DEVNULL, + start_new_session=True, + ) + try: + deadline = time.monotonic() + 2 + while not record.exists() and time.monotonic() < deadline: + time.sleep(0.01) + self.assertTrue(record.exists()) + with ( + mock.patch.object(probe_process, "process_start_identity", return_value=None), + mock.patch.object(probe_process.os, "killpg", wraps=os.killpg) as signal_group, + self.assertRaisesRegex(ContractError, "cleanup unconfirmed"), + ): + probe_process.close_probe(process, start_identity=None) + signal_group.assert_not_called() + self.assertIsNotNone(process.returncode) + self.assertTrue(_live_group_exists(process.pid)) + self.assertTrue(process.stdout.closed) + finally: + child = json.loads(record.read_text()) if record.exists() else None + if child and process_start_identity(child["pid"]) == child["start"]: + try: + os.kill(child["pid"], signal.SIGKILL) + except ProcessLookupError: + pass + if process.poll() is None: + process.kill() + process.wait(timeout=2) + process.stdout.close() + deadline = time.monotonic() + 2 + while _live_group_exists(process.pid) and time.monotonic() < deadline: + time.sleep(0.02) + self.assertFalse(_live_group_exists(process.pid)) + + def test_doctor_unidentified_auth_is_unknown_and_every_owned_parent_is_absent(self): + self.codex() + self.install("claude", response={"loggedIn": True, "authMethod": "claude.ai"}) + processes = [] + real_spawn, real_capture = diagnostics.subprocess.Popen, diagnostics.capture_probe_identity + def spawn(*args, **kwargs): + process = real_spawn(*args, **kwargs) + processes.append(process) + return process + def capture(process): + return real_capture(process) if process.args[-1] == "--version" else None + with ( + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=spawn), + mock.patch.object(diagnostics, "capture_probe_identity", side_effect=capture), + mock.patch.object(diagnostics, "JsonLinePeer") as peer, + ): + report = self.report() + peer.assert_not_called() + self.assertFalse(report["ready"]) + self.assertFalse(report["adapter_ready"]) + for name in ("codex", "claude"): + row = self.row(report, name) + self.assertTrue(row["supported"]) + self.assertIsNone(row["authenticated"]) + self.assertFalse(row["ready"]) + for process in processes: + self.assertIsNotNone(process.returncode) + self.assertFalse(_live_group_exists(process.pid)) + if process.stdout is not None: + self.assertTrue(process.stdout.closed) + + def test_missing_group_is_confirmed_before_reap_without_signaling_stale_pid(self): + process = mock.Mock(pid=987654, returncode=0, stderr=None) + with ( + mock.patch.object(probe_process, "_probe_group_exists", return_value=False), + mock.patch.object(probe_process, "process_start_identity", return_value="new-start"), + mock.patch.object(probe_process.os, "killpg") as signal_group, + ): + probe_process.close_probe(process, start_identity="original-start") + signal_group.assert_not_called() + process.wait.assert_called_once() + + def test_identity_change_after_term_blocks_kill_escalation(self): + process = mock.Mock(pid=987654, returncode=None, stderr=None) + identity = {"value": "original-start"} + def record_signal(pgid, value): + identity["value"] = "reused-start" + with ( + mock.patch.object(probe_process, "_probe_group_exists", return_value=True), + mock.patch.object(probe_process, "process_start_identity", side_effect=lambda pid: identity["value"]), + mock.patch.object(probe_process.os, "killpg", side_effect=record_signal) as signal_group, + self.assertRaisesRegex(ContractError, "identity changed"), + ): + probe_process.close_probe(process, start_identity="original-start") + signal_group.assert_called_once_with(process.pid, signal.SIGTERM) + + def test_missing_cli_never_claims_authentication_or_support(self): + self.codex().unlink() + report = self.report() + row = self.row(report, "codex") + self.assertEqual(row["status"], "unavailable") + self.assertFalse(row["installed"]) + self.assertFalse(row["supported"]) + self.assertIsNone(row["authenticated"]) + self.assertIsNone(row["operation_verified"]) + self.assertFalse(report["ready"]) + self.assertFalse(self.log.exists()) + + def test_cli_and_mcp_doctor_keep_the_same_machine_envelope(self): + self.codex() + report = self.report() + stdout = io.StringIO() + with mock.patch.object(cli, "build_doctor_report", return_value=report): + with contextlib.redirect_stdout(stdout): + code = cli.main(["doctor", "--json"]) + self.assertEqual(code, 0) + envelope = json.loads(stdout.getvalue()) + self.assertEqual(envelope["schema_version"], 1) + self.assertTrue(envelope["ok"]) + self.assertEqual(envelope["data"], report) + bridge = mcp_server.MCPBridge(self.root / "runtime", mock.Mock(), environment={}) + with mock.patch.object(mcp_server, "build_doctor_report", return_value=report): + self.assertEqual(bridge.doctor(), envelope) + + def test_read_only_report_keeps_config_and_ledger_unchanged(self): + self.codex() + configuration = self.home / ".claude.json" + configuration.write_text('{"secret":"private-token","keep":true}') + database = self.project / "state.sqlite" + database.write_bytes(b"unchanged-ledger") + before = {str(path): path.read_bytes() for directory in (self.home, self.project) + for path in directory.rglob("*") if path.is_file()} + self.report() + after = {str(path): path.read_bytes() for directory in (self.home, self.project) + for path in directory.rglob("*") if path.is_file()} + self.assertEqual(before, after) + self.manager.setup.assert_not_called() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_experiment_assignment_store.py b/test/core/test_experiment_assignment_store.py new file mode 100644 index 0000000..2d782ce --- /dev/null +++ b/test/core/test_experiment_assignment_store.py @@ -0,0 +1,197 @@ +"""Prelaunch assignment persistence, not a claim of public trial execution.""" + +import copy +import hashlib +import json +from pathlib import Path +import sqlite3 +import subprocess +import tempfile +import unittest + +from test_experiment_provenance import frozen_snapshot, spec_v2, digest, execution_digest +from test_learning import experiment +from test_lifecycle import profile +from devsquad.contracts import ContractError +from devsquad.experiment_provenance import assignment_for, paired_input_identity +from devsquad.store import ConflictError, Store, canonical_json, git_common_dir, SUPPORTED_SCHEMA_VERSION + + +class ExperimentAssignmentStoreTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix='devsquad-assignment-') + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name).resolve() + self.repo = self.root / 'repo' + subprocess.run(['git', 'init', '-q', str(self.repo)], check=True) + self.store = Store(self.root / 'state.sqlite3', self.root / 'artifacts') + self.addCleanup(self.store.close) + self.common = str(git_common_dir(self.repo)) + self.snapshot = frozen_snapshot() + self.snapshot['task']['project']['repo_path'] = str(self.repo) + self.spec = spec_v2() + self.spec['project_path'] = str(self.repo) + self.spec['cases'][0].update(paired_input_identity( + self.snapshot, role='reviewer', package_digest='e' * 64, + )) + + def snapshot_for(self, arm='control', *, spec=None): + spec = spec or self.spec + snapshot = copy.deepcopy(self.snapshot) + if arm == 'candidate': + candidate = profile('profile-b', 'model-b') + snapshot['routing']['roles']['reviewer']['selected'] = { + 'profile_id': candidate['id'], 'profile': candidate, + 'profile_sha256': digest(candidate), + } + snapshot['experiment_spec'] = copy.deepcopy(spec) + snapshot['experiment_assignment'] = assignment_for( + spec, 'eval-1', arm, project_common_dir=self.common, + ) + return snapshot + + def claim(self, key): + return self.store.claim_start(self.repo, key, {'task': self.snapshot['task']}, 'owner') + + def prepare(self, claim, snapshot): + return self.store.complete_preparation( + claim.run_id, claim.fencing_token, snapshot, package_digest='e' * 64, + ) + + def test_assignment_freezes_before_attempts_with_preparation_fence(self): + claim = self.claim('first') + snapshot = self.snapshot_for() + version = self.prepare(claim, snapshot) + row = self.store.connection.execute( + 'SELECT * FROM experiment_assignments WHERE run_id=?', (claim.run_id,), + ).fetchone() + self.assertEqual(json.loads(row['assignment_json']), snapshot['experiment_assignment']) + self.assertEqual(json.loads(row['snapshot_json']), snapshot) + self.assertEqual(row['assignment_sha256'], digest(snapshot['experiment_assignment'])) + self.assertEqual(row['preparation_fencing_token'], claim.fencing_token) + self.assertEqual(row['frozen_run_version'], version) + self.assertEqual(self.store.connection.execute( + 'SELECT COUNT(*) FROM attempts WHERE run_id=?', (claim.run_id,), + ).fetchone()[0], 0) + # Later mutable workflow snapshots cannot rewrite the original witness. + self.store.connection.execute('UPDATE runs SET mutable_snapshot=? WHERE id=?', + (canonical_json({'altered': True}), claim.run_id)) + self.assertEqual(self.store.connection.execute( + 'SELECT snapshot_json FROM experiment_assignments WHERE run_id=?', (claim.run_id,), + ).fetchone()[0], canonical_json(snapshot)) + + def test_distinct_arms_share_one_predeclared_spec(self): + for arm in ('control', 'candidate'): + self.prepare(self.claim(arm), self.snapshot_for(arm)) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_specs').fetchone()[0], 1) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 2) + + def test_invalid_assignment_rolls_back_spec_and_preparation_atomically(self): + for mutation in ('input', 'profile', 'project', 'missing_spec', 'digest'): + with self.subTest(mutation=mutation): + claim = self.claim(mutation) + snapshot = self.snapshot_for() + if mutation == 'input': + snapshot['task']['goal'] = 'A different goal.' + elif mutation == 'profile': + snapshot['experiment_assignment'] = assignment_for( + self.spec, 'eval-1', 'candidate', project_common_dir=self.common, + ) + elif mutation == 'project': + snapshot['experiment_assignment']['project_common_dir'] = '/tmp/foreign/.git' + elif mutation == 'missing_spec': + del snapshot['experiment_spec'] + else: + snapshot['experiment_assignment']['spec_sha256'] = '0' * 64 + with self.assertRaises(ContractError): + self.prepare(claim, snapshot) + self.assertEqual(self.store.run(claim.run_id)['phase'], 'preparing') + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_specs').fetchone()[0], 0) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 0) + + def test_duplicate_arm_cannot_acquire_a_second_run(self): + first = self.claim('first') + self.prepare(first, self.snapshot_for()) + second = self.claim('second') + with self.assertRaisesRegex(ConflictError, 'already assigned'): + self.prepare(second, self.snapshot_for()) + self.assertEqual(self.store.run(second.run_id)['phase'], 'preparing') + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 1) + + def test_changed_spec_and_stale_preparation_cannot_rebind(self): + first = self.claim('first') + self.prepare(first, self.snapshot_for()) + with self.assertRaisesRegex(ConflictError, 'stale'): + self.prepare(first, self.snapshot_for('candidate')) + changed = copy.deepcopy(self.spec) + changed['hypothesis'] = 'Post-hoc replacement.' + second = self.claim('candidate') + with self.assertRaisesRegex(ConflictError, 'frozen differently'): + self.prepare(second, self.snapshot_for('candidate', spec=changed)) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 1) + + def test_plain_preparation_creates_no_experiment_records(self): + claim = self.claim('plain') + self.prepare(claim, self.snapshot) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_specs').fetchone()[0], 0) + with self.assertRaisesRegex(ConflictError, 'stale'): + self.prepare(claim, self.snapshot_for()) + + def test_tested_native_version_and_binary_must_match_predeclared_execution(self): + snapshot = copy.deepcopy(self.snapshot) + selected = snapshot['routing']['roles']['reviewer']['selected'] + selected['profile']['harness'] = 'codex' + selected['profile_sha256'] = digest(selected['profile']) + adapter = {'harness': 'codex', 'harness_version': 'fixture-version-1', + 'transport': 'native_protocol', 'model_provider': 'openai', + 'binary_sha256': '1' * 64} + snapshot['review_adapters'] = {'profile-a': adapter} + spec = copy.deepcopy(self.spec) + spec['variable']['control_profile_sha256'] = selected['profile_sha256'] + spec['variable']['control_execution_sha256'] = execution_digest(selected['profile'], adapter) + snapshot['experiment_spec'] = spec + snapshot['experiment_assignment'] = assignment_for( + spec, 'eval-1', 'control', project_common_dir=self.common, + ) + for field, replacement in (('harness_version', 'fixture-version-2'), + ('binary_sha256', '2' * 64)): + changed = copy.deepcopy(snapshot) + changed['review_adapters']['profile-a'][field] = replacement + with self.subTest(field=field): + with self.assertRaisesRegex(ContractError, 'native execution'): + self.prepare(self.claim(field), changed) + # Correct explicit native context is accepted; no CLI is executed. + self.prepare(self.claim('native-context'), snapshot) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 1) + + def test_schema_13_migration_preserves_legacy_evaluation_bytes(self): + database = self.root / 'old.sqlite3' + connection = sqlite3.connect(database) + migrations = Path(__file__).resolve().parents[2] / 'plugin/core/src/devsquad/migrations' + for path in sorted(migrations.glob('*.sql')): + version = int(path.name.split('_', 1)[0]) + if version > 13: + continue + connection.executescript(path.read_text()) + connection.execute('INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)', (version, 'fixture')) + spec = canonical_json(experiment(self.repo)) + evaluation = canonical_json({'schema_version': 1, 'historical': True}) + connection.execute( + 'INSERT INTO experiments(experiment_id,project_path,spec_json,spec_sha256,evaluation_json,' + 'evaluation_sha256,verdict,recorded_at) VALUES(?,?,?,?,?,?,?,?)', + ('old', str(self.repo), spec, hashlib.sha256(spec.encode()).hexdigest(), + evaluation, hashlib.sha256(evaluation.encode()).hexdigest(), 'no_change', 'fixture'), + ) + connection.commit() + connection.close() + upgraded = Store(database, self.root / 'old-artifacts') + self.addCleanup(upgraded.close) + saved = upgraded.connection.execute('SELECT * FROM experiments').fetchone() + self.assertEqual(saved['spec_json'], spec) + self.assertEqual(saved['evaluation_json'], evaluation) + self.assertEqual(upgraded.connection.execute('SELECT MAX(version) FROM schema_migrations').fetchone()[0], SUPPORTED_SCHEMA_VERSION) + self.assertEqual(upgraded.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 0) + + +if __name__ == '__main__': + unittest.main() diff --git a/test/core/test_experiment_eligibility.py b/test/core/test_experiment_eligibility.py new file mode 100644 index 0000000..ecf40fa --- /dev/null +++ b/test/core/test_experiment_eligibility.py @@ -0,0 +1,411 @@ +"""New lifecycle authority is bound to current public saved-run evidence.""" + +import copy +from datetime import datetime, timedelta, timezone +import hashlib +import json +from pathlib import Path +import sqlite3 +import subprocess +import tempfile +import threading +import unittest +from unittest.mock import patch + +from experiment_runtime_fixture import ExperimentRuntimeFixture +import test_learning as learning_fixtures +from test_learning import experimental_final +import test_lifecycle as lifecycle_fixtures +from devsquad.catalog import update_last_good +from devsquad.contracts import ContractError +from devsquad.learning import evaluate_experiment +from devsquad.service import Service +from devsquad.store import ConflictError, Store, canonical_json, SUPPORTED_SCHEMA_VERSION + + +class ExperimentEligibilityTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-eligibility-") + self.addCleanup(self.temporary.cleanup) + self.fixture = ExperimentRuntimeFixture(Path(self.temporary.name)) + self.addCleanup(self.fixture.close) + self.fixture.run_all() + self.evaluation = self.fixture.service.policy_evaluate(self.fixture.spec) + self.store = self.fixture.store() + self.addCleanup(self.store.close) + self.template = lifecycle_fixtures.lifecycle_template() + self.store.bootstrap_profile_binding(self.template, self.fixture.profiles["control"], version=7) + helper = lifecycle_fixtures.ProfileLifecycleTest() + helper.candidate = self.fixture.profiles["candidate"] + self.qualification = helper.qualification(self.evaluation) + self.qualification["budget"].update(max_worker_invocations=4, worker_invocations=4) + + def correct(self, fixture=None, *, arm="candidate"): + fixture = fixture or self.fixture + correction = experimental_final(f"escaped-held-out-{fixture.spec['experiment_id']}-{arm}", "succeeded") + correction.update( + kind="late_correction", verdict="escaped_defect", + corrects_outcome_id=fixture.outcome_id("hold-1", arm), + observed_at=datetime.now(timezone.utc).isoformat(), + summary="A real later escaped-defect observation invalidates eligibility.", + ) + fixture.service.outcome_add(fixture.runs[("hold-1", arm)], correction) + + def test_unmeasured_ratios_cannot_bypass_finite_qualification_gates(self): + template = copy.deepcopy(self.template) + template["template_id"] = "finite-measurement-template" + template["gate"].update(max_latency_ratio=2.0, max_usage_ratio=2.0) + self.store.register_profile_template(template) + qualification = copy.deepcopy(self.qualification) + qualification["template_id"] = template["template_id"] + for fields in (("latency_ratio",), ("usage_ratio",), ("latency_ratio", "usage_ratio")): + with self.subTest(invented=fields): + changed = copy.deepcopy(qualification) + for field in fields: + changed["measured"][field] = 0.0 + with self.assertRaisesRegex(ContractError, "measurements|unmeasured"): + self.store.record_profile_qualification(changed) + with self.assertRaisesRegex(ContractError, "latency_ratio_missing.*usage_ratio_missing"): + self.store.record_profile_qualification(qualification) + + def test_live_outcomes_require_current_clock_not_module_import_time(self): + # A full suite may spend more than the permitted skew before this + # module's first lifecycle case. Keep the real clock fence strict. + with self.assertRaisesRegex(ContractError, "invalid_saved_run_evidence"): + self.store.record_profile_qualification( + self.qualification, now=datetime.now(timezone.utc) - timedelta(minutes=10), + ) + self.assertEqual(self.store.record_profile_qualification(self.qualification)["gate_failures"], []) + + def next_fixture(self, experiment_id, *, profiles=None, candidate_succeeds=False): + fixture = ExperimentRuntimeFixture( + self.fixture.root / experiment_id, service=self.fixture.service, repo=self.fixture.repo, + experiment_id=experiment_id, profiles=profiles, candidate_succeeds=candidate_succeeds, + ) + self.addCleanup(fixture.close) + fixture.run_all() + return fixture, fixture.service.policy_evaluate(fixture.spec) + + @staticmethod + def rollback(evaluation, *, target="profile-a", target_version=7, expected=8): + return { + "schema_version": 1, "decision_id": "regression-rollback", + "action": "rollback", "alias": "review.deep", "expected_binding_version": expected, + "qualification_id": None, + "rollback_target": {"profile_id": target, "binding_version": target_version}, + "experiment_id": evaluation["experiment"]["experiment_id"], + "evaluation_sha256": evaluation["evaluation_sha256"], "actor": "human", + "reason": "The predeclared held-out regression requires rollback.", + "evidence_refs": ["evaluation.json"], + } + + def test_stale_regression_cannot_roll_back_until_explicit_current_review(self): + self.store.record_profile_qualification(self.qualification) + self.store.change_profile_binding(lifecycle_fixtures.ProfileLifecycleTest.promotion("initial-promotion")) + fixture, evaluation = self.next_fixture("post-promotion-regression") + self.assertEqual(evaluation["evaluation"]["verdict"], "no_change") + self.correct(fixture, arm="control") + with self.assertRaisesRegex(ContractError, "current|stale"): + self.store.change_profile_binding(self.rollback(evaluation)) + self.assertEqual(self.store.profile_binding("review.deep")["version"], 8) + reviewed = fixture.service.policy_evaluate( + fixture.spec, revision_id="regression-review", + previous_evaluation_sha256=evaluation["evaluation_sha256"], + ) + request = self.rollback(reviewed) + first = self.store.change_profile_binding(request) + self.assertEqual(first["receipt"]["to"]["binding_version"], 9) + self.assertEqual(self.store.change_profile_binding(request), {**first, "replayed": True}) + + def test_stale_qualified_target_blocks_regression_and_is_skipped_by_catalog_fallback(self): + self.store.record_profile_qualification(self.qualification) + self.store.change_profile_binding(lifecycle_fixtures.ProfileLifecycleTest.promotion("promote-b")) + profiles = { + "control": self.fixture.profiles["candidate"], + "candidate": lifecycle_fixtures.profile("profile-c", "model-c"), + } + _, evaluation = self.next_fixture("qualify-c", profiles=profiles, candidate_succeeds=True) + helper = lifecycle_fixtures.ProfileLifecycleTest() + helper.candidate = profiles["candidate"] + qualification = helper.qualification(evaluation, qualification_id="qualification-c") + self.store.record_profile_qualification(qualification) + request = lifecycle_fixtures.ProfileLifecycleTest.promotion("promote-c", expected=8) + request["qualification_id"] = "qualification-c" + self.store.change_profile_binding(request) + self.correct() + _, regression = self.next_fixture("c-regression", profiles=profiles) + with self.assertRaisesRegex(ContractError, "current|stale"): + self.store.change_profile_binding(self.rollback(regression, target="profile-b", target_version=8, expected=9)) + self.assertEqual(self.store.profile_binding("review.deep")["version"], 9) + all_profiles = [self.fixture.profiles["control"], *profiles.values()] + for available, should_block in ((["model-b"], True), (["model-a", "model-b"], False)): + catalog_path = self.fixture.root / ("blocked-catalog.json" if should_block else "safe-catalog.json") + update_last_good(catalog_path, harness="fixture", version="1", complete=True, + models=[{"id": value["model_id"]} for value in all_profiles], profiles=all_profiles) + change = update_last_good(catalog_path, harness="fixture", version="1", complete=True, + models=[{"id": value} for value in available], profiles=all_profiles)["catalog_change"] + fallback = { + "schema_version": 1, "decision_id": "catalog-blocked" if should_block else "catalog-safe", + "action": "rollback", "alias": "review.deep", "expected_binding_version": 9, + "catalog_change": change, "actor": "human", "reason": "The incumbent was removed.", + "evidence_refs": ["catalog.json"], + } + if should_block: + with self.assertRaisesRegex(ContractError, "no available qualified predecessor"): + self.store.fallback_unavailable_profile_binding(fallback) + self.assertEqual(self.store.profile_binding("review.deep")["version"], 9) + else: + result = self.store.fallback_unavailable_profile_binding(fallback) + self.assertEqual(result["receipt"]["to"]["profile_id"], "profile-a") + self.assertEqual(result["receipt"]["to"]["binding_version"], 10) + + def test_correction_after_evaluation_blocks_new_qualification_atomically(self): + self.correct() + with self.assertRaisesRegex(ContractError, "current|stale|changed"): + self.store.record_profile_qualification(self.qualification) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM qualification_runs").fetchone()[0], 0) + self.assertEqual(self.store.profile_binding("review.deep")["version"], 7) + + def test_correction_after_qualification_blocks_replay_and_new_promotion(self): + self.store.record_profile_qualification(self.qualification) + self.correct() + with self.assertRaisesRegex(ContractError, "current|stale|changed"): + self.store.record_profile_qualification(self.qualification) + with self.assertRaisesRegex(ContractError, "current|stale|changed"): + self.store.change_profile_binding(lifecycle_fixtures.ProfileLifecycleTest.promotion("stale-promotion")) + self.assertEqual(self.store.profile_binding("review.deep")["version"], 7) + self.assertEqual(self.store.profile_binding_decisions("review.deep"), []) + + def test_completed_decision_replay_preserves_historical_receipt_after_correction(self): + self.store.record_profile_qualification(self.qualification) + request = lifecycle_fixtures.ProfileLifecycleTest.promotion("completed-promotion") + first = self.store.change_profile_binding(request) + self.correct() + replay = self.store.change_profile_binding(request) + self.assertEqual(replay, {**first, "replayed": True}) + self.assertEqual(self.store.profile_binding("review.deep")["version"], 8) + self.assertEqual(len(self.store.profile_binding_decisions("review.deep")), 1) + + def test_correction_cannot_commit_between_qualification_validation_and_commit(self): + attempted, completed = threading.Event(), threading.Event() + failures = [] + original = self.store._qualification_evidence + + def correction(): + attempted.set() + try: + self.correct() + except Exception as exc: + failures.append(exc) + finally: + completed.set() + + worker = threading.Thread(target=correction) + + def checked(*args, **kwargs): + result = original(*args, **kwargs) + self.assertTrue(self.store.connection.in_transaction) + worker.start() + self.assertTrue(attempted.wait(2)) + # The other connection's write cannot commit while this validated + # qualification transaction still holds its SQLite writer fence. + self.assertFalse(completed.wait(0.1)) + return result + + try: + with patch.object(self.store, "_qualification_evidence", side_effect=checked): + qualification = self.store.record_profile_qualification(self.qualification) + self.assertTrue(qualification["eligibility"]["eligible"]) + finally: + if worker.ident is not None: + worker.join(10) + self.assertFalse(worker.is_alive()) + self.assertEqual(failures, []) + self.assertTrue(completed.is_set()) + with self.assertRaisesRegex(ContractError, "current|stale"): + self.store.record_profile_qualification(self.qualification) + with self.assertRaisesRegex(ContractError, "current|stale"): + self.store.change_profile_binding(lifecycle_fixtures.ProfileLifecycleTest.promotion("raced-promotion")) + self.assertEqual(self.store.profile_binding("review.deep")["version"], 7) + + def test_proposal_shows_stale_evidence_without_rewriting_saved_evaluation(self): + current = self.fixture.service.learning_propose(str(self.fixture.repo))["proposal"] + self.assertEqual(current["verdict"], "promotion_proposal") + self.correct() + stale = self.fixture.service.learning_propose(str(self.fixture.repo))["proposal"] + self.assertEqual(stale["verdict"], "no_change") + self.assertIn("saved_evaluation_stale_evidence_changed", stale["reasons"]) + self.assertEqual(stale["evidence"]["experiment"]["evaluation_sha256"], self.evaluation["evaluation_sha256"]) + self.assertFalse(stale["evidence"]["experiment"]["eligibility"]["eligible"]) + + def test_same_profile_id_with_different_fingerprint_cannot_qualify(self): + changed = copy.deepcopy(self.qualification) + changed["candidate_profile"]["required_tools"] = ["read", "web"] + with self.assertRaisesRegex(ContractError, "fingerprint|candidate"): + self.store.record_profile_qualification(changed) + + def test_task_class_claim_must_match_actual_frozen_tasks(self): + template = copy.deepcopy(self.template) + template["template_id"] = "template-multiple-classes" + template["allowed_task_classes"].append("untested-task-class") + self.store.register_profile_template(template) + changed = copy.deepcopy(self.qualification) + changed.update(template_id=template["template_id"], task_class="untested-task-class") + with self.assertRaisesRegex(ContractError, "task.class|context"): + self.store.record_profile_qualification(changed) + + def test_explicit_revision_keeps_original_bytes_and_reuses_original_assignments(self): + original = dict(self.store.connection.execute( + "SELECT * FROM experiments WHERE experiment_id=?", (self.fixture.spec["experiment_id"],), + ).fetchone()) + self.correct() + revision = self.fixture.service.policy_evaluate( + self.fixture.spec, revision_id="review-after-escape", + previous_evaluation_sha256=self.evaluation["evaluation_sha256"], + ) + self.assertEqual(revision["evaluation"]["verdict"], "no_change") + self.assertNotEqual(revision["evaluation_sha256"], self.evaluation["evaluation_sha256"]) + self.assertEqual(revision["revision_id"], "review-after-escape") + self.assertEqual(revision["previous_evaluation_sha256"], self.evaluation["evaluation_sha256"]) + self.assertTrue(revision["eligibility"]["eligible"]) + replay = self.fixture.service.policy_evaluate( + self.fixture.spec, revision_id="review-after-escape", + previous_evaluation_sha256=self.evaluation["evaluation_sha256"], + ) + self.assertEqual(replay, {**revision, "replayed": True}) + saved = dict(self.store.connection.execute( + "SELECT * FROM experiments WHERE experiment_id=?", (self.fixture.spec["experiment_id"],), + ).fetchone()) + self.assertEqual(saved, original) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM experiment_assignments").fetchone()[0], 4) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM experiment_evaluation_revisions").fetchone()[0], 1) + for arguments in ( + {"revision_id": "missing-predecessor"}, + {"previous_evaluation_sha256": revision["evaluation_sha256"]}, + {"revision_id": "bad-predecessor", "previous_evaluation_sha256": "not-a-hash"}, + {"revision_id": "review-after-escape", "previous_evaluation_sha256": "0" * 64}, + ): + with self.subTest(arguments=arguments), self.assertRaises(ContractError): + self.fixture.service.policy_evaluate(self.fixture.spec, **arguments) + self.store.connection.execute( + "UPDATE experiment_evaluation_revisions SET previous_evaluation_sha256=? WHERE revision_id=?", + ("0" * 64, "review-after-escape"), + ) + try: + with self.assertRaisesRegex(ContractError, "predecessor"): + self.fixture.service.policy_evaluate(self.fixture.spec) + finally: + self.store.connection.execute( + "UPDATE experiment_evaluation_revisions SET previous_evaluation_sha256=? WHERE revision_id=?", + (self.evaluation["evaluation_sha256"], "review-after-escape"), + ) + with self.assertRaises(ConflictError): + self.fixture.service.policy_evaluate( + self.fixture.spec, revision_id="stale-predecessor-review", + previous_evaluation_sha256=self.evaluation["evaluation_sha256"], + ) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM experiment_evaluation_revisions").fetchone()[0], 1) + + def test_fresh_reviewed_revision_can_qualify_and_pins_new_decision(self): + correction = experimental_final("reviewed-followup", "succeeded") + correction.update( + kind="late_correction", verdict="corrected", corrects_outcome_id="candidate-hold-1", + observed_at=datetime.now(timezone.utc).isoformat(), + summary="A bounded follow-up review recorded corrected evidence without an escaped defect.", + ) + self.fixture.service.outcome_add(self.fixture.runs[("hold-1", "candidate")], correction) + with self.assertRaisesRegex(ContractError, "current|stale"): + self.store.record_profile_qualification(self.qualification) + reviewed = self.fixture.service.policy_evaluate( + self.fixture.spec, revision_id="reviewed-followup", + previous_evaluation_sha256=self.evaluation["evaluation_sha256"], + ) + self.assertEqual(reviewed["evaluation"]["verdict"], "promotion_proposal") + qualification = copy.deepcopy(self.qualification) + qualification.update(qualification_id="qualification-reviewed", evaluation_sha256=reviewed["evaluation_sha256"]) + result = self.store.record_profile_qualification(qualification) + self.assertTrue(result["eligibility"]["eligible"]) + request = lifecycle_fixtures.ProfileLifecycleTest.promotion("reviewed-promotion") + request["qualification_id"] = qualification["qualification_id"] + promoted = self.store.change_profile_binding(request) + self.assertEqual(promoted["receipt"]["qualification"]["evaluation_sha256"], reviewed["evaluation_sha256"]) + self.assertEqual(promoted["receipt"]["to"]["binding_version"], 8) + + +class HistoricalExperimentEligibilityTest(unittest.TestCase): + def test_upgraded_reused_outcome_history_remains_readable_but_cannot_authorize_new_decisions(self): + temporary = tempfile.TemporaryDirectory(prefix="devsquad-historical-eligibility-") + self.addCleanup(temporary.cleanup) + root = Path(temporary.name).resolve() + repo = root / "repo" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + runtime = root / "runtime" + runtime.mkdir() + database = runtime / "state.sqlite3" + connection = sqlite3.connect(database) + migrations = Path(__file__).resolve().parents[2] / "plugin/core/src/devsquad/migrations" + for path in sorted(migrations.glob("*.sql")): + version = int(path.name.split("_", 1)[0]) + if version <= 13: + connection.executescript(path.read_text()) + connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?, 'historical')", (version,)) + spec = learning_fixtures.experiment(repo) + chains = {} + for case in spec["cases"]: + for arm, verdict in (("control", "failed"), ("candidate", "succeeded")): + chains[case[f"{arm}_outcome_id"]] = { + "final": experimental_final(case[f"{arm}_outcome_id"], verdict), "late_corrections": [], + } + evaluation = evaluate_experiment(spec, chains, evaluated_at=learning_fixtures.NOW.isoformat()) + # Explicit historical negative: the old bug could count two reused + # outcomes as three pairs. Never fabricate new successful runtime runs. + for case in spec["cases"]: + case.update(control_outcome_id="reused-control", candidate_outcome_id="reused-candidate") + spec_json = canonical_json(spec) + spec_sha256 = hashlib.sha256(spec_json.encode()).hexdigest() + evaluation["spec_sha256"] = spec_sha256 + evaluation_json = canonical_json(evaluation) + evaluation_sha256 = hashlib.sha256(evaluation_json.encode()).hexdigest() + connection.execute( + "INSERT INTO experiments(experiment_id,project_path,spec_json,spec_sha256,evaluation_json,evaluation_sha256,verdict,recorded_at) VALUES(?,?,?,?,?,?,?,?)", + (spec["experiment_id"], str(repo), spec_json, spec_sha256, evaluation_json, evaluation_sha256, + evaluation["verdict"], evaluation["evaluated_at"]), + ) + connection.commit() + connection.close() + service = Service(runtime) + store = Store(database, runtime / "artifacts") + self.addCleanup(store.close) + original = dict(store.connection.execute("SELECT * FROM experiments").fetchone()) + self.assertEqual(store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], SUPPORTED_SCHEMA_VERSION) + report = service.learning_report(str(repo)) + self.assertEqual(report["sample_size"], 0) + proposal = service.learning_propose(str(repo))["proposal"] + self.assertEqual(proposal["verdict"], "no_change") + self.assertIn("legacy_unverified_evidence", proposal["reasons"]) + self.assertFalse(proposal["decision"]["review_required"]) + self.assertEqual(proposal["evidence"]["experiment"]["evaluation_sha256"], evaluation_sha256) + self.assertEqual(dict(store.connection.execute("SELECT * FROM experiments").fetchone()), original) + helper = lifecycle_fixtures.ProfileLifecycleTest() + helper.candidate = lifecycle_fixtures.profile("profile-b", "model-b") + qualification = helper.qualification({"experiment": spec, "evaluation_sha256": evaluation_sha256}) + store.bootstrap_profile_binding(lifecycle_fixtures.lifecycle_template(), lifecycle_fixtures.profile("profile-a", "model-a"), version=7) + with self.assertRaisesRegex(ContractError, "legacy_unverified_evidence"): + store.record_profile_qualification(qualification) + # Import an old qualification as historical test data, then prove its + # old qualified label cannot bypass the repaired new-decision gate. + store._insert_concrete_profile(helper.candidate, evaluation["evaluated_at"]) + payload = canonical_json(qualification) + store.connection.execute( + "INSERT INTO qualification_runs(qualification_id,alias,template_id,profile_id,experiment_id,evaluation_sha256,verdict,payload_json,payload_sha256,gate_failures_json,recorded_at) VALUES(?,?,?,?,?,?,?,?,?,?,?)", + (qualification["qualification_id"], qualification["alias"], qualification["template_id"], helper.candidate["id"], + spec["experiment_id"], evaluation_sha256, "qualified", payload, hashlib.sha256(payload.encode()).hexdigest(), "[]", evaluation["evaluated_at"]), + ) + with self.assertRaisesRegex(ContractError, "legacy_unverified_evidence"): + store.change_profile_binding(lifecycle_fixtures.ProfileLifecycleTest.promotion("unsafe-history-promotion")) + self.assertEqual(store.profile_binding("review.deep")["version"], 7) + self.assertEqual(store.profile_binding_decisions("review.deep"), []) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_experiment_evidence_integrity.py b/test/core/test_experiment_evidence_integrity.py new file mode 100644 index 0000000..4ca4558 --- /dev/null +++ b/test/core/test_experiment_evidence_integrity.py @@ -0,0 +1,180 @@ +"""Corrupt only negative copies of otherwise publicly executed arm evidence.""" + +import copy +import hashlib +import json +from pathlib import Path +import tempfile +import unittest +from unittest.mock import patch + +from experiment_runtime_fixture import ExperimentRuntimeFixture +from devsquad.contracts import ContractError +from devsquad.experiment_evidence import read_experiment_chains +from devsquad.experiment_provenance import assignment_for +from devsquad.store import canonical_json +from test_experiment_provenance import digest + + +class ExperimentEvidenceIntegrityTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-evidence-integrity-") + self.addCleanup(self.temporary.cleanup) + self.fixture = ExperimentRuntimeFixture(Path(self.temporary.name)) + self.addCleanup(self.fixture.close) + self.run_id = self.fixture.run_arm("eval-1", "candidate") + self.store = self.fixture.store() + self.addCleanup(self.store.close) + + def assert_rejected_without_persistence(self): + with self.assertRaises(ContractError): + self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM experiments").fetchone()[0], 0) + + def test_fence_event_and_outcome_corruptions_cannot_be_hidden_by_missing_partner(self): + assignment = dict(self.store.connection.execute( + "SELECT * FROM experiment_assignments WHERE run_id=?", (self.run_id,), + ).fetchone()) + run = self.store.run(self.run_id) + outcome = dict(self.store.connection.execute( + "SELECT * FROM outcomes WHERE run_id=?", (self.run_id,), + ).fetchone()) + queued = dict(self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? AND run_version=?", + (self.run_id, assignment["frozen_run_version"]), + ).fetchone()) + claimed = dict(self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? AND type='supervisor.claimed'", (self.run_id,), + ).fetchone()) + running = dict(self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? AND type='run.running'", (self.run_id,), + ).fetchone()) + changed_launch = json.loads(running["payload"]) + changed_launch["pid"] += 1 + mutations = [ + ("fence_version", "experiment_assignments", "run_id", self.run_id, + "frozen_run_version", 1, assignment["frozen_run_version"]), + ("fence_token", "experiment_assignments", "run_id", self.run_id, + "preparation_fencing_token", 999, assignment["preparation_fencing_token"]), + ("assignment_hash", "experiment_assignments", "run_id", self.run_id, + "assignment_sha256", "0" * 64, assignment["assignment_sha256"]), + ("queued_event", "events", "id", queued["id"], "type", "ignored", queued["type"]), + ("claim_event", "events", "id", claimed["id"], "type", "ignored", claimed["type"]), + ("launch_event", "events", "id", running["id"], "payload", canonical_json(changed_launch), running["payload"]), + ("final_hash", "outcomes", "id", outcome["id"], "payload_sha256", "0" * 64, outcome["payload_sha256"]), + ("final_row_verdict", "outcomes", "id", outcome["id"], "verdict", "failed", outcome["verdict"]), + ("run_package", "runs", "id", self.run_id, "package_digest", "0" * 64, run["package_digest"]), + ("run_terminal", "runs", "id", self.run_id, "state", "failed", run["state"]), + ] + for name, table, key, identity, column, changed, original in mutations: + with self.subTest(mutation=name): + statement = f"UPDATE {table} SET {column}=? WHERE {key}=?" + self.store.connection.execute(statement, (changed, identity)) + try: + self.assert_rejected_without_persistence() + finally: + self.store.connection.execute(statement, (original, identity)) + self.assertEqual(self.fixture.service.policy_evaluate(self.fixture.spec)["evaluation"]["verdict"], "no_change") + + def test_saved_output_bytes_must_match_the_durable_capture(self): + artifact = self.store.connection.execute( + "SELECT f.path FROM attempts a JOIN artifacts f ON f.id=a.stdout_artifact_id WHERE a.run_id=?", + (self.run_id,), + ).fetchone() + path = Path(artifact["path"]) + original = path.read_bytes() + try: + path.write_bytes(original + b"corrupt") + self.assert_rejected_without_persistence() + finally: + path.write_bytes(original) + + def test_imported_profile_drift_is_rejected_even_with_a_valid_artifact_hash(self): + row = dict(self.store.connection.execute( + "SELECT * FROM artifacts WHERE run_id=? AND name LIKE 'review-attempt-%.json'", + (self.run_id,), + ).fetchone()) + path = Path(row["path"]) + original = path.read_bytes() + changed = json.loads(original) + selected = changed["selected_profile"] + selected["profile"]["model_id"] = "a-different-executed-model" + selected["profile_sha256"] = digest(selected["profile"]) + content = (canonical_json(changed) + "\n").encode() + # Simulate internally hash-consistent imported identity drift. A file + # checksum alone cannot bind its selected profile to the original arm. + try: + path.write_bytes(content) + self.store.connection.execute( + "UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (hashlib.sha256(content).hexdigest(), len(content), row["id"]), + ) + self.assert_rejected_without_persistence() + finally: + path.write_bytes(original) + self.store.connection.execute( + "UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (row["sha256"], row["byte_size"], row["id"]), + ) + + def test_trial_reservation_rejects_changed_inputs_before_any_worker(self): + resolve = self.fixture.service._resolve_snapshot + + def assigned(*args, **kwargs): + snapshot = resolve(*args, **kwargs) + snapshot["experiment_spec"] = copy.deepcopy(self.fixture.spec) + snapshot["experiment_assignment"] = assignment_for( + self.fixture.spec, "hold-1", "control", project_common_dir=self.fixture.common, + ) + return snapshot + + with (patch.object(self.fixture.service, "_resolve_snapshot", side_effect=assigned), + patch.object(self.fixture.service, "_spawn_daemon", return_value=0)): + started = self.fixture.service.start( + self.fixture.task("hold-1", "control"), "reserved-negative", + _internal_review_fixture={"verdict": "clean", "summary": "Fixture.", "findings": []}, + ) + run_id = started["run_id"] + self.fixture.runs[("hold-1", "control")] = run_id + self.assertEqual(started["state"], "queued") + run = self.store.run(run_id) + original = json.loads(run["mutable_snapshot"]) + selected = original["routing"]["roles"]["reviewer"]["selected"] + changed_profile = copy.deepcopy(original) + slot = changed_profile["routing"]["roles"]["reviewer"]["selected"] + slot["profile"]["model_id"] = "changed-after-preparation" + slot["profile_sha256"] = digest(slot["profile"]) + changed_task = copy.deepcopy(original) + changed_task["task"]["goal"] = "A different task after preparation." + for changed in (changed_profile, changed_task, []): + with self.subTest(snapshot_type=type(changed).__name__): + self.store.connection.execute( + "UPDATE runs SET mutable_snapshot=? WHERE id=?", (canonical_json(changed), run_id), + ) + try: + with self.assertRaisesRegex(ContractError, "experiment launch"): + self.store.reserve_attempt( + run_id, run["version"], "negative-worker", run["package_digest"], "reviewer", + account_pool_id=selected["profile"]["account_pool_id"], + profile_id=selected["profile_id"], profile_index=0, + ) + self.assertEqual(self.store.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=?", (run_id,), + ).fetchone()[0], 0) + finally: + self.store.connection.execute( + "UPDATE runs SET mutable_snapshot=? WHERE id=?", (run["mutable_snapshot"], run_id), + ) + + def test_reader_requires_one_consistent_transaction(self): + self.assertFalse(self.store.connection.in_transaction) + with self.assertRaisesRegex(ContractError, "consistent ledger transaction"): + read_experiment_chains( + self.store.connection, spec=self.fixture.spec, + project_id=self.store.run(self.run_id)["project_id"], + project_common_dir=self.fixture.common, evaluated_at="2026-10-01T00:00:00Z", + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_experiment_provenance.py b/test/core/test_experiment_provenance.py new file mode 100644 index 0000000..b9777c6 --- /dev/null +++ b/test/core/test_experiment_provenance.py @@ -0,0 +1,261 @@ +"""Versioned experiment assignment and paired-input identity contracts.""" + +import copy +import hashlib +import importlib +from pathlib import Path +import unittest + +from test_learning import NOW, experiment, experimental_final +from test_lifecycle import profile, review_task +from devsquad.contracts import ContractError +from devsquad.learning import evaluate_experiment, validate_experiment +from devsquad.store import canonical_json + + +def digest(value): + return hashlib.sha256(canonical_json(value).encode()).hexdigest() + + +def execution_digest(selected, adapter=None): + return digest({'profile_sha256': digest(selected), + 'adapter': adapter if adapter is not None else {'harness': 'fixture'}}) + + +def spec_v2(): + spec = experiment(Path('/tmp/provenance-project')) + spec['schema_version'] = 2 + spec['variable'].update({ + 'role': 'reviewer', + 'control_profile_sha256': digest(profile('profile-a', 'model-a')), + 'candidate_profile_sha256': digest(profile('profile-b', 'model-b')), + 'control_execution_sha256': execution_digest(profile('profile-a', 'model-a')), + 'candidate_execution_sha256': execution_digest(profile('profile-b', 'model-b')), + }) + for case in spec['cases']: + case['input_sha256'] = digest(['input', case['case_id']]) + case['case_sha256'] = digest(['case', case['case_id']]) + return spec + + +def frozen_snapshot(): + selected = profile('profile-a', 'model-a') + return { + 'task': review_task(Path('/tmp/provenance-project')), + 'base_oid': 'a' * 40, + 'target_oid': 'b' * 40, + 'workspace': { + 'candidate_sha256': 'c' * 64, 'path': '/tmp/run-a/review', + 'base_oid': 'a' * 40, 'target_oid': 'b' * 40, + }, + 'configs': {'policy_file': {'path': '/tmp/a/policy.json', 'sha256': 'd' * 64}}, + 'routing': { + 'policy': {'sha256': 'd' * 64}, + 'roles': {'reviewer': { + 'selected': {'profile_id': selected['id'], 'profile': selected, + 'profile_sha256': digest(selected)}, + 'fallbacks': [], 'fallback_mode': 'none', + }}, + 'capacity': {}, + }, + } + + +class ExperimentProvenanceContractTest(unittest.TestCase): + def api(self): + return importlib.import_module('devsquad.experiment_provenance') + + def test_v2_spec_retains_concrete_profiles_and_both_input_identities(self): + spec = spec_v2() + self.assertEqual(validate_experiment(spec), spec) + self.assertEqual(spec['variable']['role'], 'reviewer') + + def test_v2_spec_requires_role_fingerprints_and_exact_case_fields(self): + changes = [ + ('variable', 'role', 'lead'), + ('variable', 'role', []), + ('variable', 'control_profile_sha256', 'short'), + ('variable', 'candidate_profile_sha256', True), + ('variable', 'control_execution_sha256', None), + ('variable', 'candidate_execution_sha256', 'short'), + ('case', 'case_sha256', None), + ('case', 'input_sha256', 'A' * 64), + ] + for target, field, value in changes: + with self.subTest(field=field, value=value): + spec = spec_v2() + row = spec['variable'] if target == 'variable' else spec['cases'][0] + row[field] = value + with self.assertRaises(ContractError): + validate_experiment(spec) + for field in ('role', 'control_profile_sha256', 'candidate_profile_sha256', + 'control_execution_sha256', 'candidate_execution_sha256'): + spec = spec_v2() + del spec['variable'][field] + with self.assertRaises(ContractError): + validate_experiment(spec) + + def test_v2_cannot_evaluate_unbound_outcome_labels_as_provenance(self): + spec = spec_v2() + chains = {} + for case in spec['cases']: + for arm, verdict in (('control', 'failed'), ('candidate', 'succeeded')): + outcome_id = case[f'{arm}_outcome_id'] + chains[outcome_id] = { + 'final': experimental_final(outcome_id, verdict), 'late_corrections': [], + } + with self.assertRaisesRegex(ContractError, 'provenance'): + evaluate_experiment(spec, chains, evaluated_at=NOW.isoformat()) + + def test_same_corpus_case_cannot_be_relabelled_held_out_by_context_change(self): + spec = spec_v2() + spec['cases'][2]['case_sha256'] = spec['cases'][0]['case_sha256'] + self.assertNotEqual(spec['cases'][2]['input_sha256'], spec['cases'][0]['input_sha256']) + with self.assertRaisesRegex(ContractError, 'case.*unique'): + validate_experiment(spec) + + def test_assignment_is_exactly_bound_to_spec_case_split_arm_and_project(self): + api = self.api() + spec = spec_v2() + assignment = api.assignment_for( + spec, 'eval-1', 'candidate', project_common_dir='/tmp/provenance-project/.git', + ) + self.assertEqual(assignment['spec_sha256'], digest(spec)) + self.assertEqual(assignment['profile_sha256'], spec['variable']['candidate_profile_sha256']) + self.assertEqual(api.validate_assignment( + assignment, spec=spec, project_common_dir='/tmp/provenance-project/.git', + ), assignment) + for field, replacement in ( + ('schema_version', True), ('spec_sha256', '0' * 64), + ('experiment_id', 'foreign-experiment'), ('split', 'held_out'), + ('arm', 'control'), ('role', 'implementer'), + ('profile_id', 'profile-a'), ('profile_sha256', '0' * 64), + ('execution_sha256', '0' * 64), + ('input_sha256', '0' * 64), ('case_sha256', '0' * 64), + ('project_common_dir', '/tmp/foreign/.git'), ('extra', 1), + ): + with self.subTest(field=field): + invalid = {**assignment, field: replacement} + with self.assertRaises(ContractError): + api.validate_assignment(invalid, spec=spec, + project_common_dir='/tmp/provenance-project/.git') + + def test_legacy_specs_and_unknown_case_or_arm_cannot_make_assignments(self): + api = self.api() + for spec, case_id, arm in ( + (experiment(Path('/tmp/provenance-project')), 'eval-1', 'candidate'), + (spec_v2(), 'missing', 'candidate'), + (spec_v2(), 'eval-1', []), + ): + with self.subTest(case=case_id, arm=arm): + with self.assertRaises(ContractError): + api.assignment_for(spec, case_id, arm, + project_common_dir='/tmp/provenance-project/.git') + + def test_incidental_paths_observations_and_tested_profile_do_not_change_inputs(self): + api = self.api() + before = frozen_snapshot() + changed = copy.deepcopy(before) + changed['task']['project']['repo_path'] = '/tmp/other-worktree' + changed['task']['project']['base_ref'] = 'renamed-base' + changed['task']['origin'] = {'surface': 'another-host', 'session_ref': 'session-b'} + changed['workspace']['path'] = '/tmp/run-b/review' + changed['configs']['policy_file']['path'] = '/tmp/b/policy.json' + changed['routing']['capacity'] = {'irrelevant': 'runtime observation'} + candidate = profile('profile-b', 'model-b') + changed['routing']['roles']['reviewer']['selected'] = { + 'profile_id': candidate['id'], 'profile': candidate, + 'profile_sha256': digest(candidate), + } + # Assignment is deliberately excluded: its spec hash includes input hashes. + changed['experiment_assignment'] = {'not': 'fingerprint input'} + self.assertEqual(api.paired_input_identity(before, role='reviewer', package_digest='e' * 64), + api.paired_input_identity(changed, role='reviewer', package_digest='e' * 64)) + + def test_runtime_policy_and_peer_changes_affect_pair_not_corpus_identity(self): + api = self.api() + before = frozen_snapshot() + base = api.paired_input_identity(before, role='reviewer', package_digest='e' * 64) + for kind in ('runtime', 'policy', 'peer'): + with self.subTest(kind=kind): + changed = copy.deepcopy(before) + package = 'e' * 64 + if kind == 'runtime': + package = 'f' * 64 + elif kind == 'policy': + changed['configs']['policy_file']['sha256'] = 'f' * 64 + changed['routing']['policy']['sha256'] = 'f' * 64 + else: + changed['task']['lead']['mode'] = 'headless' + peer = profile('lead', 'lead-model') + changed['routing']['roles']['lead'] = { + 'selected': {'profile_id': 'lead', 'profile': peer, + 'profile_sha256': digest(peer)}, 'fallbacks': [], + 'fallback_mode': 'none', + } + after = api.paired_input_identity(changed, role='reviewer', package_digest=package) + self.assertEqual(base['case_sha256'], after['case_sha256']) + self.assertNotEqual(base['input_sha256'], after['input_sha256']) + + def test_tested_role_fallback_policy_is_not_the_experimental_variable(self): + api = self.api() + before = frozen_snapshot() + base = api.paired_input_identity(before, role='reviewer', package_digest='e' * 64) + changed = copy.deepcopy(before) + changed['routing']['roles']['reviewer']['fallback_mode'] = 'policy' + after = api.paired_input_identity(changed, role='reviewer', package_digest='e' * 64) + self.assertEqual(base['case_sha256'], after['case_sha256']) + self.assertNotEqual(base['input_sha256'], after['input_sha256']) + + def test_task_check_and_candidate_changes_cannot_share_an_input_identity(self): + api = self.api() + before = frozen_snapshot() + base = api.paired_input_identity(before, role='reviewer', package_digest='e' * 64) + for kind in ('goal', 'check', 'candidate', 'scope'): + with self.subTest(kind=kind): + changed = copy.deepcopy(before) + if kind == 'goal': + changed['task']['goal'] = 'A different bounded task.' + elif kind == 'check': + changed['task']['checks'] = [{ + 'id': 'required', 'argv': ['true'], 'cwd': '.', + 'timeout_seconds': 1, 'required_to_pass': True, + }] + elif kind == 'candidate': + changed['workspace']['candidate_sha256'] = 'f' * 64 + else: + changed['task']['scope']['read_paths'] = ['different'] + after = api.paired_input_identity(changed, role='reviewer', package_digest='e' * 64) + self.assertNotEqual(base['case_sha256'], after['case_sha256']) + self.assertNotEqual(base['input_sha256'], after['input_sha256']) + + def test_missing_or_inconsistent_frozen_inputs_fail_closed(self): + api = self.api() + for kind in ('policy', 'role', 'profile', 'candidate', 'package'): + with self.subTest(kind=kind): + changed = frozen_snapshot() + package = 'e' * 64 + if kind == 'policy': + changed['routing']['policy']['sha256'] = '0' * 64 + elif kind == 'role': + changed['routing']['roles'] = {} + elif kind == 'profile': + changed['routing']['roles']['reviewer']['selected']['profile_sha256'] = '0' * 64 + elif kind == 'candidate': + del changed['workspace'] + else: + package = None + with self.assertRaises(ContractError): + api.paired_input_identity(changed, role='reviewer', package_digest=package) + + def test_missing_tested_native_adapter_fails_closed(self): + snapshot = frozen_snapshot() + selected = snapshot['routing']['roles']['reviewer']['selected'] + selected['profile']['harness'] = 'codex' + selected['profile_sha256'] = digest(selected['profile']) + with self.assertRaisesRegex(ContractError, 'adapter'): + self.api().paired_input_identity(snapshot, role='reviewer', package_digest='e' * 64) + + +if __name__ == '__main__': + unittest.main() diff --git a/test/core/test_experiment_reuse.py b/test/core/test_experiment_reuse.py new file mode 100644 index 0000000..53eb064 --- /dev/null +++ b/test/core/test_experiment_reuse.py @@ -0,0 +1,110 @@ +"""Reject experiment sample inflation through reused outcome observations.""" + +from __future__ import annotations + +from pathlib import Path +import subprocess +import tempfile +import unittest + +from test_learning import NOW, experiment, experimental_final + +from devsquad.contracts import ContractError +from devsquad.learning import evaluate_experiment, validate_experiment +from devsquad.store import Store + + +def repeated_pair_spec(project_path: Path) -> dict: + """The original regression: two observations relabelled as three pairs.""" + spec = experiment(project_path) + control_id = spec["cases"][0]["control_outcome_id"] + candidate_id = spec["cases"][0]["candidate_outcome_id"] + for case in spec["cases"]: + case["control_outcome_id"] = control_id + case["candidate_outcome_id"] = candidate_id + return spec + + +def outcome_chains(spec: dict) -> dict: + chains = {} + for case in spec["cases"]: + for arm, verdict in (("control", "failed"), ("candidate", "succeeded")): + outcome_id = case[f"{arm}_outcome_id"] + chains.setdefault(outcome_id, { + "final": experimental_final(outcome_id, verdict), + "late_corrections": [], + }) + return chains + + +class ExperimentObservationReuseTest(unittest.TestCase): + def test_validator_rejects_two_outcomes_relabelled_as_three_cases(self): + spec = repeated_pair_spec(Path("/tmp/experiment-reuse-project")) + self.assertEqual(len(outcome_chains(spec)), 2) + self.assertEqual(len(spec["cases"]), 3) + with self.assertRaises(ContractError): + validate_experiment(spec) + + def test_validator_rejects_cross_arm_reuse_in_different_cases(self): + spec = experiment(Path("/tmp/experiment-reuse-project")) + spec["cases"][1]["control_outcome_id"] = spec["cases"][0]["candidate_outcome_id"] + # Each local pair is distinct; uniqueness must cover all arms/cases. + self.assertTrue(all( + case["control_outcome_id"] != case["candidate_outcome_id"] + for case in spec["cases"] + )) + with self.assertRaises(ContractError): + validate_experiment(spec) + + def test_validator_rejects_same_arm_reuse_across_splits(self): + spec = experiment(Path("/tmp/experiment-reuse-project")) + spec["cases"][2]["candidate_outcome_id"] = spec["cases"][0]["candidate_outcome_id"] + self.assertNotEqual(spec["cases"][0]["split"], spec["cases"][2]["split"]) + with self.assertRaises(ContractError): + validate_experiment(spec) + + def test_evaluator_cannot_promote_two_observations_as_three_pairs(self): + spec = repeated_pair_spec(Path("/tmp/experiment-reuse-project")) + chains = outcome_chains(spec) + with self.assertRaises(ContractError): + evaluate_experiment(spec, chains, evaluated_at=NOW.isoformat()) + + def test_store_rejects_reused_outcomes_without_persisting_evaluation(self): + with tempfile.TemporaryDirectory(prefix="devsquad-experiment-reuse-") as temporary: + root = Path(temporary) + repo = root / "repo" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + store = Store(root / "state.sqlite3", root / "artifacts") + try: + spec = repeated_pair_spec(repo) + chains = outcome_chains(spec) + for outcome_id, chain in chains.items(): + final = chain["final"] + claim = store.claim_start(repo, f"run-{outcome_id}", {}, "owner") + # Seed the same legacy terminal fixture used by test_learning. + # It must never make reused observations valid evidence. + store.connection.execute( + "UPDATE runs SET state=?,phase=NULL WHERE id=?", + (final["verdict"], claim.run_id), + ) + store.record_outcome(claim.run_id, final, now=NOW) + rejected = None + try: + store.evaluate_learning_experiment(spec, now=NOW) + except ContractError as exc: + rejected = exc + self.assertEqual( + store.connection.execute( + "SELECT COUNT(*) FROM experiments WHERE experiment_id=?", + (spec["experiment_id"],), + ).fetchone()[0], + 0, + "invalid reused evidence must not leave a saved evaluation", + ) + self.assertIsInstance(rejected, ContractError) + finally: + store.close() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_experiment_saved_runs.py b/test/core/test_experiment_saved_runs.py new file mode 100644 index 0000000..6794a24 --- /dev/null +++ b/test/core/test_experiment_saved_runs.py @@ -0,0 +1,347 @@ +"""Saved-run v2 evaluation through real offline workers and public completion.""" + +from datetime import datetime, timezone +import json +import tempfile +from pathlib import Path +import unittest + +from experiment_runtime_fixture import ExperimentRuntimeFixture +from test_experiment_provenance import digest +from test_learning import experimental_final + +from devsquad.contracts import ContractError +from devsquad.store import canonical_json + + +class ExperimentSavedRunsTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-saved-experiment-") + self.addCleanup(self.temporary.cleanup) + self.fixture = ExperimentRuntimeFixture(Path(self.temporary.name)) + self.addCleanup(self.fixture.close) + + def test_disjoint_public_runs_supply_v2_evaluation_without_submitted_provenance(self): + self.fixture.run_all() + self.assertEqual(len(set(self.fixture.runs.values())), 4) + store = self.fixture.store() + try: + rows = store.connection.execute("SELECT id,status,role,profile_id FROM attempts").fetchall() + self.assertEqual(len(rows), 4) + self.assertTrue(all(row["status"] == "finished" and row["role"] == "reviewer" for row in rows)) + self.assertEqual({row["profile_id"] for row in rows}, {"profile-a", "profile-b"}) + self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM experiment_assignments").fetchone()[0], 4) + finally: + store.close() + result = self.fixture.service.policy_evaluate(self.fixture.spec) + evaluation = result["evaluation"] + self.assertEqual(evaluation["schema_version"], 2) + self.assertEqual(evaluation["verdict"], "promotion_proposal") + self.assertEqual(evaluation["metrics"]["evaluation"]["available_pairs"], 1) + self.assertEqual(evaluation["metrics"]["held_out"]["available_pairs"], 1) + self.assertEqual(len(evaluation["evidence_sha256"]), 64) + replay = self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertTrue(replay["replayed"]) + self.assertEqual(replay, {**result, "replayed": True}) + + def test_public_issue_delivery_pairs_bind_baseline_and_all_real_attempts(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "delivery", workflow="issue-delivery") + self.addCleanup(fixture.close) + source_before = fixture.git("status", "--porcelain") + head_before = fixture.git("rev-parse", "HEAD") + fixture.run_all() + result = fixture.service.policy_evaluate(fixture.spec) + self.assertEqual(result["evaluation"]["verdict"], "promotion_proposal") + self.assertTrue(result["eligibility"]["eligible"]) + self.assertEqual(fixture.spec["variable"]["role"], "implementer") + store = fixture.store() + try: + for run_id in fixture.runs.values(): + attempts = store.attempts_for_run(run_id) + self.assertEqual([row["role"] for row in attempts], ["implementer", "reviewer"]) + self.assertTrue(all(row["status"] == "finished" for row in attempts)) + snapshot = json.loads(store.run(run_id)["mutable_snapshot"]) + self.assertNotEqual(snapshot["workspace"]["target_oid"], snapshot["target_oid"]) + finally: + store.close() + self.assertEqual(fixture.git("status", "--porcelain"), source_before) + self.assertEqual(fixture.git("rev-parse", "HEAD"), head_before) + + def test_project_symlink_alias_preserves_the_predeclared_spec_hash(self): + alias = self.fixture.root / "project-alias" + alias.symlink_to(self.fixture.repo, target_is_directory=True) + self.fixture.spec["project_path"] = str(alias) + frozen_hash = digest(self.fixture.spec) + self.fixture.run_all() + result = self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(result["evaluation"]["verdict"], "promotion_proposal") + self.assertEqual(result["experiment"]["project_path"], str(alias)) + self.assertEqual(result["evaluation"]["spec_sha256"], frozen_hash) + store = self.fixture.store() + try: + declared = store.connection.execute( + "SELECT spec_json,spec_sha256 FROM experiment_specs WHERE experiment_id=?", + (self.fixture.spec["experiment_id"],), + ).fetchone() + saved = store.connection.execute( + "SELECT spec_json,spec_sha256 FROM experiments WHERE experiment_id=?", + (self.fixture.spec["experiment_id"],), + ).fetchone() + self.assertEqual(tuple(saved), tuple(declared)) + self.assertEqual(saved["spec_sha256"], frozen_hash) + finally: + store.close() + + def test_real_terminal_worker_failure_is_retained_as_failed_exposure(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "terminal-failure", fail_candidate=True) + self.addCleanup(fixture.close) + fixture.run_all() + store = fixture.store() + try: + for case_id in ("eval-1", "hold-1"): + run_id = fixture.runs[(case_id, "candidate")] + self.assertEqual(store.run(run_id)["state"], "failed") + attempt = store.connection.execute( + "SELECT * FROM attempts WHERE run_id=?", (run_id,), + ).fetchone() + self.assertEqual(attempt["status"], "finished") + metadata = json.loads(attempt["output_metadata"]) + self.assertNotIn("failure", metadata) + self.assertEqual(metadata["stdout"]["captured_bytes"], 0) + self.assertGreater(metadata["stderr"]["captured_bytes"], 0) + finally: + store.close() + evaluation = fixture.service.policy_evaluate(fixture.spec)["evaluation"] + self.assertEqual(evaluation["verdict"], "no_change") + for split in ("evaluation", "held_out"): + self.assertEqual(evaluation["metrics"][split]["available_pairs"], 1) + self.assertEqual(evaluation["metrics"][split]["candidate_success_rate"], 0.0) + self.assertTrue(all(case["candidate_verdict"] == "failed" for case in evaluation["cases"])) + + def test_failed_receipt_requires_matching_terminal_attempt_even_with_valid_hash(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "failure-receipt", fail_candidate=True) + self.addCleanup(fixture.close) + run_id = fixture.run_arm("eval-1", "candidate") + store = fixture.store() + try: + artifact = dict(store.connection.execute( + "SELECT * FROM artifacts WHERE run_id=? AND name='result-receipt.json'", (run_id,), + ).fetchone()) + path = Path(artifact["path"]) + original = path.read_bytes() + receipt = json.loads(original) + changed_profile = json.loads(original) + changed_profile["attempts"][0]["selected_profile"]["profile"]["model_id"] = "wrong-model" + changed_status = json.loads(original) + changed_status["attempts"][0]["status"] = "succeeded" + mutations = [ + {**receipt, "attempts": None}, + {**receipt, "attempts": []}, + {**receipt, "state": "succeeded"}, + {**receipt, "run_id": "another-run"}, + changed_profile, changed_status, + ] + # Corrupt only negative evidence; each real failed run and its + # opaque native streams were produced through the public runtime. + for changed in mutations: + with self.subTest(receipt=changed): + content = canonical_json(changed).encode() + path.write_bytes(content) + store.connection.execute( + "UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (digest(changed), len(content), artifact["id"]), + ) + try: + with self.assertRaises(ContractError): + fixture.service.policy_evaluate(fixture.spec) + finally: + path.write_bytes(original) + store.connection.execute( + "UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (artifact["sha256"], artifact["byte_size"], artifact["id"]), + ) + finally: + store.close() + + def test_success_cannot_hide_missing_review_behind_failure_metadata(self): + run_id = self.fixture.run_arm("eval-1", "candidate") + store = self.fixture.store() + try: + attempt = store.connection.execute( + "SELECT id,output_metadata FROM attempts WHERE run_id=?", (run_id,), + ).fetchone() + artifact_name = f"review-attempt-{attempt['id']}.json" + metadata = json.loads(attempt["output_metadata"]) + metadata["failure"] = {"code": "CLI_ERROR"} + store.connection.execute( + "UPDATE artifacts SET name='hidden-review.json' WHERE run_id=? AND name=?", + (run_id, artifact_name), + ) + store.connection.execute( + "UPDATE attempts SET output_metadata=? WHERE id=?", + (canonical_json(metadata), attempt["id"]), + ) + with self.assertRaisesRegex(ContractError, "failed reviewer receipt"): + self.fixture.service.policy_evaluate(self.fixture.spec) + finally: + store.close() + + def test_reader_uses_prelaunch_snapshot_not_later_mutable_snapshot(self): + self.fixture.run_all() + run_id = self.fixture.runs[("hold-1", "candidate")] + store = self.fixture.store() + try: + original = store.run(run_id)["mutable_snapshot"] + changed = json.loads(original) + changed["task"]["goal"] = "Mutable continuation no longer describes the original trial." + changed["routing"]["roles"]["reviewer"]["selected"]["profile"]["model_id"] = "changed-model" + changed["experiment_assignment"]["spec_sha256"] = "0" * 64 + store.connection.execute( + "UPDATE runs SET mutable_snapshot=? WHERE id=?", (canonical_json(changed), run_id), + ) + try: + result = self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(result["evaluation"]["verdict"], "promotion_proposal") + finally: + store.connection.execute("UPDATE runs SET mutable_snapshot=? WHERE id=?", (original, run_id)) + replay = self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(replay, {**result, "replayed": True}) + finally: + store.close() + + def test_success_after_real_wrong_profile_fallback_cannot_credit_declared_arm(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "fallback", with_fallback=True) + self.addCleanup(fixture.close) + run_id = fixture.run_arm("eval-1", "candidate") + self.assertEqual(fixture.service.status(run_id)["state"], "succeeded") + store = fixture.store() + try: + attempts = store.connection.execute( + "SELECT profile_id,profile_index,output_metadata FROM attempts WHERE run_id=? ORDER BY profile_index", + (run_id,), + ).fetchall() + self.assertEqual([row["profile_id"] for row in attempts], ["candidate-fixture-fail", "profile-fallback"]) + self.assertEqual([row["profile_index"] for row in attempts], [0, 1]) + self.assertIsNotNone(json.loads(attempts[0]["output_metadata"])["failure"]) + self.assertEqual(len(store.outcomes_for_run(run_id)), 1) + with self.assertRaisesRegex(ContractError, "declared arm"): + fixture.service.policy_evaluate(fixture.spec) + self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM experiments").fetchone()[0], 0) + finally: + store.close() + + def test_missing_partner_is_visible_without_an_invented_pair(self): + self.fixture.run_all(skip=("hold-1", "control")) + evaluation = self.fixture.service.policy_evaluate(self.fixture.spec)["evaluation"] + self.assertEqual(evaluation["verdict"], "no_change") + self.assertEqual(evaluation["metrics"]["evaluation"]["available_pairs"], 1) + self.assertEqual(evaluation["metrics"]["held_out"]["available_pairs"], 0) + self.assertIn("insufficient_held_out_pairs", evaluation["reasons"]) + row = next(row for row in evaluation["cases"] if row["case_id"] == "hold-1") + self.assertEqual(row["status"], "missing") + self.assertEqual(row["missing"], ["control"]) + + def test_public_prelaunch_cancel_has_no_trial_exposure(self): + self.fixture.run_all(no_attempt=("hold-1", "candidate")) + cancelled_run = self.fixture.runs[("hold-1", "candidate")] + store = self.fixture.store() + try: + self.assertEqual(store.run(cancelled_run)["state"], "cancelled") + self.assertEqual(store.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=?", (cancelled_run,), + ).fetchone()[0], 0) + self.assertEqual(len(store.outcomes_for_run(cancelled_run)), 1) + finally: + store.close() + evaluation = self.fixture.service.policy_evaluate(self.fixture.spec)["evaluation"] + self.assertEqual(evaluation["verdict"], "no_change") + self.assertEqual(evaluation["metrics"]["evaluation"]["available_pairs"], 1) + self.assertEqual(evaluation["metrics"]["held_out"]["available_pairs"], 0) + row = next(row for row in evaluation["cases"] if row["case_id"] == "hold-1") + self.assertEqual(row["status"], "missing") + self.assertIn("candidate", row["missing"]) + + def test_saved_tampering_is_rejected_even_when_partner_is_missing(self): + self.fixture.run_all(skip=("hold-1", "control")) + run_id = self.fixture.runs[("hold-1", "candidate")] + store = self.fixture.store() + try: + attempt = dict(store.connection.execute( + "SELECT * FROM attempts WHERE run_id=?", (run_id,), + ).fetchone()) + assignment = dict(store.connection.execute( + "SELECT * FROM experiment_assignments WHERE run_id=?", (run_id,), + ).fetchone()) + changed_assignment = json.loads(assignment["assignment_json"]) + changed_assignment["profile_id"] = "profile-a" + changed_snapshot = json.loads(assignment["snapshot_json"]) + changed_snapshot["task"]["goal"] = "A changed input after preparation." + mutations = [ + ("actual_profile", "attempts", "id", attempt["id"], + {"profile_id": "profile-a"}, {"profile_id": attempt["profile_id"]}), + ("actual_index", "attempts", "id", attempt["id"], + {"profile_index": 1}, {"profile_index": attempt["profile_index"]}), + ("actual_package", "attempts", "id", attempt["id"], + {"package_digest": "0" * 64}, {"package_digest": attempt["package_digest"]}), + ("saved_assignment", "experiment_assignments", "run_id", run_id, + {"assignment_json": canonical_json(changed_assignment), "assignment_sha256": digest(changed_assignment)}, + {"assignment_json": assignment["assignment_json"], "assignment_sha256": assignment["assignment_sha256"]}), + ("paired_input", "experiment_assignments", "run_id", run_id, + {"snapshot_json": canonical_json(changed_snapshot)}, {"snapshot_json": assignment["snapshot_json"]}), + ] + # SQL is intentionally limited to these negative corruptions. The + # positive runs above were prepared/executed/completed publicly. + for name, table, key, identity, changed, original in mutations: + with self.subTest(mutation=name): + columns = ",".join(f"{column}=?" for column in changed) + statement = f"UPDATE {table} SET {columns} WHERE {key}=?" + store.connection.execute(statement, [*changed.values(), identity]) + try: + with self.assertRaises(ContractError): + self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(store.connection.execute( + "SELECT COUNT(*) FROM experiments WHERE experiment_id=?", + (self.fixture.spec["experiment_id"],), + ).fetchone()[0], 0) + finally: + store.connection.execute(statement, [*original.values(), identity]) + finally: + store.close() + + def test_late_correction_invalidates_replay_without_rewriting_historical_evaluation(self): + self.fixture.run_all() + first = self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(first["evaluation"]["verdict"], "promotion_proposal") + store = self.fixture.store() + try: + original = dict(store.connection.execute( + "SELECT spec_json,spec_sha256,evaluation_json,evaluation_sha256,recorded_at " + "FROM experiments WHERE experiment_id=?", (self.fixture.spec["experiment_id"],), + ).fetchone()) + finally: + store.close() + correction = experimental_final("escaped-candidate-hold-1", "succeeded") + correction.update({ + "kind": "late_correction", "verdict": "escaped_defect", + "corrects_outcome_id": "candidate-hold-1", + "observed_at": datetime.now(timezone.utc).isoformat(), + "summary": "An escaped defect was found after the saved evaluation.", + }) + run_id = self.fixture.runs[("hold-1", "candidate")] + self.fixture.service.outcome_add(run_id, correction) + with self.assertRaises(ContractError): + self.fixture.service.policy_evaluate(self.fixture.spec) + store = self.fixture.store() + try: + saved = dict(store.connection.execute( + "SELECT spec_json,spec_sha256,evaluation_json,evaluation_sha256,recorded_at " + "FROM experiments WHERE experiment_id=?", (self.fixture.spec["experiment_id"],), + ).fetchone()) + self.assertEqual(saved, original) + self.assertEqual(len(store.outcomes_for_run(run_id)), 2) + finally: + store.close() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_experiment_v2_evaluation.py b/test/core/test_experiment_v2_evaluation.py new file mode 100644 index 0000000..51084ab --- /dev/null +++ b/test/core/test_experiment_v2_evaluation.py @@ -0,0 +1,212 @@ +"""Pure normalized-chain tests, not public or real-run provenance proof.""" + +from __future__ import annotations + +import copy +from datetime import timedelta +import unittest + +from test_experiment_provenance import digest, spec_v2 +from test_learning import NOW, experimental_final + +from devsquad.contracts import ContractError +from devsquad.experiment_provenance import assignment_for +from devsquad.learning import evaluate_experiment + + +PROJECT_COMMON_DIR = "/tmp/provenance-project/.git" + + +def normalized_chains(spec): + """Fabricate the normalized reader boundary, never saved-run authority.""" + chains = {} + for case in spec["cases"]: + for arm, verdict in (("control", "failed"), ("candidate", "succeeded")): + outcome_id = case[f"{arm}_outcome_id"] + final = experimental_final(outcome_id, verdict) + assignment = assignment_for( + spec, case["case_id"], arm, project_common_dir=PROJECT_COMMON_DIR, + ) + attempt_id = f"attempt-{outcome_id}" + chains[outcome_id] = { + "final": final, + "late_corrections": [], + "provenance": { + "run_id": f"run-{outcome_id}", + "assignment": assignment, + "profile_sha256": assignment["profile_sha256"], + "execution_sha256": assignment["execution_sha256"], + "input_sha256": assignment["input_sha256"], + "case_sha256": assignment["case_sha256"], + "attempt_ids": [attempt_id], + "attempts_sha256": digest([attempt_id]), + "final_outcome_sha256": digest(final), + "correction_sha256": [], + }, + } + return chains + + +def correction_for(final): + correction = experimental_final(f"escaped-{final['outcome_id']}", "succeeded") + correction.update({ + "kind": "late_correction", + "verdict": "escaped_defect", + "corrects_outcome_id": final["outcome_id"], + "observed_at": (NOW + timedelta(seconds=1)).isoformat(), + "summary": "A later check found an escaped defect in this exact arm.", + }) + return correction + + +class ExperimentV2EvaluationTest(unittest.TestCase): + def setUp(self): + self.spec = spec_v2() + self.chains = normalized_chains(self.spec) + + def evaluate(self, chains=None): + return evaluate_experiment( + self.spec, self.chains if chains is None else chains, + evaluated_at=(NOW + timedelta(seconds=2)).isoformat(), + project_common_dir=PROJECT_COMMON_DIR, + ) + + def test_disjoint_paired_observations_produce_a_bound_proposal(self): + result = self.evaluate() + self.assertEqual(result["schema_version"], 2) + self.assertEqual(result["verdict"], "promotion_proposal") + self.assertFalse(result["active_policy_changed"]) + self.assertEqual(result["metrics"]["evaluation"]["available_pairs"], 2) + self.assertEqual(result["metrics"]["held_out"]["available_pairs"], 1) + self.assertEqual(result["metrics"]["evaluation"]["control_successes"], 0) + self.assertEqual(result["evidence_sha256"], digest({ + outcome_id: chain["provenance"] + for outcome_id, chain in self.chains.items() + })) + self.assertEqual(self.evaluate(), result) + + def test_failed_candidate_remains_visible_and_does_not_qualify(self): + chain = self.chains["candidate-hold-1"] + chain["final"]["verdict"] = "failed" + chain["provenance"]["final_outcome_sha256"] = digest(chain["final"]) + result = self.evaluate() + self.assertEqual(result["verdict"], "no_change") + self.assertEqual(result["metrics"]["held_out"]["available_pairs"], 1) + self.assertEqual(result["metrics"]["held_out"]["candidate_successes"], 0) + self.assertIn({"case_id": "hold-1", "reason": "candidate_not_successful"}, result["failures"]) + + def test_missing_partner_remains_visible_without_counting_a_pair(self): + before = self.evaluate() + for arm in ("control", "candidate"): + with self.subTest(arm=arm): + chains = copy.deepcopy(self.chains) + missing_id = f"{arm}-hold-1" + del chains[missing_id] + result = self.evaluate(chains) + self.assertEqual(result["verdict"], "no_change") + self.assertEqual(result["metrics"]["held_out"]["available_pairs"], 0) + row = next(row for row in result["cases"] if row["case_id"] == "hold-1") + self.assertEqual(row["status"], "missing") + self.assertEqual(row["missing"], [arm]) + self.assertIn("insufficient_held_out_pairs", result["reasons"]) + self.assertNotEqual(result["evidence_sha256"], before["evidence_sha256"]) + + def test_duplicate_runs_are_rejected_even_with_a_missing_partner(self): + for missing_partner in (False, True): + with self.subTest(missing_partner=missing_partner): + chains = copy.deepcopy(self.chains) + chains["candidate-eval-2"]["provenance"]["run_id"] = chains["candidate-eval-1"]["provenance"]["run_id"] + if missing_partner: + del chains["control-eval-2"] + with self.assertRaisesRegex(ContractError, "run ids.*unique"): + self.evaluate(chains) + + def test_duplicate_attempts_are_rejected_even_with_a_missing_partner(self): + for missing_partner in (False, True): + with self.subTest(missing_partner=missing_partner): + chains = copy.deepcopy(self.chains) + prior = chains["candidate-eval-1"]["provenance"] + current = chains["candidate-eval-2"]["provenance"] + current["attempt_ids"] = list(prior["attempt_ids"]) + current["attempts_sha256"] = prior["attempts_sha256"] + if missing_partner: + del chains["control-eval-2"] + with self.assertRaisesRegex(ContractError, "attempt ids.*unique"): + self.evaluate(chains) + + def test_crossed_assignments_profiles_and_inputs_fail_closed(self): + changes = [ + (("assignment", "experiment_id"), "foreign-experiment"), + (("assignment", "spec_sha256"), "0" * 64), + (("assignment", "project_common_dir"), "/tmp/foreign-project/.git"), + (("assignment", "case_id"), "eval-2"), + (("assignment", "split"), "held_out"), + (("assignment", "arm"), "control"), + (("assignment", "role"), "implementer"), + (("assignment", "profile_id"), "profile-a"), + (("profile_sha256",), self.spec["variable"]["control_profile_sha256"]), + (("execution_sha256",), "0" * 64), + (("input_sha256",), self.spec["cases"][1]["input_sha256"]), + (("case_sha256",), self.spec["cases"][1]["case_sha256"]), + ] + for path, replacement in changes: + with self.subTest(path=path): + chains = copy.deepcopy(self.chains) + target = chains["candidate-eval-1"]["provenance"] + for field in path[:-1]: + target = target[field] + target[path[-1]] = replacement + with self.assertRaises(ContractError): + self.evaluate(chains) + + def test_final_outcome_bytes_must_match_the_saved_hash(self): + chains = copy.deepcopy(self.chains) + chains["candidate-eval-1"]["final"]["summary"] = "Changed after the witness was frozen." + with self.assertRaisesRegex(ContractError, "final outcome hash"): + self.evaluate(chains) + + def test_strict_late_correction_blocks_promotion_and_changes_evidence_digest(self): + before = self.evaluate() + chain = self.chains["candidate-hold-1"] + correction = correction_for(chain["final"]) + chain["late_corrections"] = [correction] + chain["provenance"]["correction_sha256"] = [digest(correction)] + after = self.evaluate() + self.assertEqual(after["verdict"], "no_change") + self.assertEqual(after["metrics"]["held_out"]["candidate_escaped_defects"], 1) + self.assertIn("candidate_escaped_defect_limit_exceeded", after["reasons"]) + self.assertIn({"case_id": "hold-1", "reason": "candidate_escaped_defect"}, after["failures"]) + self.assertNotEqual(after["evidence_sha256"], before["evidence_sha256"]) + + def test_malformed_or_crossed_late_corrections_are_rejected(self): + changes = [ + ("corrects_outcome_id", "candidate-eval-1"), + ("selection_mode", "automatic"), + ("observed_at", (NOW - timedelta(seconds=1)).isoformat()), + ("evidence_refs", []), + ("kind", "late"), + ] + for field, replacement in changes: + with self.subTest(field=field): + chains = copy.deepcopy(self.chains) + chain = chains["candidate-hold-1"] + correction = correction_for(chain["final"]) + correction[field] = replacement + chain["late_corrections"] = [correction] + chain["provenance"]["correction_sha256"] = [digest(correction)] + with self.assertRaises(ContractError): + self.evaluate(chains) + + def test_correction_hash_cannot_omit_or_misrepresent_saved_correction(self): + for invalid_hashes in ([], ["0" * 64]): + with self.subTest(hashes=invalid_hashes): + chains = copy.deepcopy(self.chains) + chain = chains["candidate-hold-1"] + chain["late_corrections"] = [correction_for(chain["final"])] + chain["provenance"]["correction_sha256"] = invalid_hashes + with self.assertRaises(ContractError): + self.evaluate(chains) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_handoff_service.py b/test/core/test_handoff_service.py new file mode 100644 index 0000000..e6e3b7d --- /dev/null +++ b/test/core/test_handoff_service.py @@ -0,0 +1,209 @@ +from __future__ import annotations + +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.service import Service +from devsquad.store import ConflictError, Store, request_hash + + +class HandoffServiceTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-handoff-service-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], + check=True, + ) + (self.repo / "README").write_text("base\n") + subprocess.run(["git", "-C", str(self.repo), "add", "README"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.service = Service(self.runtime) + + def waiting_run(self, key: str = "handoff") -> tuple[str, dict, int]: + packet = { + "schema_version": 1, + "candidate_sha256": "c" * 64, + "instructions": "Review the frozen candidate.", + } + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + claim = store.claim_start(self.repo, key, {"task": key}, "preflight") + version = store.complete_preparation( + claim.run_id, + claim.fencing_token, + {"base_oid": "a" * 40, "target_oid": "b" * 40}, + package_path="/frozen/package", + package_digest="package-digest", + ) + reservation = store.reserve_attempt( + claim.run_id, version, "supervisor", "package-digest", + ) + version = store.mark_attempt_running( + reservation, 101, 101, "process-start-id", + ) + snapshot = store.publish_handoff( + claim.run_id, + version, + reservation.attempt_token, + reservation.supervisor_token, + packet, + ) + return claim.run_id, packet, snapshot.run_version + finally: + store.close() + + @staticmethod + def decision( + submission_id: str = "submission-1", + disposition: str = "accept", + reason: str = "accepted", + ) -> dict: + body = { + "schema_version": 1, + "submission_id": submission_id, + "disposition": disposition, + "reason": reason, + "evidence_refs": [], + } + return {**body, "submission_hash": request_hash(body)} + + def test_status_claim_complete_and_exact_replay_share_one_saved_run(self): + run_id, packet, version = self.waiting_run() + waiting = self.service.status(run_id) + self.assertEqual( + (waiting["state"], waiting["phase"], waiting["version"], waiting["next_action"]), + ("awaiting_host", None, version, "claim_handoff"), + ) + self.assertEqual(waiting["handoff"]["status"], "open") + self.assertNotIn("packet", waiting["handoff"]) + self.assertNotIn("fencing_token", waiting["handoff"]) + + acquired = self.service.handoff_claim(run_id, version, "terminal-a") + self.assertEqual(acquired["action"], "acquired") + self.assertEqual(acquired["handoff"]["packet"], packet) + self.assertEqual(acquired["handoff"]["claimed_by"], "terminal-a") + public_claim = acquired["claim"] + self.assertEqual(set(public_claim), { + "schema_version", "run_id", "handoff_id", "owner", + "fencing_token", "expires_at", "run_version", + }) + + renewed = self.service.handoff_claim( + run_id, acquired["version"], "terminal-a", public_claim, + ) + self.assertEqual(renewed["action"], "renewed") + self.assertEqual( + renewed["claim"]["fencing_token"], public_claim["fencing_token"], + ) + with self.assertRaises(ConflictError): + self.service.handoff_claim( + run_id, renewed["version"], "terminal-a", public_claim, + ) + + decision = self.decision() + completed = self.service.handoff_complete(run_id, renewed["claim"], decision) + self.assertEqual( + (completed["state"], completed["phase"], completed["disposition"]), + ("awaiting_host", "handoff_submitted", "accept"), + ) + self.assertFalse(completed["replayed"]) + replay = self.service.handoff_complete(run_id, renewed["claim"], decision) + self.assertTrue(replay["replayed"]) + self.assertEqual( + replay["recorded_run_version"], completed["recorded_run_version"], + ) + + def test_cancel_awaiting_host_is_terminal_and_keeps_replay_idempotent(self): + run_id, _, version = self.waiting_run("cancel-wait") + acquired = self.service.handoff_claim(run_id, version, "terminal-a") + decision = self.decision() + completed = self.service.handoff_complete(run_id, acquired["claim"], decision) + cancelled = self.service.cancel(run_id) + self.assertEqual(cancelled["state"], "cancelled") + result = self.service.result(run_id) + self.assertTrue(result["ready"]) + self.assertEqual(result["state"], "cancelled") + replay = self.service.handoff_complete(run_id, acquired["claim"], decision) + self.assertTrue(replay["replayed"]) + self.assertEqual(replay["state"], "cancelled") + self.assertEqual( + replay["recorded_run_version"], completed["recorded_run_version"], + ) + + def test_public_claim_shape_and_run_binding_are_strict(self): + run_id, _, version = self.waiting_run("strict-claim") + with self.assertRaises(ContractError): + self.service.handoff_complete(run_id, {"schema_version": 1}, self.decision()) + acquired = self.service.handoff_claim(run_id, version, "terminal-a") + wrong_run = dict(acquired["claim"], run_id="different-run") + with self.assertRaises(ConflictError): + self.service.handoff_complete(run_id, wrong_run, self.decision()) + naive_expiry = dict(acquired["claim"], expires_at="2026-09-15T05:00:00") + with self.assertRaises(ContractError): + self.service.handoff_complete(run_id, naive_expiry, self.decision()) + + def test_independent_cli_processes_claim_and_complete_the_saved_handoff(self): + run_id, packet, version = self.waiting_run("cli-handoff") + environment = os.environ.copy() + environment["PYTHONPATH"] = str(ROOT / "plugin/core/src") + + def invoke(arguments): + result = subprocess.run( + [sys.executable, "-P", "-m", "devsquad.cli", *arguments], + cwd=self.root, + env=environment, + text=True, + capture_output=True, + check=False, + ) + self.assertEqual(result.stderr, "") + self.assertEqual(result.returncode, 0, result.stdout) + return json.loads(result.stdout) + + claimed = invoke([ + "handoff", "claim", run_id, + "--expected-version", str(version), + "--owner", "second-terminal", + "--runtime-dir", str(self.runtime), + "--json", + ]) + self.assertTrue(claimed["ok"]) + self.assertEqual(claimed["data"]["handoff"]["packet"], packet) + claim_file = self.root / "host-claim.json" + claim_file.write_text(json.dumps(claimed["data"]["claim"])) + decision = self.decision() + decision_file = self.root / "host-decision.json" + decision_file.write_text(json.dumps(decision)) + + completed = invoke([ + "handoff", "complete", run_id, + "--claim-file", str(claim_file), + "--decision-file", str(decision_file), + "--runtime-dir", str(self.runtime), + "--json", + ]) + self.assertTrue(completed["ok"]) + self.assertEqual(completed["data"]["phase"], "handoff_submitted") + self.assertEqual(completed["data"]["submission_hash"], decision["submission_hash"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py new file mode 100644 index 0000000..cfff039 --- /dev/null +++ b/test/core/test_handoff_store.py @@ -0,0 +1,908 @@ +from datetime import datetime, timedelta, timezone +import json +import os +from pathlib import Path +import shutil +import sqlite3 +import subprocess +import sys +import tempfile +import threading +import time +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.store import ( + ConflictError, + HandoffClaim, + SUPPORTED_SCHEMA_VERSION, + Store, + canonical_json, + request_hash, +) +from devsquad.contracts import ContractError + + +class HandoffStoreTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-handoff-") + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True, + ) + (self.repo / "README").write_text("base\n") + subprocess.run(["git", "-C", str(self.repo), "add", "README"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.database = self.root / "runtime/state.sqlite3" + self.artifacts = self.root / "runtime/artifacts" + self.store = Store(self.database, self.artifacts) + + def tearDown(self): + self.store.close() + self.temp.cleanup() + + def running_run(self, key="handoff"): + claim = self.store.claim_start(self.repo, key, {"task": key}, "preflight") + version = self.store.complete_preparation( + claim.run_id, + claim.fencing_token, + {"base_oid": "a" * 40, "target_oid": "b" * 40}, + package_path="/frozen/package", + package_digest="package-digest", + ) + reservation = self.store.reserve_attempt( + claim.run_id, version, "supervisor", "package-digest", + ) + version = self.store.mark_attempt_running( + reservation, 101, 101, "process-start-id", + ) + return claim.run_id, reservation, version + + def waiting_run(self, key="handoff"): + run_id, reservation, version = self.running_run(key) + packet = { + "schema_version": 1, + "candidate_sha256": "c" * 64, + "evidence_refs": [], + } + snapshot = self.store.publish_handoff( + run_id, + version, + reservation.attempt_token, + reservation.supervisor_token, + packet, + now=datetime(2026, 9, 15, 5, 0, tzinfo=timezone.utc), + ) + return run_id, reservation, snapshot + + @staticmethod + def decision(submission_id="submission-1", disposition="accept", reason="accepted"): + body = { + "schema_version": 1, + "submission_id": submission_id, + "disposition": disposition, + "reason": reason, + "evidence_refs": [], + } + return {**body, "submission_hash": request_hash(body)} + + def claim_with_deadline(self, key, deadline): + run_id, _, snapshot = self.waiting_run(key) + claim = self.store.claim_handoff( + run_id, + snapshot.run_version, + "host", + now=deadline - timedelta(minutes=1), + ) + expires_at = deadline.isoformat() + self.store.connection.execute( + "UPDATE claims SET lease_expires_at=? WHERE run_id=?", + (expires_at, run_id), + ) + return run_id, HandoffClaim( + run_id=claim.run_id, + handoff_id=claim.handoff_id, + owner_id=claim.owner_id, + fencing_token=claim.fencing_token, + expires_at=expires_at, + run_version=claim.run_version, + action=claim.action, + ) + + def run_after_independent_writer_wait(self, deadline, operation): + """Run an operation whose BEGIN IMMEDIATE is blocked across a lease expiry.""" + ready = threading.Event() + begin_attempted = threading.Event() + release_started = threading.Event() + failures = [] + connection = self.store.connection + + class BeginNotifyingConnection: + def execute(self, statement, *args, **kwargs): + if statement == "BEGIN IMMEDIATE": + begin_attempted.set() + return connection.execute(statement, *args, **kwargs) + + def __getattr__(self, name): + return getattr(connection, name) + + def hold_lock(): + writer = sqlite3.connect(self.database, timeout=5, isolation_level=None) + try: + writer.execute("PRAGMA busy_timeout=5000") + writer.execute("BEGIN IMMEDIATE") + ready.set() + if not begin_attempted.wait(5): + raise AssertionError("handoff operation did not attempt its write transaction") + # Give the caller time to enter SQLite's busy wait before releasing the lock. + time.sleep(0.05) + release_started.set() + writer.execute("COMMIT") + except BaseException as exc: # Preserve thread failures for the assertion owner. + failures.append(exc) + ready.set() + finally: + if writer.in_transaction: + writer.execute("ROLLBACK") + writer.close() + + thread = threading.Thread(target=hold_lock) + thread.start() + self.store.connection = BeginNotifyingConnection() + try: + self.assertTrue(ready.wait(5), "independent SQLite writer did not acquire its lock") + self.assertEqual(failures, []) + before_expiry = deadline - timedelta(seconds=1) + after_expiry = deadline + timedelta(seconds=1) + with mock.patch( + "devsquad.store._authoritative_now", + side_effect=lambda value=None: ( + after_expiry if release_started.is_set() else before_expiry + ), + ): + return operation() + finally: + begin_attempted.set() + thread.join(5) + self.store.connection = connection + self.assertFalse(thread.is_alive()) + self.assertEqual(failures, []) + + def test_schema_four_fixture_migrates_to_host_handoffs(self): + old_database = self.root / "schema-four.sqlite3" + connection = sqlite3.connect(old_database) + migration_dir = ROOT / "plugin/core/src/devsquad/migrations" + for version, name in ( + (1, "001_initial.sql"), + (2, "002_supervisor.sql"), + (3, "003_durable_io.sql"), + (4, "004_run_snapshot.sql"), + ): + connection.executescript((migration_dir / name).read_text()) + connection.execute( + "INSERT INTO schema_migrations(version,applied_at) VALUES(?, 'fixture')", + (version,), + ) + connection.commit() + connection.close() + + upgraded = Store(old_database, self.root / "schema-four-artifacts") + self.addCleanup(upgraded.close) + self.assertEqual( + upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], + SUPPORTED_SCHEMA_VERSION, + ) + tables = { + row[0] + for row in upgraded.connection.execute( + "SELECT name FROM sqlite_master WHERE type='table'" + ) + } + self.assertTrue({"handoffs", "handoff_submissions"} <= tables) + claim_columns = { + row[1] for row in upgraded.connection.execute("PRAGMA table_info(claims)") + } + self.assertTrue({"handoff_id", "lease_expires_at", "renewed_at"} <= claim_columns) + with self.assertRaises(sqlite3.IntegrityError): + upgraded.connection.execute( + "INSERT INTO handoffs(id,run_id,sequence,packet_json,packet_sha256,status," + "created_run_version,created_at) VALUES('bad','missing',1,'{}',?,'invalid',1,'now')", + ("0" * 64,), + ) + + def test_publish_is_fenced_and_atomically_releases_writer(self): + run_id, reservation, version = self.running_run() + packet = {"schema_version": 1, "candidate_sha256": "c" * 64} + snapshot = self.store.publish_handoff( + run_id, + version, + reservation.attempt_token, + reservation.supervisor_token, + packet, + ) + self.assertEqual(snapshot.packet, packet) + self.assertEqual(snapshot.run_state, "awaiting_host") + self.assertEqual(snapshot.run_version, version + 1) + self.assertEqual(snapshot.created_run_version, version + 1) + self.assertEqual( + self.store.connection.execute( + "SELECT status FROM attempts WHERE attempt_token=?", + (reservation.attempt_token,), + ).fetchone()[0], + "finished", + ) + self.assertEqual( + self.store.connection.execute( + "SELECT active FROM supervisor_claims WHERE run_id=?", (run_id,), + ).fetchone()[0], + 0, + ) + event = self.store.connection.execute( + "SELECT run_version,type,payload FROM events WHERE run_id=? ORDER BY id DESC LIMIT 1", + (run_id,), + ).fetchone() + self.assertEqual((event[0], event[1]), (version + 1, "run.awaiting_host")) + self.assertEqual(json.loads(event[2])["packet_sha256"], snapshot.packet_sha256) + with self.assertRaises(ConflictError): + self.store.publish_handoff( + run_id, + version, + reservation.attempt_token, + reservation.supervisor_token, + packet, + ) + self.store.connection.execute( + "UPDATE handoffs SET packet_json='{}' WHERE id=?", (snapshot.handoff_id,), + ) + with self.assertRaises(ConflictError): + self.store.handoff_snapshot(run_id) + + def test_durable_evidence_and_handoff_cross_one_atomic_fence(self): + run_id, reservation, _ = self.running_run("durable-handoff") + artifacts = [] + for name, content in ( + (f"{reservation.attempt_id}.stdout", b'{"workflow":"review"}\n'), + (f"{reservation.attempt_id}.stderr", b""), + ("review.json", b'{"verdict":"clean"}\n'), + ): + path, digest, size = self.store.finalize_artifact(run_id, name, content) + artifacts.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + review = next(item for item in artifacts if item["name"] == "review.json") + packet = { + "schema_version": 1, + "candidate_sha256": "c" * 64, + "artifacts": [{"name": review["name"], "sha256": review["sha256"]}], + } + outcome = self.store.commit_durable_handoff( + run_id, + reservation.attempt_token, + artifacts, + {"stdout": {"captured_bytes": 22}, "stderr": {"captured_bytes": 0}}, + packet, + ) + self.assertEqual(outcome, "awaiting_host") + snapshot = self.store.handoff_snapshot(run_id) + self.assertEqual(snapshot.packet["candidate_sha256"], "c" * 64) + reference = snapshot.packet["artifacts"][0] + self.assertEqual(set(reference), {"artifact_id", "name", "sha256"}) + artifact = self.store.connection.execute( + "SELECT name,sha256 FROM artifacts WHERE id=?", (reference["artifact_id"],), + ).fetchone() + self.assertEqual((artifact["name"], artifact["sha256"]), ( + "review.json", review["sha256"], + )) + self.assertEqual( + self.store.commit_durable_handoff( + run_id, + reservation.attempt_token, + artifacts, + {"stdout": {"captured_bytes": 22}, "stderr": {"captured_bytes": 0}}, + packet, + ), + "awaiting_host", + ) + + invalid_run, invalid_reservation, _ = self.running_run("invalid-evidence") + invalid_artifacts = [] + for name, content in ( + (f"{invalid_reservation.attempt_id}.stdout", b"output"), + (f"{invalid_reservation.attempt_id}.stderr", b""), + ): + path, digest, size = self.store.finalize_artifact( + invalid_run, name, content, + ) + invalid_artifacts.append({ + "name": name, "path": path, "sha256": digest, "byte_size": size, + }) + with self.assertRaisesRegex(ContractError, "does not match durable evidence"): + self.store.commit_durable_handoff( + invalid_run, + invalid_reservation.attempt_token, + invalid_artifacts, + {}, + { + "schema_version": 1, + "artifacts": [{"name": "missing.json", "sha256": "0" * 64}], + }, + ) + self.assertEqual(self.store.run(invalid_run)["state"], "running") + + def test_claim_cas_renewal_and_expired_takeover(self): + run_id, _, snapshot = self.waiting_run() + first_time = datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc) + first = self.store.claim_handoff( + run_id, snapshot.run_version, "host-a", now=first_time, + ) + self.assertEqual(first.action, "acquired") + renewed = self.store.claim_handoff( + run_id, + first.run_version, + "host-a", + first, + now=first_time + timedelta(minutes=5), + ) + self.assertEqual(renewed.action, "renewed") + self.assertEqual(renewed.fencing_token, first.fencing_token) + self.assertGreater(renewed.expires_at, first.expires_at) + with self.assertRaises(ConflictError): + self.store.claim_handoff( + run_id, + renewed.run_version, + "host-a", + first, + now=first_time + timedelta(minutes=6), + ) + with self.assertRaises(ConflictError): + self.store.claim_handoff( + run_id, + renewed.run_version, + "host-b", + now=first_time + timedelta(minutes=6), + ) + with self.assertRaises(ConflictError): + self.store.claim_handoff( + run_id, + renewed.run_version, + "host-a", + renewed, + now=first_time + timedelta(minutes=15), + ) + takeover = self.store.claim_handoff( + run_id, + renewed.run_version, + "host-b", + now=first_time + timedelta(minutes=15), + ) + self.assertEqual(takeover.action, "taken_over") + self.assertEqual(takeover.fencing_token, first.fencing_token + 1) + + def test_two_claimants_at_one_version_yield_one_owner(self): + run_id, _, snapshot = self.waiting_run() + barrier = threading.Barrier(2) + claims = [] + conflicts = [] + + def claim(owner): + store = Store(self.database, self.artifacts) + try: + barrier.wait() + claims.append(store.claim_handoff(run_id, snapshot.run_version, owner)) + except ConflictError as exc: + conflicts.append(exc) + finally: + store.close() + + threads = [threading.Thread(target=claim, args=(owner,)) for owner in ("a", "b")] + for thread in threads: + thread.start() + for thread in threads: + thread.join() + self.assertEqual(len(claims), 1) + self.assertEqual(len(conflicts), 1) + + def test_renewal_checks_expiry_after_waiting_for_write_lock(self): + deadline = datetime(2026, 9, 15, 5, 2, tzinfo=timezone.utc) + run_id, claim = self.claim_with_deadline("renewal-lock", deadline) + version_before = self.store.run(run_id)["version"] + with self.assertRaisesRegex(ConflictError, "stale or expired"): + self.run_after_independent_writer_wait( + deadline, + lambda: self.store.claim_handoff( + run_id, + version_before, + claim.owner_id, + claim, + ), + ) + self.assertEqual(self.store.run(run_id)["version"], version_before) + persisted = self.store.connection.execute( + "SELECT lease_expires_at,renewed_at FROM claims WHERE run_id=?", (run_id,), + ).fetchone() + self.assertEqual(tuple(persisted), (claim.expires_at, None)) + + def test_completion_checks_expiry_after_waiting_for_write_lock(self): + deadline = datetime(2026, 9, 15, 5, 2, tzinfo=timezone.utc) + run_id, claim = self.claim_with_deadline("completion-lock", deadline) + version_before = self.store.run(run_id)["version"] + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.run_after_independent_writer_wait( + deadline, + lambda: self.store.record_handoff_submission( + run_id, claim, self.decision(), + ), + ) + run = self.store.run(run_id) + self.assertEqual( + (run["state"], run["phase"], run["version"]), + ("awaiting_host", None, version_before + 1), + ) + rejection = self.store.connection.execute( + "SELECT outcome,rejection_code,recorded_run_version " + "FROM handoff_submissions WHERE handoff_id=?", + (claim.handoff_id,), + ).fetchone() + self.assertEqual( + tuple(rejection), ("rejected", "expired_claim", version_before + 1), + ) + + def test_expired_submission_is_audited_and_cannot_displace_takeover(self): + run_id, _, snapshot = self.waiting_run() + started = datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc) + old_claim = self.store.claim_handoff( + run_id, snapshot.run_version, "old-host", now=started, + ) + takeover = self.store.claim_handoff( + run_id, + old_claim.run_version, + "new-host", + now=started + timedelta(minutes=11), + ) + before = self.store.run(run_id)["version"] + with self.assertRaises(ConflictError): + self.store.record_handoff_submission( + run_id, + old_claim, + self.decision(), + now=started + timedelta(minutes=11), + ) + self.assertEqual(self.store.run(run_id)["version"], before + 1) + rejection = self.store.connection.execute( + "SELECT outcome,rejection_code FROM handoff_submissions WHERE handoff_id=?", + (snapshot.handoff_id,), + ).fetchone() + self.assertEqual(tuple(rejection), ("rejected", "stale_claim")) + current_claim = self.store.handoff_snapshot(run_id).claim + self.assertEqual(current_claim.owner_id, takeover.owner_id) + self.assertEqual(current_claim.fencing_token, takeover.fencing_token) + + def guided_expiry_recovery(self, key, *, guided=True): + run_id, _, snapshot = self.waiting_run(key) + started = datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc) + decision = self.decision(submission_id=f"terminal-{snapshot.handoff_id}") + options = {"initial_only": True, "terminal_decision": decision} if guided else {} + old = self.store.claim_handoff(run_id, snapshot.run_version, "terminal-operator", now=started, **options) + later = datetime.fromisoformat(old.expires_at) + timedelta(seconds=1) + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.store.record_handoff_submission(run_id, old, decision, now=later) + rejected = dict(self.store.connection.execute("SELECT * FROM handoff_submissions WHERE handoff_id=?", (old.handoff_id,)).fetchone()) + current = self.store.claim_handoff(run_id, self.store.run(run_id)["version"], "terminal-operator", now=later, **options) + return run_id, old, current, decision, later, rejected + + def handoff_rows(self, run_id): + return {table: [dict(row) for row in self.store.connection.execute(query, (run_id,))] for table, query in { + "runs": "SELECT * FROM runs WHERE id=?", + "claims": "SELECT * FROM claims WHERE run_id=?", + "handoffs": "SELECT * FROM handoffs WHERE run_id=?", + "submissions": "SELECT s.* FROM handoff_submissions s JOIN handoffs h ON h.id=s.handoff_id WHERE h.run_id=?", + "events": "SELECT * FROM events WHERE run_id=? ORDER BY id", + }.items()} + + def test_guided_expiry_audit_and_projection_commit_or_roll_back_together(self): + for interruption in ("before-audit", "after-audit", "after-projection"): + with self.subTest(interruption=interruption): + run_id, old, current, decision, later, rejected = self.guided_expiry_recovery(interruption) + before = self.handoff_rows(run_id) + connection = self.store.connection + class InterruptedConnection: + def execute(self, statement, *args, **kwargs): + audit = "'handoff.completion_recovered'" in statement + projection = statement.startswith("UPDATE handoff_submissions SET") + if audit and interruption == "before-audit": + raise KeyboardInterrupt + result = connection.execute(statement, *args, **kwargs) + if (audit and interruption == "after-audit") or (projection and interruption == "after-projection"): + raise KeyboardInterrupt + return result + def __getattr__(self, name): + return getattr(connection, name) + self.store.connection = InterruptedConnection() + try: + with self.assertRaises(KeyboardInterrupt): + self.store.record_handoff_submission(run_id, current, decision, now=later) + finally: + self.store.close() + self.store = Store(self.database, self.artifacts) + self.assertEqual(self.handoff_rows(run_id), before) + recorded = self.store.record_handoff_submission(run_id, current, decision, now=later) + self.assertFalse(recorded.replayed) + after = self.handoff_rows(run_id) + self.assertEqual(after["events"][:-2], before["events"]) + audit_event, submitted = after["events"][-2:] + self.assertEqual((audit_event["type"], submitted["type"]), ("handoff.completion_recovered", "handoff.submitted")) + payload = json.loads(audit_event["payload"]) + self.assertEqual(audit_event["payload"], canonical_json(payload)) + self.assertEqual(payload["rejected_submission"], rejected) + self.assertEqual(payload["rejected_submission_sha256"], request_hash(rejected)) + self.assertEqual(submitted["run_version"], audit_event["run_version"] + 1) + self.assertEqual(recorded.recorded_run_version, submitted["run_version"]) + self.assertEqual(after["runs"][0]["version"], submitted["run_version"]) + self.assertEqual(after["handoffs"][0]["submitted_run_version"], submitted["run_version"]) + self.assertEqual(after["submissions"][0]["recorded_run_version"], submitted["run_version"]) + replay = self.store.record_handoff_submission(run_id, old, decision, now=later + timedelta(hours=1)) + self.assertTrue(replay.replayed) + self.assertEqual(replay.recorded_run_version, recorded.recorded_run_version) + self.assertEqual(self.handoff_rows(run_id), after) + + def test_guided_expiry_recovery_refuses_malformed_prior_row_or_marker(self): + corruptions = { + "decision": ("decision_json", "{}"), + "evidence": ("evidence_refs_json", "[{}]"), + "owner": ("owner_id", "other-app"), + "disposition": ("disposition", "reject"), + "rejection": ("rejection_code", "stale_claim"), + "created": ("created_at", "not-a-time"), + "future-created": ("created_at", "2099-01-01T00:00:00+00:00"), + "same-fence": ("fencing_token", None), + "same-version": ("recorded_run_version", None), + "marker-structure": None, + "marker-packet": None, + } + for name, corruption in corruptions.items(): + with self.subTest(corruption=name): + run_id, _, current, decision, later, rejected = self.guided_expiry_recovery(name) + if corruption is None: + event = self.store.connection.execute("SELECT id,payload FROM events WHERE run_id=? AND type='handoff.taken_over' ORDER BY id DESC LIMIT 1", (run_id,)).fetchone() + payload = json.loads(event["payload"]) + if name == "marker-structure": + payload["terminal_finish"]["unexpected"] = True + else: + payload["terminal_finish"]["packet_sha256"] = "0" * 64 + self.store.connection.execute("UPDATE events SET payload=? WHERE id=?", (canonical_json(payload), event["id"])) + else: + field, value = corruption + if name == "same-fence": + value = current.fencing_token + elif name == "same-version": + value = current.run_version + self.store.connection.execute(f"UPDATE handoff_submissions SET {field}=? WHERE id=?", (value, rejected["id"])) + before = self.handoff_rows(run_id) + with self.assertRaises(ConflictError): + self.store.record_handoff_submission(run_id, current, decision, now=later) + self.assertEqual(self.handoff_rows(run_id), before) + + def test_guided_expiry_recovery_requires_current_marker_not_same_name_app(self): + for operation in ("ordinary-app", "renew", "takeover", "changed-intent"): + with self.subTest(operation=operation): + run_id, _, current, decision, later, _ = self.guided_expiry_recovery(operation, guided=operation != "ordinary-app") + if operation == "renew": + current = self.store.claim_handoff(run_id, current.run_version, "terminal-operator", current, now=later + timedelta(seconds=1)) + later += timedelta(seconds=1) + elif operation == "takeover": + later = datetime.fromisoformat(current.expires_at) + timedelta(seconds=1) + current = self.store.claim_handoff(run_id, current.run_version, "terminal-operator", now=later) + elif operation == "changed-intent": + decision = self.decision(submission_id=decision["submission_id"], reason="different exact intent") + before = self.handoff_rows(run_id) + with self.assertRaises(ConflictError): + self.store.record_handoff_submission(run_id, current, decision, now=later) + after = self.handoff_rows(run_id) + self.assertFalse(any(row["type"] == "handoff.completion_recovered" for row in after["events"])) + self.assertEqual(after["submissions"][0], before["submissions"][0]) + if operation != "changed-intent": + self.assertEqual(after, before) + self.assertIsNone(self.store.terminal_finish_decision(run_id, current.handoff_id)) + + def test_guided_expiry_recovery_requires_exact_original_rejection_event(self): + for corruption in ("missing", "type", "created", "row-created", "row-version", "run", "version", "noncanonical", "extra", "handoff", "submission-id", "hash", "reason"): + with self.subTest(corruption=corruption): + run_id, _, current, decision, later, rejected = self.guided_expiry_recovery(f"audit-{corruption}") + event = self.store.connection.execute("SELECT * FROM events WHERE run_id=? AND run_version=?", (run_id, rejected["recorded_run_version"])).fetchone() + if corruption == "missing": + self.store.connection.execute("DELETE FROM events WHERE id=?", (event["id"],)) + elif corruption == "type": + self.store.connection.execute("UPDATE events SET type='handoff.submitted' WHERE id=?", (event["id"],)) + elif corruption == "created": + self.store.connection.execute("UPDATE events SET created_at=? WHERE id=?", ((later + timedelta(seconds=1)).isoformat(), event["id"])) + elif corruption == "row-created": + self.store.connection.execute("UPDATE handoff_submissions SET created_at=? WHERE id=?", ((later - timedelta(seconds=1)).isoformat(), rejected["id"])) + elif corruption == "row-version": + self.store.connection.execute("UPDATE handoff_submissions SET recorded_run_version=? WHERE id=?", (rejected["recorded_run_version"] - 1, rejected["id"])) + elif corruption == "run": + other, _, _ = self.waiting_run("audit-other-run") + self.store.connection.execute("UPDATE events SET run_id=? WHERE id=?", (other, event["id"])) + elif corruption == "version": + self.store.connection.execute("UPDATE events SET run_version=? WHERE id=?", (current.run_version + 100, event["id"])) + elif corruption == "noncanonical": + self.store.connection.execute("UPDATE events SET payload=? WHERE id=?", (json.dumps(json.loads(event["payload"])), event["id"])) + else: + payload = json.loads(event["payload"]) + if corruption == "extra": + payload["extra"] = True + else: + field = {"handoff": "handoff_id", "submission-id": "submission_id", "hash": "submission_hash", "reason": "reason"}[corruption] + payload[field] = "different" + self.store.connection.execute("UPDATE events SET payload=? WHERE id=?", (canonical_json(payload), event["id"])) + before = self.handoff_rows(run_id) + with self.assertRaisesRegex(ConflictError, "rejection audit"): + self.store.record_handoff_submission(run_id, current, decision, now=later) + self.assertEqual(self.handoff_rows(run_id), before) + + def test_repeated_guided_expiry_retains_one_rejection_and_one_recovery(self): + run_id, _, current, decision, later, rejected = self.guided_expiry_recovery("repeated-expiry") + again = datetime.fromisoformat(current.expires_at) + timedelta(seconds=1) + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.store.record_handoff_submission(run_id, current, decision, now=again) + self.assertEqual(dict(self.store.connection.execute("SELECT * FROM handoff_submissions WHERE id=?", (rejected["id"],)).fetchone()), rejected) + latest = self.store.claim_handoff(run_id, self.store.run(run_id)["version"], "terminal-operator", now=again, initial_only=True, terminal_decision=decision) + self.assertGreater(latest.fencing_token, current.fencing_token) + self.store.record_handoff_submission(run_id, latest, decision, now=again) + rows = self.handoff_rows(run_id) + self.assertEqual(sum(row["type"] == "handoff.completion_rejected" for row in rows["events"]), 1) + self.assertEqual(sum(row["type"] == "handoff.completion_recovered" for row in rows["events"]), 1) + self.assertEqual(len(rows["submissions"]), 1) + + def test_submission_replays_and_terminal_late_rejection_preserves_run_and_events(self): + run_id, _, snapshot = self.waiting_run() + claimed_at = datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc) + claim = self.store.claim_handoff( + run_id, snapshot.run_version, "host", now=claimed_at, + ) + decision = self.decision() + recorded = self.store.record_handoff_submission( + run_id, claim, decision, now=claimed_at + timedelta(minutes=1), + ) + self.assertFalse(recorded.replayed) + waiting = self.store.run(run_id) + self.assertEqual((waiting["state"], waiting["phase"]), ("awaiting_host", "handoff_submitted")) + replay = self.store.record_handoff_submission( + run_id, claim, decision, now=claimed_at + timedelta(minutes=20), + ) + self.assertTrue(replay.replayed) + self.assertEqual(replay.recorded_run_version, recorded.recorded_run_version) + terminal_version = self.store.cancel_host_wait( + run_id, now=claimed_at + timedelta(minutes=21), + ) + replay_after_cancel = self.store.record_handoff_submission( + run_id, claim, decision, now=claimed_at + timedelta(minutes=22), + ) + self.assertTrue(replay_after_cancel.replayed) + self.assertEqual(self.store.run(run_id)["version"], terminal_version) + + terminal_before = self.store.connection.execute( + "SELECT * FROM runs WHERE id=?", (run_id,), + ).fetchone() + terminal_before = tuple(terminal_before) + events_before = [ + tuple(row) + for row in self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? ORDER BY id", (run_id,), + ) + ] + + conflicting = self.decision(reason="different body") + with self.assertRaises(ConflictError): + self.store.record_handoff_submission( + run_id, claim, conflicting, now=claimed_at + timedelta(minutes=23), + ) + terminal = self.store.run(run_id) + self.assertEqual(terminal["state"], "cancelled") + self.assertEqual(terminal["phase"], None) + self.assertEqual(terminal["version"], terminal_version) + terminal_after = tuple(self.store.connection.execute( + "SELECT * FROM runs WHERE id=?", (run_id,), + ).fetchone()) + events_after = [ + tuple(row) + for row in self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? ORDER BY id", (run_id,), + ) + ] + self.assertEqual(terminal_after, terminal_before) + self.assertEqual(events_after, events_before) + rejection = self.store.connection.execute( + "SELECT outcome,rejection_code,recorded_run_version FROM handoff_submissions " + "WHERE handoff_id=? AND submission_hash=?", + (snapshot.handoff_id, conflicting["submission_hash"]), + ).fetchone() + self.assertEqual( + tuple(rejection), ("rejected", "submission_id_reused", terminal_version), + ) + + with self.assertRaises(ConflictError): + self.store.record_handoff_submission( + run_id, claim, conflicting, now=claimed_at + timedelta(minutes=24), + ) + self.assertEqual(tuple(self.store.connection.execute( + "SELECT * FROM runs WHERE id=?", (run_id,), + ).fetchone()), terminal_before) + self.assertEqual([ + tuple(row) + for row in self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? ORDER BY id", (run_id,), + ) + ], events_before) + + def test_cancel_wait_invalidates_claim_and_publishes_terminal_receipt(self): + run_id, _, snapshot = self.waiting_run() + claim = self.store.claim_handoff( + run_id, snapshot.run_version, "host", + now=datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc), + ) + version = self.store.cancel_host_wait( + run_id, now=datetime(2026, 9, 15, 5, 2, tzinfo=timezone.utc), + ) + run = self.store.run(run_id) + self.assertEqual((run["state"], run["phase"], run["version"]), ("cancelled", None, version)) + persisted_claim = self.store.connection.execute( + "SELECT active,fencing_token FROM claims WHERE run_id=?", (run_id,), + ).fetchone() + self.assertEqual(tuple(persisted_claim), (0, claim.fencing_token + 1)) + receipt = self.store.artifact_named(run_id, "result-receipt.json") + self.assertIsNotNone(receipt) + receipt_payload = json.loads(Path(receipt["path"]).read_text()) + self.assertEqual( + (receipt_payload["state"], receipt_payload["phase"]), + ("cancelled", "awaiting_host"), + ) + self.assertEqual(self.store.cancel_host_wait(run_id), version) + + +class InstalledWheelHandoffMigrationTest(unittest.TestCase): + @staticmethod + def build_python(): + candidates = [ + os.environ.get("DEVSQUAD_BUILD_PYTHON"), + sys.executable, + str(Path.home() / ".cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3"), + shutil.which("python3.13"), + shutil.which("python3.12"), + shutil.which("python3.11"), + ] + for candidate in dict.fromkeys(value for value in candidates if value): + try: + result = subprocess.run( + [ + candidate, + "-c", + "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68", + ], + text=True, + capture_output=True, + ) + except OSError: + continue + if result.returncode == 0: + return candidate + return None + + def test_build_python_skips_missing_first_candidate_for_supported_interpreter(self): + missing = str(Path(tempfile.mkdtemp(prefix="devsquad-missing-")) / "no-such-python") + probed = [] + + def fake_run(argv, **kwargs): + probed.append(argv[0]) + if argv[0] == missing: + raise FileNotFoundError(argv[0]) + return subprocess.CompletedProcess(argv, 0) + + with ( + mock.patch.dict(os.environ, {"DEVSQUAD_BUILD_PYTHON": missing}), + mock.patch("subprocess.run", side_effect=fake_run), + ): + result = self.build_python() + self.assertEqual(probed[0], missing) + self.assertEqual(result, sys.executable) + + def test_installed_wheel_applies_schema_four_to_twelve(self): + build_python = self.build_python() + if build_python is None: + self.skipTest("offline wheel gate requires setuptools>=68 and wheel") + with tempfile.TemporaryDirectory(prefix="devsquad-handoff-wheel-") as directory: + root = Path(directory) + source = root / "core" + shutil.copytree(ROOT / "plugin/core", source) + wheels = root / "wheels" + wheels.mkdir() + subprocess.run( + [ + build_python, + "-m", + "pip", + "wheel", + str(source), + "--wheel-dir", + str(wheels), + "--no-index", + "--no-deps", + "--no-build-isolation", + ], + check=True, + text=True, + capture_output=True, + ) + wheel = next(wheels.glob("devsquad_core-*.whl")) + environment = os.environ.copy() + environment.pop("PYTHONPATH", None) + venv = root / "venv" + subprocess.run([build_python, "-m", "venv", str(venv)], check=True, env=environment) + python = venv / ("Scripts/python.exe" if os.name == "nt" else "bin/python") + subprocess.run( + [str(python), "-m", "pip", "install", "--no-index", "--no-deps", str(wheel)], + check=True, + text=True, + capture_output=True, + env=environment, + ) + probe = r''' +from importlib.resources import files +from pathlib import Path +import sqlite3 +import sys +from devsquad.store import Store, SUPPORTED_SCHEMA_VERSION + +root = Path(sys.argv[1]) +root.mkdir(parents=True) +database = root / "state.sqlite3" +connection = sqlite3.connect(database) +migrations = files("devsquad.migrations") +for version, name in ( + (1, "001_initial.sql"), + (2, "002_supervisor.sql"), + (3, "003_durable_io.sql"), + (4, "004_run_snapshot.sql"), +): + connection.executescript(migrations.joinpath(name).read_text()) + connection.execute( + "INSERT INTO schema_migrations(version, applied_at) VALUES(?, 'fixture')", (version,) + ) +connection.commit() +connection.close() +store = Store(database, root / "artifacts") +try: + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == SUPPORTED_SCHEMA_VERSION + assert store.connection.execute( + "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" + ).fetchone() + claim_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(claims)")} + assert {"handoff_id", "lease_expires_at", "renewed_at"} <= claim_columns + attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} + assert {"profile_id", "profile_index"} <= attempt_columns +finally: + store.close() +''' + subprocess.run( + [str(python), "-P", "-c", probe, str(root / "probe-runtime")], + check=True, + text=True, + capture_output=True, + cwd=root, + env=environment, + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_implementation_identity_evidence.py b/test/core/test_implementation_identity_evidence.py new file mode 100644 index 0000000..4840864 --- /dev/null +++ b/test/core/test_implementation_identity_evidence.py @@ -0,0 +1,225 @@ +"""Adversarial native evidence parsing and coordinator import validation.""" + +from __future__ import annotations + +import copy +import hashlib +import json +import unittest +from unittest.mock import patch + +import test_claude_identity as fixtures +from devsquad.claude_delivery_worker import run +from devsquad.claude_identity import ( + ClaudeResultError, failure_envelope, strict_json, validate_failure, +) +from devsquad.contracts import ContractError +from devsquad.supervisor import Supervisor +from devsquad.workflows import validate_implementation_evidence + + +class ImplementationIdentityEvidenceTest(unittest.TestCase): + def setUp(self): + self.fixture = fixtures.ClaudeImplementationIdentityTest(methodName="runTest") + self.fixture.setUp() + self.addCleanup(self.fixture.doCleanups) + + def evidence(self): + self.fixture.output.write_text(json.dumps(self.fixture.document())) + snapshot = self.fixture.snapshot() + return snapshot, run(snapshot) + + def test_strict_native_json_rejects_duplicates_and_nonfinite_numbers(self): + document = json.dumps(self.fixture.document()) + for payload in ( + document[:-1] + ', "session_id": "other"}', + document[:-1] + ', "unrelated": NaN}', + document[:-1] + ', "unrelated": Infinity}', + document[:-1] + ', "unrelated": 1e9999}', + ): + with self.subTest(payload=payload[-80:]): + self.fixture.output.write_text(payload) + with self.assertRaises(ClaudeResultError): + run(self.fixture.snapshot()) + + def test_optional_native_usage_metadata_is_typed_and_bounded(self): + for field, invalid in ( + ("costUSD", -1), ("costUSD", True), ("costUSD", 10 ** 400), + ("cacheReadInputTokens", -1), ("contextWindow", True), + ("canonicalModel", {}), ("provider", "other-provider"), + ("inputTokens", 2 ** 63), + ): + with self.subTest(field=field, invalid_type=type(invalid).__name__): + document = self.fixture.document() + document["modelUsage"][self.fixture.MODEL][field] = invalid + with self.assertRaises(ContractError): + self.fixture.run_document(document) + + def test_missing_aggregate_usage_stays_unknown_with_valid_identity(self): + document = self.fixture.document() + del document["usage"] + attempt = self.fixture.run_document(document)["attempt"] + self.assertEqual(attempt["usage"]["source"], "unavailable") + self.assertIsNone(attempt["usage"]["total_tokens"]) + self.assertEqual(attempt["observed_identity"]["verification_scope"], "reported_model") + self.assertIsNone(attempt["native_model_requests"]) + + def test_coordinator_rejects_tampering_before_candidate_finalization(self): + snapshot, evidence = self.evidence() + mutations = [ + (("schema_version",), 1), + (("attempt", "observed_identity"), {}), + (("attempt", "observed_identity", "model_id"), "claude-other"), + (("attempt", "observed_identity", "effort"), "high"), + (("attempt", "observed_identity", "backing_revision"), "invented"), + (("attempt", "observed_identity", "verification_scope"), "everything"), + (("attempt", "observed_identity", "model_source"), "requested_argv"), + (("attempt", "observed_identity", "harness_version"), "different"), + (("attempt", "observed_identity", "native_evidence"), None), + (("attempt", "observed_identity", "native_evidence", "schema_version"), True), + (("attempt", "observed_identity", "native_evidence", "is_error"), 0), + (("attempt", "observed_identity", "native_evidence", "model_usage"), {}), + (("attempt", "native_ids", "session_id"), "different-session"), + (("attempt", "usage", "input_tokens"), 999), + ] + for path, value in mutations: + with self.subTest(path=path): + tampered = copy.deepcopy(evidence) + target = tampered + for field in path[:-1]: + target = target[field] + target[path[-1]] = value + with self.assertRaises(ContractError): + validate_implementation_evidence(tampered, snapshot) + with patch("devsquad.supervisor.freeze_delivery_candidate") as freeze: + with self.assertRaises(ContractError): + # This import boundary validates before accessing its store. + Supervisor._commit_delivery_candidate( + object(), "unused", {}, [], {}, snapshot, + json.dumps(tampered).encode(), + ) + freeze.assert_not_called() + + def test_native_claim_cannot_downgrade_to_fixture_evidence(self): + snapshot, evidence = self.evidence() + evidence["schema_version"] = 1 + evidence["attempt"]["observed_identity"] = None + evidence["attempt"]["native_ids"] = {} + with self.assertRaises(ContractError): + validate_implementation_evidence(evidence, snapshot) + + def test_frozen_fallback_is_validated_against_its_own_profile(self): + snapshot, _ = self.evidence() + route = snapshot["routing"]["roles"]["implementer"] + fallback = copy.deepcopy(route["selected"]) + fallback["profile"]["id"] = fallback["profile_id"] = "fallback-writer" + fallback["reference"]["id"] = "fallback-writer" + fallback["profile"]["model_id"] = "sonnet" + fallback["profile_sha256"] = hashlib.sha256(json.dumps( + fallback["profile"], sort_keys=True, separators=(",", ":"), + ).encode()).hexdigest() + route["fallbacks"] = [fallback] + snapshot["implementation_adapters"]["fallback-writer"] = snapshot["implementation_adapter"] + launched = copy.deepcopy(snapshot) + launched["routing"]["roles"]["implementer"]["selected"] = fallback + evidence = run(launched) + self.assertEqual(validate_implementation_evidence(evidence, snapshot), evidence) + self.assertEqual(evidence["attempt"]["selected_profile"]["profile_id"], "fallback-writer") + + def test_failed_identity_keeps_only_typed_native_diagnostics(self): + document = self.fixture.document() + document["result"] = "private provider prose must not enter diagnostics" + document["modelUsage"]["claude-other-model"] = { + **self.fixture.model_usage(), "unrelated_secret": "do not retain", + } + with self.assertRaises(ClaudeResultError) as caught: + self.fixture.run_document(document) + diagnostics = caught.exception.diagnostics + self.assertEqual(len(diagnostics["model_usage"]), 2) + self.assertEqual(diagnostics["usage"]["total_tokens"], 19) + self.assertNotIn("private provider prose", json.dumps(diagnostics)) + self.assertNotIn("unrelated_secret", json.dumps(diagnostics)) + self.assertNotIn("do not retain", json.dumps(diagnostics)) + value = failure_envelope(caught.exception, "a" * 64) + self.assertEqual(validate_failure(value, "a" * 64), value) + with self.assertRaises(ContractError): + validate_failure(value, "b" * 64) + for field, invalid in (("identity_status", "verified"), + ("output_sha256", "bad"), ("output_bytes", True)): + altered = copy.deepcopy(value) + altered["native_diagnostics"][field] = invalid + with self.assertRaises(ContractError): + validate_failure(altered, "a" * 64) + + def test_import_decoder_rejects_duplicate_native_evidence_keys(self): + snapshot, evidence = self.evidence() + encoded = json.dumps(evidence) + encoded = encoded[:-1] + ', "schema_version": 2}' + with self.assertRaises(ContractError): + strict_json(encoded) + with patch("devsquad.supervisor.freeze_delivery_candidate") as freeze: + with self.assertRaises(ContractError): + Supervisor._commit_delivery_candidate( + object(), "unused", {}, [], {}, snapshot, encoded.encode(), + ) + freeze.assert_not_called() + + def test_import_must_match_actual_attempt_not_another_allowed_fallback(self): + snapshot, evidence = self.evidence() + route = snapshot["routing"]["roles"]["implementer"] + primary = route["selected"] + fallback = copy.deepcopy(primary) + fallback["profile"]["id"] = fallback["profile_id"] = "fallback-writer" + fallback["reference"]["id"] = "fallback-writer" + fallback["profile_sha256"] = hashlib.sha256(json.dumps( + fallback["profile"], sort_keys=True, separators=(",", ":"), + ).encode()).hexdigest() + route["fallbacks"] = [fallback] + snapshot["implementation_adapters"]["fallback-writer"] = snapshot["implementation_adapter"] + evidence["attempt"]["selected_profile"] = fallback + # It is valid evidence for this frozen fallback, but not for the actual + # primary invocation importing it. Validate the coordinator boundary. + validate_implementation_evidence(evidence, snapshot) + for attempt in ( + {"profile_index": 0, "profile_id": primary["profile_id"]}, + {"profile_index": 0, "profile_id": fallback["profile_id"]}, + {"profile_index": True, "profile_id": fallback["profile_id"]}, + ): + with self.subTest(attempt=attempt): + with patch("devsquad.supervisor.freeze_delivery_candidate", side_effect=AssertionError("unbound evidence reached freeze")) as freeze: + with self.assertRaises(ContractError): + Supervisor._commit_delivery_candidate( + object(), "unused", attempt, [], {}, snapshot, + json.dumps(evidence).encode(), + ) + freeze.assert_not_called() + + def test_invalid_utf8_stdout_retains_original_native_byte_hash(self): + payload = b"\xff\xfe\r\n" + self.fixture.output.write_bytes(payload) + with self.assertRaises(ClaudeResultError) as caught: + run(self.fixture.snapshot()) + diagnostics = caught.exception.diagnostics + self.assertEqual(diagnostics["output_bytes"], len(payload)) + self.assertEqual(diagnostics["output_sha256"], hashlib.sha256(payload).hexdigest()) + self.assertIsNone(diagnostics["session_id"]) + + def test_failed_result_hash_preserves_crlf_native_bytes(self): + document = self.fixture.document() + document["modelUsage"]["claude-other-model"] = self.fixture.model_usage() + payload = (json.dumps(document, indent=2) + "\n").replace("\n", "\r\n").encode() + self.fixture.output.write_bytes(payload) + with self.assertRaises(ClaudeResultError) as caught: + run(self.fixture.snapshot()) + self.assertEqual(caught.exception.diagnostics["output_bytes"], len(payload)) + self.assertEqual(caught.exception.diagnostics["output_sha256"], hashlib.sha256(payload).hexdigest()) + + def test_invalid_utf8_stderr_cannot_preempt_valid_result_parsing(self): + content = self.fixture.binary.read_text() + self.fixture.binary.write_text(content.replace("/bin/cat", "printf '\\377' >&2\n/bin/cat")) + evidence = self.fixture.run_document(self.fixture.document()) + self.assertEqual(evidence["attempt"]["observed_identity"]["model_id"], self.fixture.MODEL) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_install_core.py b/test/core/test_install_core.py new file mode 100644 index 0000000..65766a4 --- /dev/null +++ b/test/core/test_install_core.py @@ -0,0 +1,623 @@ +import json +import io +from contextlib import closing +import os +from pathlib import Path +import re +import shutil +import sqlite3 +import subprocess +import sys +import tempfile +import tarfile +import time +import unittest +import zipfile +from unittest.mock import patch + +from devsquad_test_fixtures import branch_review_routing_documents +from devsquad.store import SUPPORTED_SCHEMA_VERSION + + +ROOT = Path(__file__).resolve().parents[2] +CORE = ROOT / "plugin/core" +INSTALLER = ROOT / "scripts/install-core.sh" +COMPOSITE_INSTALLER = ROOT / "install.sh" + + +class StandaloneInstallerTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-install-") + self.root = Path(self.temp.name) + self.install_root = self.root / "installed" + self.bin_dir = self.root / "bin" + self.environment = os.environ.copy() + self.environment.update({ + "DEVSQUAD_INSTALL_ROOT": str(self.install_root), + "DEVSQUAD_BIN_DIR": str(self.bin_dir), + "DEVSQUAD_PYTHON": sys.executable, + }) + self.environment.pop("PYTHONPATH", None) + + def tearDown(self): + self.temp.cleanup() + + def install(self, source=CORE, *arguments): + result = subprocess.run( + [ + "/bin/bash", str(INSTALLER), + "--source-core", str(source), + "--json", *arguments, + ], + check=True, + text=True, + capture_output=True, + env=self.environment, + cwd=ROOT, + ) + self.assertEqual(result.stderr, "") + lines = result.stdout.splitlines() + self.assertEqual(len(lines), 1, result.stdout) + return json.loads(lines[0]) + + def changed_source(self): + destination = self.root / "core-v2" + shutil.copytree( + CORE, + destination, + ignore=shutil.ignore_patterns( + "__pycache__", "*.pyc", ".pytest_cache", "*.egg-info", + ), + ) + pyproject = destination / "pyproject.toml" + pyproject.write_text( + pyproject.read_text().replace('version = "0.1.0"', 'version = "0.1.1"') + ) + package = destination / "src/devsquad/__init__.py" + package.write_text( + package.read_text().replace('__version__ = "0.1.0"', '__version__ = "0.1.1"') + ) + return destination + + def mcp_wheelhouse(self): + wheelhouse = self.root / "mcp-wheelhouse" + wheelhouse.mkdir() + lock = CORE / "requirements-mcp.lock" + for raw_line in lock.read_text().splitlines(): + line = raw_line.strip() + if not line or line.startswith("#"): + continue + name, version = line.split("==", 1) + normalized = re.sub(r"[-_.]+", "_", name) + dist_info = f"{normalized}-{version}.dist-info" + wheel = wheelhouse / f"{normalized}-{version}-py3-none-any.whl" + with zipfile.ZipFile(wheel, "w", zipfile.ZIP_DEFLATED) as archive: + archive.writestr( + f"{dist_info}/METADATA", + "Metadata-Version: 2.1\n" + f"Name: {name}\n" + f"Version: {version}\n", + ) + archive.writestr( + f"{dist_info}/WHEEL", + "Wheel-Version: 1.0\n" + "Generator: devsquad-test\n" + "Root-Is-Purelib: true\n" + "Tag: py3-none-any\n", + ) + archive.writestr(f"{dist_info}/RECORD", "") + if name == "mcp": + archive.writestr("mcp/__init__.py", "") + return wheelhouse + + def test_fresh_install_requires_no_claude_and_reinstall_is_idempotent(self): + self.environment["PATH"] = "/usr/bin:/bin" + + first = self.install() + self.assertTrue(first["changed"]) + self.assertIsNone(first["error"]) + self.assertEqual(first["drift"], { + "plugin_installed": False, + "source_installed": False, + "source_plugin": False, + }) + self.assertFalse(first["installed"]["mcp"]) + self.assertTrue(first["installed_manifest_matches"]) + launcher = self.bin_dir / "squad" + self.assertTrue(os.access(launcher, os.X_OK)) + version = subprocess.run( + [str(launcher), "--version"], + check=True, + text=True, + capture_output=True, + env=self.environment, + ) + self.assertEqual(version.stdout.strip(), "squad 0.1.0") + self.assertEqual(version.stderr, "") + + release = Path(first["current_target"]) + self.assertTrue((release / "core/adapters/codex/adapter.json").is_file()) + self.assertTrue((release / "core/integrations/grok/registration.json").is_file()) + self.assertTrue((release / "venv/lib").is_dir()) + before_state = (self.install_root / "install-state.json").read_bytes() + before_launcher = launcher.read_bytes() + + second = self.install() + self.assertFalse(second["changed"]) + self.assertEqual(second["current_target"], first["current_target"]) + self.assertEqual((self.install_root / "install-state.json").read_bytes(), before_state) + self.assertEqual(launcher.read_bytes(), before_launcher) + self.assertEqual(len(list((self.install_root / "releases").iterdir())), 1) + + def test_offline_mcp_install_keeps_json_clean_and_passes_dependency_check(self): + report = self.install( + CORE, + "--with-mcp", + "--mcp-wheelhouse", + str(self.mcp_wheelhouse()), + ) + + self.assertTrue(report["changed"]) + self.assertTrue(report["installed"]["mcp"]) + release = Path(report["current_target"]) + checked = subprocess.run( + [str(release / "venv/bin/python"), "-m", "pip", "check"], + check=True, + text=True, + capture_output=True, + env=self.environment, + ) + self.assertEqual(checked.stdout.strip(), "No broken requirements found.") + self.assertEqual(checked.stderr, "") + + def test_composite_installer_succeeds_without_claude(self): + self.environment["PATH"] = "/usr/bin:/bin" + result = subprocess.run( + ["/bin/bash", str(COMPOSITE_INSTALLER), "--core-only", "--json"], + check=True, + text=True, + capture_output=True, + env=self.environment, + cwd=ROOT, + ) + report = json.loads(result.stdout) + self.assertTrue(report["changed"]) + self.assertIn("Claude plugin: skipped", result.stderr) + self.assertEqual( + subprocess.run( + [str(self.bin_dir / "squad"), "--version"], + check=True, + text=True, + capture_output=True, + env=self.environment, + ).stdout.strip(), + "squad 0.1.0", + ) + + def test_known_legacy_release_symlink_is_migrated_but_arbitrary_launcher_is_not(self): + legacy = self.install_root / "releases/0.1.0+legacy/venv/bin" + legacy.mkdir(parents=True) + legacy_squad = legacy / "squad" + legacy_squad.write_text("#!/bin/sh\nexit 0\n") + legacy_squad.chmod(0o755) + self.bin_dir.mkdir() + launcher = self.bin_dir / "squad" + launcher.symlink_to(legacy_squad) + + migrated = self.install() + self.assertTrue(migrated["changed"]) + self.assertFalse(launcher.is_symlink()) + self.assertIn("managed-by: devsquad-install-core-v1", launcher.read_text()) + self.assertTrue(legacy_squad.is_file()) + + other_root = self.root / "other-install" + other_bin = self.root / "other-bin" + other_bin.mkdir() + arbitrary = self.root / "arbitrary-squad" + arbitrary.write_text("#!/bin/sh\nexit 0\n") + arbitrary.chmod(0o755) + (other_bin / "squad").symlink_to(arbitrary) + environment = { + **self.environment, + "DEVSQUAD_INSTALL_ROOT": str(other_root), + "DEVSQUAD_BIN_DIR": str(other_bin), + } + refused = subprocess.run( + ["/bin/bash", str(INSTALLER), "--json"], + text=True, + capture_output=True, + env=environment, + cwd=ROOT, + ) + self.assertNotEqual(refused.returncode, 0) + self.assertIn("refusing to replace unmanaged launcher", refused.stderr) + self.assertTrue((other_bin / "squad").is_symlink()) + + def test_composite_claude_reinstall_uses_plugin_hooks_once(self): + fake_bin = self.root / "fake-bin" + fake_bin.mkdir() + fake_claude = fake_bin / "claude" + fake_claude.write_text("""#!/bin/sh +set -eu +printf '%s\\n' "$*" >> "$DEVSQUAD_FAKE_CLAUDE_LOG" +case "$*" in + "plugin marketplace list") + if [ -f "$DEVSQUAD_FAKE_CLAUDE_STATE/marketplace" ]; then echo devsquad-marketplace; fi ;; + "plugin marketplace add "*) touch "$DEVSQUAD_FAKE_CLAUDE_STATE/marketplace" ;; + "plugin marketplace update "*) : ;; + "plugin list") + if [ -f "$DEVSQUAD_FAKE_CLAUDE_STATE/plugin" ]; then echo devsquad@devsquad-marketplace; fi ;; + "plugin install "*) touch "$DEVSQUAD_FAKE_CLAUDE_STATE/plugin" ;; + "plugin update "*) : ;; + "plugin enable "*) : ;; + *) exit 64 ;; +esac +""") + fake_claude.chmod(0o755) + state = self.root / "fake-claude-state" + state.mkdir() + log = self.root / "fake-claude.log" + home = self.root / "home" + settings = home / ".claude/settings.json" + settings.parent.mkdir(parents=True) + original_settings = '{"hooks":{"sentinel":[]}}\n' + settings.write_text(original_settings) + environment = self.environment.copy() + environment.update({ + "PATH": f"{fake_bin}:/usr/bin:/bin", + "HOME": str(home), + "DEVSQUAD_FAKE_CLAUDE_LOG": str(log), + "DEVSQUAD_FAKE_CLAUDE_STATE": str(state), + }) + command = ["/bin/bash", str(COMPOSITE_INSTALLER), "--with-claude"] + first = subprocess.run( + command, text=True, capture_output=True, env=environment, cwd=ROOT, + ) + self.assertEqual(first.returncode, 0, (first.stdout, first.stderr)) + second = subprocess.run( + command, text=True, capture_output=True, env=environment, cwd=ROOT, + ) + self.assertEqual(second.returncode, 0, (second.stdout, second.stderr)) + self.assertIn("Claude plugin ready", first.stdout) + self.assertIn("Claude plugin ready", second.stdout) + self.assertEqual(settings.read_text(), original_settings) + calls = log.read_text().splitlines() + self.assertEqual(calls, [ + "plugin marketplace list", + "plugin marketplace add https://github.com/joshidikshant/devsquad.git", + "plugin list", + "plugin install devsquad@devsquad-marketplace", + "plugin enable devsquad@devsquad-marketplace", + "plugin marketplace list", + "plugin marketplace update devsquad-marketplace", + "plugin list", + "plugin update devsquad@devsquad-marketplace", + "plugin enable devsquad@devsquad-marketplace", + ]) + + def test_marketplace_package_source_contains_the_complete_core(self): + marketplace = json.loads((ROOT / ".claude-plugin/marketplace.json").read_text()) + entries = [item for item in marketplace["plugins"] if item["name"] == "devsquad"] + self.assertEqual(len(entries), 1) + plugin_root = (ROOT / entries[0]["source"]).resolve() + self.assertEqual(plugin_root, (ROOT / "plugin").resolve()) + required = { + "core/pyproject.toml", + "core/bin/squad", + "core/src/devsquad/cli.py", + "core/src/devsquad/migrations/013_decision_observations.sql", + "core/integrations/codex/registration.json", + "core/schemas/task.schema.json", + } + self.assertEqual( + {path for path in required if not (plugin_root / path).is_file()}, + set(), + ) + + def test_status_reports_drift_and_update_selects_a_new_immutable_release(self): + first = self.install() + old_release = Path(first["current_target"]) + old_manifest = (old_release / "release.json").read_bytes() + source = self.changed_source() + + status = self.install(source, "--status") + self.assertFalse(status["changed"]) + self.assertEqual(status["drift"], { + "plugin_installed": False, + "source_installed": True, + "source_plugin": True, + }) + + updated = self.install(source) + self.assertTrue(updated["changed"]) + self.assertNotEqual(updated["current_target"], first["current_target"]) + self.assertEqual(updated["installed"]["version"], "0.1.1") + self.assertEqual(updated["drift"], { + "plugin_installed": True, + "source_installed": False, + "source_plugin": True, + }) + self.assertTrue(old_release.is_dir()) + self.assertEqual((old_release / "release.json").read_bytes(), old_manifest) + self.assertEqual(len(list((self.install_root / "releases").iterdir())), 2) + version = subprocess.run( + [str(self.bin_dir / "squad"), "--version"], + check=True, + text=True, + capture_output=True, + env=self.environment, + ) + self.assertEqual(version.stdout.strip(), "squad 0.1.1") + + def test_update_does_not_break_an_active_release_pinned_run(self): + first = self.install() + old_release = Path(first["current_target"]) + old_python = old_release / "venv/bin/python" + repo = self.root / "repo" + runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + subprocess.run( + ["git", "-C", str(repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(repo), "config", "user.name", "Test"], + check=True, + ) + (repo / "src").mkdir() + (repo / "tests").mkdir() + (repo / "src/app.py").write_text("VALUE = 'base'\n") + (repo / "tests/test_app.py").write_text("# fixture test\n") + profiles, policy = branch_review_routing_documents() + (repo / "profiles.json").write_text(profiles) + (repo / "policy.json").write_text(policy) + subprocess.run(["git", "-C", str(repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(repo), "commit", "-qm", "base"], check=True) + task = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + task["project"] = { + "repo_path": str(repo), "base_ref": "HEAD", "target_ref": "HEAD", + } + task["routing"] = { + "profiles_file": "profiles.json", "policy_file": "policy.json", + } + task_path = self.root / "active-task.json" + task_path.write_text(json.dumps(task)) + start_script = """ +import json +from pathlib import Path +import sys +from devsquad.service import Service +task = json.loads(Path(sys.argv[1]).read_text()) +print(json.dumps(Service(Path(sys.argv[2])).start( + task, "installer-update", _internal_fake_delay=4, +))) +""" + started_process = subprocess.run( + [str(old_python), "-P", "-c", start_script, str(task_path), str(runtime)], + check=True, + text=True, + capture_output=True, + env=self.environment, + cwd=self.root, + ) + started = json.loads(started_process.stdout) + self.assertEqual(started["state"], "queued", started) + + launcher = self.bin_dir / "squad" + deadline = time.monotonic() + 8 + before_update = None + while time.monotonic() < deadline: + before_update = self.cli_json(launcher, "status", started["run_id"], "--runtime-dir", str(runtime)) + if before_update["data"]["state"] == "running": + break + time.sleep(0.05) + self.assertEqual(before_update["data"]["state"], "running", before_update) + + updated = self.install(self.changed_source()) + self.assertNotEqual(updated["current_target"], str(old_release)) + self.assertTrue(old_release.is_dir()) + + deadline = time.monotonic() + 10 + after_update = None + while time.monotonic() < deadline: + after_update = self.cli_json(launcher, "status", started["run_id"], "--runtime-dir", str(runtime)) + if after_update["data"]["state"] == "succeeded": + break + time.sleep(0.05) + self.assertEqual(after_update["data"]["state"], "succeeded", after_update) + result = self.cli_json( + launcher, "result", started["run_id"], "--runtime-dir", str(runtime), + ) + self.assertTrue(result["data"]["ready"]) + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + package_path, package_digest = connection.execute( + "SELECT package_path, package_digest FROM runs WHERE id=?", + (started["run_id"],), + ).fetchone() + self.assertTrue(Path(package_path).is_dir()) + self.assertEqual(len(package_digest), 64) + + def test_schema_13_update_defers_for_active_and_recoverable_old_runs(self): + """Exercise the historical package, not a same-schema version bump.""" + old_source = self.root / "schema13-core" + old_source.mkdir() + archived = subprocess.run(["git", "archive", "f4fa657:plugin/core"], cwd=ROOT, + check=True, capture_output=True).stdout + with tarfile.open(fileobj=io.BytesIO(archived)) as archive: + archive.extractall(old_source, filter="data") + first = self.install(old_source) + old_release = Path(first["current_target"]) + old_python = old_release / "venv/bin/python" + runtime = self.install_root / "runtime" + repo = self.root / "schema13-repo" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + subprocess.run(["git", "-C", str(repo), "config", "user.name", "Test"], check=True) + subprocess.run(["git", "-C", str(repo), "config", "user.email", "test@example.invalid"], check=True) + (repo / "README").write_text("baseline\n") + subprocess.run(["git", "-C", str(repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(repo), "commit", "-qm", "baseline"], check=True) + profiles, policy = branch_review_routing_documents() + task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) + task.update(project={"repo_path": str(repo), "base_ref": "HEAD", "target_ref": "HEAD"}, + scope={"read_paths": ["README"], "write_paths": []}, checks=[], + routing={"profiles": json.loads(profiles), "policy": json.loads(policy)}) + task_path = self.root / "schema13-task.json" + task_path.write_text(json.dumps(task)) + script = """ +import json,sys +from pathlib import Path +from unittest.mock import patch +from devsquad.service import Service +s=Service(Path(sys.argv[2])) +task=json.loads(Path(sys.argv[1]).read_text()) +live=s.start(task,'schema13-active',_internal_fake_delay=30) +with patch.object(s,'_spawn_daemon',return_value=0): + queued=s.start(task,'schema13-recoverable',_internal_fake_delay=0.1) +print(json.dumps([live,queued])) +""" + process = subprocess.run([str(old_python), "-P", "-c", script, str(task_path), str(runtime)], + check=True, capture_output=True, text=True, env=self.environment, cwd=self.root) + live, queued = json.loads(process.stdout) + launcher = self.bin_dir / "squad" + try: + deadline = time.monotonic() + 8 + while time.monotonic() < deadline: + status = self.cli_json(launcher, "status", live["run_id"], "--runtime-dir", str(runtime)) + if status["data"]["state"] == "running": + break + time.sleep(0.05) + self.assertEqual(status["data"]["state"], "running", status) + result = subprocess.run(["/bin/bash", str(INSTALLER), "--json"], capture_output=True, + text=True, env=self.environment, cwd=ROOT) + self.assertNotEqual(result.returncode, 0) + self.assertIn("active/recoverable", result.stderr) + self.assertEqual((self.install_root / "current").resolve(), old_release) + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 13) + self.cli_json(launcher, "cancel", live["run_id"], "--runtime-dir", str(runtime)) + deadline = time.monotonic() + 8 + while time.monotonic() < deadline: + state = self.cli_json(launcher, "status", live["run_id"], "--runtime-dir", str(runtime))["data"]["state"] + if state == "cancelled": + break + time.sleep(0.05) + self.assertEqual(state, "cancelled") + # A queued old-package run remains recoverable after the deferral. + self.cli_json(launcher, "resume", queued["run_id"], "--runtime-dir", str(runtime)) + deadline = time.monotonic() + 8 + while time.monotonic() < deadline: + state = self.cli_json(launcher, "status", queued["run_id"], "--runtime-dir", str(runtime))["data"]["state"] + if state == "succeeded": + break + time.sleep(0.05) + self.assertEqual(state, "succeeded") + updated = self.install() + self.assertNotEqual(updated["current_target"], str(old_release)) + self.assertTrue(old_release.is_dir()) + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 13) + self.assertTrue(self.cli_json(launcher, "result", queued["run_id"], "--runtime-dir", str(runtime))["data"]["ready"]) + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], SUPPORTED_SCHEMA_VERSION) + finally: + # Use the old package explicitly even if a later assertion fails. + cleanup_script = "from pathlib import Path; import sys; from devsquad.service import Service; s=Service(Path(sys.argv[1])); [s.cancel(r) for r in sys.argv[2:]]" + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + schema = connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] + if schema == 13: + subprocess.run([str(old_python), "-P", "-c", cleanup_script, str(runtime), live["run_id"], queued["run_id"]], + capture_output=True, env=self.environment, cwd=self.root) + + def test_failed_selector_activation_never_advances_the_old_ledger(self): + from devsquad.release_activation import activate_release + runtime = self.root / "activation-runtime" + runtime.mkdir() + database = runtime / "state.sqlite3" + with closing(sqlite3.connect(database)) as connection: + for migration in sorted((CORE / "src/devsquad/migrations").glob("*.sql")): + version = int(migration.name.split("_", 1)[0]) + if version > 13: + break + connection.executescript(migration.read_text()) + connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?, 'fixture')", (version,)) + connection.commit() + selector, temporary = self.root / "current", self.root / "prepared-selector" + selector.symlink_to("old-release") + temporary.symlink_to("new-release") + with patch("devsquad.release_activation.os.replace", side_effect=OSError("injected activation failure")): + with self.assertRaisesRegex(OSError, "injected activation failure"): + activate_release(temporary, selector, runtime, supported_schema_version=SUPPORTED_SCHEMA_VERSION) + self.assertEqual(os.readlink(selector), "old-release") + with closing(sqlite3.connect(database, timeout=1)) as connection: + connection.execute("BEGIN EXCLUSIVE") + self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 13) + connection.rollback() + + def test_preopened_schema13_store_cannot_admit_work_after_new_schema_commit(self): + from devsquad.store import Store + old_source = self.root / "preopened-schema13-core" + old_source.mkdir() + archived = subprocess.run(["git", "archive", "f4fa657:plugin/core"], cwd=ROOT, + check=True, capture_output=True).stdout + with tarfile.open(fileobj=io.BytesIO(archived)) as archive: + archive.extractall(old_source, filter="data") + first = self.install(old_source) + old_python = Path(first["current_target"]) / "venv/bin/python" + runtime = self.install_root / "runtime" + repo = self.root / "preopened-repo" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + script = """ +import json,sys +from pathlib import Path +from devsquad.store import Store,git_common_dir +s=Store(Path(sys.argv[1])/'state.sqlite3',Path(sys.argv[1])/'artifacts') +s._project(git_common_dir(Path(sys.argv[2]))) +print('ready',flush=True) +sys.stdin.readline() +try: + s.claim_start(Path(sys.argv[2]),'late-old-client',{'task':'bounded'},'old-owner') +except Exception as exc: + print(json.dumps({'error':str(exc)}),flush=True) +else: + print(json.dumps({'unsafe_admission':True}),flush=True) +finally: + s.close() +""" + process = subprocess.Popen([str(old_python), "-P", "-c", script, str(runtime), str(repo)], + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + text=True, env=self.environment, cwd=self.root) + try: + self.assertEqual(process.stdout.readline().strip(), "ready") + self.install() + store = Store(runtime / "state.sqlite3", runtime / "artifacts") + try: + self.assertEqual(store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], SUPPORTED_SCHEMA_VERSION) + finally: + store.close() + output, stderr = process.communicate("admit\n", timeout=8) + self.assertEqual(process.returncode, 0, stderr) + self.assertIn("devsquad_connection_schema", json.loads(output)["error"]) + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + self.assertEqual(connection.execute("SELECT COUNT(*) FROM runs").fetchone()[0], 0) + finally: + if process.poll() is None: + process.terminate() + process.communicate(timeout=5) + + def cli_json(self, launcher, *arguments): + result = subprocess.run( + [str(launcher), *arguments, "--json"], + check=True, + text=True, + capture_output=True, + env=self.environment, + cwd=self.root, + ) + self.assertEqual(result.stderr, "") + return json.loads(result.stdout) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_installed_usability.py b/test/core/test_installed_usability.py new file mode 100644 index 0000000..ce72bf6 --- /dev/null +++ b/test/core/test_installed_usability.py @@ -0,0 +1,182 @@ +"""Fresh installed normal commands, with only provider binaries replaced.""" + +import hashlib +import json +import os +from pathlib import Path +import re +import shutil +import subprocess +import sys +import tempfile +import time +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +import test_cli as cli_fixtures + + +class InstalledNormalUsabilityTest(unittest.TestCase): + def setUp(self): + python = cli_fixtures.InstalledWheelMigrationTest.build_python() + if python is None: + self.skipTest("offline supported build interpreter is unavailable") + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-installed-ux-") + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.home = self.root / "home" + self.home.mkdir() + self.provider_bin = self.root / "providers" + self.provider_bin.mkdir() + for name, target in ( + ("codex", ROOT / "test/core/fakes/codex_review_cli.py"), + ("claude", ROOT / "test/core/fakes/claude_delivery_cli.py"), + ("python3", Path(python)), + ("git", Path(shutil.which("git"))), + ): + (self.provider_bin / name).symlink_to(target) + codex_home = self.home / ".codex" + codex_home.mkdir() + (codex_home / "auth.json").write_text("{}\n") + (codex_home / "auth.json").chmod(0o600) + claude_home = self.home / ".claude" + claude_home.mkdir() + (claude_home / ".credentials.json").write_text("{}\n") + (claude_home / ".credentials.json").chmod(0o600) + self.runtime = self.root / "runtime" + self.environment = { + "HOME": str(self.home), "USER": "devsquad-offline-test", + "PATH": f"{self.provider_bin}{os.pathsep}/usr/bin{os.pathsep}/bin", + "CODEX_HOME": str(codex_home), "PYTHONWARNINGS": "error::ResourceWarning", + "PYTHONDONTWRITEBYTECODE": "1", + "PIP_NO_INDEX": "1", "PIP_DISABLE_PIP_VERSION_CHECK": "1", + "DEVSQUAD_PYTHON": python, "DEVSQUAD_RUNTIME_DIR": str(self.runtime), + "DEVSQUAD_INSTALL_ROOT": str(self.root / "installed"), + "DEVSQUAD_BIN_DIR": str(self.root / "bin"), + } + installed = subprocess.run( + ["/bin/bash", str(ROOT / "scripts/install-core.sh"), "--source-core", str(ROOT / "plugin/core"), "--json"], + cwd=self.root, env=self.environment, text=True, capture_output=True, timeout=60, + ) + self.assertEqual((installed.returncode, installed.stderr), (0, ""), installed.stdout) + self.installation = json.loads(installed.stdout) + self.assertTrue(self.installation["changed"]) + self.assertFalse(self.installation["installed"]["mcp"]) + self.launcher = self.root / "bin/squad" + self.run_ids = set() + self.addCleanup(self.cancel_unfinished_runs) + + def git(self, repo, *arguments): + return subprocess.run([str(self.provider_bin / "git"), "-C", str(repo), *arguments], + env=self.environment, text=True, capture_output=True, check=True).stdout + + def project(self, name, *, defective): + repo = self.root / name + repo.mkdir() + self.git(repo, "init", "-qb", "main") + self.git(repo, "config", "user.name", "Offline Test") + self.git(repo, "config", "user.email", "test@example.invalid") + (repo / "src").mkdir() + (repo / "tests").mkdir() + (repo / "src/app.py").write_text(f"def add(a, b):\n return a {'-' if defective else '+'} b\n") + (repo / "tests/test_app.py").write_text( + "import unittest\nfrom src.app import add\n" + "class AdditionTest(unittest.TestCase):\n" + " def test_addition(self):\n self.assertEqual(add(2, 3), 5)\n" + ) + self.git(repo, "add", ".") + self.git(repo, "commit", "-qm", "seeded fixture") + return repo + + def command(self, repo, *arguments, expected=0): + result = subprocess.run([str(self.launcher), *arguments], cwd=repo, env=self.environment, + text=True, capture_output=True, timeout=60) + self.run_ids.update(re.findall(r"^Run: ([a-f0-9-]{36})$", result.stdout, re.MULTILINE)) + evidence = result.stdout + if result.returncode != expected: + for run_id in re.findall(r"^Run: ([a-f0-9-]{36})$", result.stdout, re.MULTILINE): + report = subprocess.run([str(self.launcher), "result", run_id, "--json"], cwd=repo, + env=self.environment, text=True, capture_output=True, timeout=15) + if report.returncode == 0: + for artifact in json.loads(report.stdout)["data"]["artifacts"]: + if artifact["name"] == "receipt.json": + evidence += "\n" + Path(artifact["path"]).read_text() + self.assertEqual((result.returncode, result.stderr), (expected, ""), evidence) + return result.stdout + + def receipt(self, repo, run_id): + result = json.loads(self.command(repo, "result", run_id, "--json"))["data"] + self.assertTrue(result["ready"]) + self.assertEqual(result["state"], "succeeded") + for artifact in result["artifacts"]: + self.assertEqual(hashlib.sha256(Path(artifact["path"]).read_bytes()).hexdigest(), artifact["sha256"]) + reference = next(item for item in result["artifacts"] if item["name"] == "receipt.json") + return json.loads(Path(reference["path"]).read_text()) + + def cancel_unfinished_runs(self): + for run_id in self.run_ids: + subprocess.run([str(self.launcher), "cancel", run_id, "--json"], cwd=self.root, + env=self.environment, text=True, capture_output=True, timeout=15) + deadline = time.monotonic() + 10 + while time.monotonic() < deadline: + result = subprocess.run([str(self.launcher), "status", run_id, "--json"], cwd=self.root, + env=self.environment, text=True, capture_output=True, timeout=5) + if result.returncode == 0 and json.loads(result.stdout)["data"]["state"] in {"succeeded", "failed", "cancelled"}: + break + time.sleep(0.05) + else: + self.fail(f"installed fixture cleanup did not terminalize {run_id}") + + def test_fresh_installed_review_and_fix_finish_using_normal_commands(self): + review_repo = self.project("review-project", defective=False) + output = self.command(review_repo, "status", expected=64) + self.assertIn("No saved runs", output) + output = self.command(self.root, "review", "--project-dir", str(review_repo), "--base", "main", "--wait", expected=2) + self.assertIn("Review: clean", output) + self.assertIn("squad finish", output) + review_id = json.loads(self.command(review_repo, "status", "--json"))["data"]["run_id"] + self.command(review_repo, "finish", "--accept", "--reason", "Inspected the exact saved review.") + self.assertIn("receipt.md", self.command(review_repo, "result")) + review_receipt = self.receipt(review_repo, review_id) + self.assertEqual(review_receipt["accounting"]["worker_invocations"], 1) + + fix_repo = self.project("fix-project", defective=True) + self.assertIn("No saved runs", self.command(fix_repo, "status", expected=64)) + original_oid = self.git(fix_repo, "rev-parse", "HEAD") + original_source = (fix_repo / "src/app.py").read_bytes() + seeded_failure = subprocess.run([str(self.provider_bin / "python3"), "-m", "unittest", "discover", "-s", "tests"], + cwd=fix_repo, env=self.environment, capture_output=True, text=True) + self.assertEqual(seeded_failure.returncode, 1) + output = self.command(self.root, "fix", "Correct add so it returns the sum.", "--project-dir", str(fix_repo), "--write-path", "src/app.py", "--wait", expected=2) + self.assertIn("Check detected-tests: passed", output) + fix_id = json.loads(self.command(fix_repo, "status", "--json"))["data"]["run_id"] + self.assertNotEqual(fix_id, review_id) + self.command(fix_repo, "finish", "--accept", "--reason", "Seeded addition test and independent review pass.") + receipt = self.receipt(fix_repo, fix_id) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + identities = [item["observed_identity"] for item in receipt["attempts"]] + self.assertEqual([item["harness"] for item in identities], ["claude", "codex"]) + self.assertNotEqual(identities[0]["model_id"], identities[1]["model_id"]) + self.assertEqual(identities[0]["verification_scope"], "reported_model") + self.assertEqual(identities[0]["native_evidence"]["writer_messages"]["message_count"], 1) + self.assertTrue(all(item["required_to_pass"] and item["status"] == "passed" for item in receipt["checks"])) + self.assertTrue(all(item["integrity"]["status"] == "verified" for item in receipt["checks"])) + self.assertEqual(self.git(fix_repo, "rev-parse", "HEAD"), original_oid) + self.assertEqual((fix_repo / "src/app.py").read_bytes(), original_source) + self.assertEqual(self.git(fix_repo, "status", "--porcelain"), "") + + # Omitted IDs stay project-scoped even when newer unrelated runs exist. + self.assertEqual(json.loads(self.command(review_repo, "status", "--json"))["data"]["run_id"], review_id) + self.command(review_repo, "review", "--base", "main", "--idempotency-key", "installed-second-review", "--wait", expected=2) + output = self.command(review_repo, "status", expected=75) + self.assertIn("Multiple saved runs", output) + self.assertIn(review_id, output) + self.assertNotIn(fix_id, output) + output = self.command(review_repo, "result", expected=75) + self.assertIn("specify RUN", output) + self.assertTrue(json.loads(self.command(review_repo, "result", review_id, "--json"))["data"]["ready"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_jev_decision_probe.py b/test/core/test_jev_decision_probe.py new file mode 100644 index 0000000..2b469ca --- /dev/null +++ b/test/core/test_jev_decision_probe.py @@ -0,0 +1,236 @@ +from __future__ import annotations + +import importlib.util +import contextlib +import io +import json +import math +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from unittest import mock + + +ROOT = Path(__file__).resolve().parents[2] +PROBE = ROOT / "test" / "core" / "probes" / "jev_decision_eval.py" +SPEC = ( + ROOT + / "docs" + / "plans" + / "engineering-team" + / "experiments" + / "jev-pilot-v1.json" +) +MODULE_SPEC = importlib.util.spec_from_file_location("jev_decision_eval", PROBE) +assert MODULE_SPEC and MODULE_SPEC.loader +jev = importlib.util.module_from_spec(MODULE_SPEC) +MODULE_SPEC.loader.exec_module(jev) + + +class JevDecisionProbeTest(unittest.TestCase): + def setUp(self): + self.spec = jev.load_spec(SPEC) + self.request = jev.build_request(self.spec) + + def valid_response(self): + answers = {} + purpose_map = {item["id"]: item for item in self.spec["purposes"]} + for case in self.spec["cases"]: + for purpose_id, purpose in purpose_map.items(): + expected = case["expected"][purpose_id] + labels = list(purpose["criteria"]) + remaining = (1.0 - 0.85) / (len(labels) - 1) + probabilities = { + label: 0.85 if label == expected else remaining + for label in labels + } + answers[f"{case['id']}__{purpose_id}"] = { + "type": "choice", + "choice": expected, + "confidence": 0.8, + "probabilities": probabilities, + } + return { + "model": "jev-1.13.0", + "answers": answers, + "usage": {"input_tokens": 2100, "output_tokens": 420}, + } + + def test_frozen_probe_is_one_request_and_dry_run_needs_no_key(self): + expected_questions = len(self.spec["cases"]) * len(self.spec["purposes"]) + self.assertEqual(len(self.request["questions"]), expected_questions) + completed = subprocess.run( + [sys.executable, str(PROBE), "--spec", str(SPEC)], + check=True, + capture_output=True, + text=True, + ) + dry_run = json.loads(completed.stdout) + self.assertEqual(dry_run["billable_requests"], 1) + self.assertEqual(dry_run["retries"], 0) + self.assertLessEqual( + dry_run["documented_max_request_cost_usd"], + dry_run["cost_ceiling_usd"], + ) + + def test_valid_response_is_redacted_and_scored(self): + validated = jev.validate_response( + self.spec, self.request, self.valid_response() + ) + summary = jev.summarize( + self.spec, validated, elapsed_ms=123, request_sha256="a" * 64 + ) + self.assertEqual(summary["billable_requests"], 1) + self.assertEqual(summary["per_purpose"]["task_family"]["accuracy_percent"], 100.0) + self.assertNotIn("Correct two spelling mistakes", json.dumps(summary)) + self.assertTrue(math.isclose(summary["pricing"]["estimated_cost_usd"], 0.0000882)) + + def test_unknown_choice_and_non_finite_probability_are_rejected(self): + response = self.valid_response() + key = next(iter(response["answers"])) + response["answers"][key]["choice"] = "not-a-label" + with self.assertRaisesRegex(jev.ProbeError, "unknown choice"): + jev.validate_response(self.spec, self.request, response) + + response = self.valid_response() + key = next(iter(response["answers"])) + label = next(iter(response["answers"][key]["probabilities"])) + response["answers"][key]["probabilities"][label] = math.nan + with self.assertRaisesRegex(jev.ProbeError, "probabilities are invalid"): + jev.validate_response(self.spec, self.request, response) + + def test_live_mode_requires_output_before_network(self): + completed = subprocess.run( + [sys.executable, str(PROBE), "--spec", str(SPEC), "--execute"], + capture_output=True, + text=True, + ) + self.assertEqual(completed.returncode, 2) + self.assertIn("--output is required", completed.stderr) + + def test_live_result_cannot_be_written_into_the_repository(self): + completed = subprocess.run( + [ + sys.executable, + str(PROBE), + "--spec", + str(SPEC), + "--execute", + "--output", + str(ROOT / "jev-result.json"), + ], + capture_output=True, + text=True, + ) + self.assertEqual(completed.returncode, 2) + self.assertIn("outside the Git repository", completed.stderr) + + +class JevEnvFileTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.addCleanup(self.temporary.cleanup) + self.env_file = Path(self.temporary.name) / ".env" + environment = mock.patch.dict(os.environ, {}, clear=True) + environment.start() + self.addCleanup(environment.stop) + + def write_env(self, text): + self.env_file.write_text(text, encoding="utf-8") + self.env_file.chmod(0o600) + + def test_explicit_env_file_supports_plain_and_quoted_key(self): + for declaration in ( + "TYPESAFE_API_KEY=fake-test-key", + "TYPESAFE_API_KEY='fake-test-key'", + 'export TYPESAFE_API_KEY="fake-test-key"', + ): + with self.subTest(declaration=declaration): + self.write_env("# Local credentials\nUNRELATED=ignored\n" + declaration + "\n") + self.assertEqual(jev.load_api_key(self.env_file), "fake-test-key") + self.assertNotIn("TYPESAFE_API_KEY", os.environ) + + def test_exported_key_wins_without_reading_env_file(self): + with mock.patch.dict(os.environ, {"TYPESAFE_API_KEY": "exported-test-key"}): + self.assertEqual(jev.load_api_key(self.env_file), "exported-test-key") + + def test_no_implicit_env_loading_and_blank_template_has_no_key(self): + self.write_env("TYPESAFE_API_KEY=local-test-key\n") + with mock.patch.object(jev.os, "open", side_effect=AssertionError("implicit read")): + self.assertIsNone(jev.load_api_key()) + self.write_env("TYPESAFE_API_KEY=\n") + self.assertIsNone(jev.load_api_key(self.env_file)) + + def test_values_are_literal_not_shell_expanded(self): + self.write_env("TYPESAFE_API_KEY='$UNDEFINED_KEY'\n") + self.assertEqual(jev.load_api_key(self.env_file), "$UNDEFINED_KEY") + + def test_malformed_and_duplicate_key_errors_never_echo_contents(self): + for contents in ( + "TYPESAFE_API_KEY='private-test-secret\n", + "TYPESAFE_API_KEY=private-test-secret with-spaces\n", + "TYPESAFE_API_KEY=private-test-secret\nTYPESAFE_API_KEY=second-key\n", + ): + with self.subTest(): + self.write_env(contents) + with self.assertRaises(jev.ProbeError) as caught: + jev.load_api_key(self.env_file) + self.assertNotIn("private-test-secret", str(caught.exception)) + + def test_missing_insecure_symlink_and_oversized_files_fail_safely(self): + with self.assertRaises(jev.ProbeError): + jev.load_api_key(self.env_file) + self.write_env("TYPESAFE_API_KEY=private-test-secret\n") + self.env_file.chmod(0o644) + with self.assertRaisesRegex(jev.ProbeError, "permissions"): + jev.load_api_key(self.env_file) + self.env_file.chmod(0o600) + link = Path(self.temporary.name) / "linked.env" + link.symlink_to(self.env_file) + with self.assertRaises(jev.ProbeError): + jev.load_api_key(link) + self.write_env("#" * (jev.MAX_ENV_BYTES + 1)) + with self.assertRaisesRegex(jev.ProbeError, "size"): + jev.load_api_key(self.env_file) + + def test_env_file_dry_run_only_reports_presence_and_never_calls_api(self): + for value, configured in (("", False), ("private-test-secret", True)): + with self.subTest(configured=configured): + self.write_env(f"TYPESAFE_API_KEY={value}\n") + stdout, stderr = io.StringIO(), io.StringIO() + with mock.patch.object(jev, "urlopen") as network, \ + contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): + result = jev.main(["--env-file", str(self.env_file)]) + self.assertEqual(result, 0, stderr.getvalue()) + self.assertEqual(json.loads(stdout.getvalue())["api_key_configured"], configured) + self.assertNotIn("private-test-secret", stdout.getvalue() + stderr.getvalue()) + network.assert_not_called() + + def test_blank_key_blocks_execution_before_network(self): + self.write_env("TYPESAFE_API_KEY=\n") + spec = jev.load_spec(SPEC) + with mock.patch.object(jev, "urlopen") as network: + with self.assertRaisesRegex(jev.ProbeError, "not set"): + jev.execute(spec, jev.build_request(spec), env_file=self.env_file) + network.assert_not_called() + + def test_explicit_key_is_used_for_exactly_one_mocked_request(self): + self.write_env("TYPESAFE_API_KEY=private-test-secret\n") + spec = jev.load_spec(SPEC) + fixture = JevDecisionProbeTest() + fixture.setUp() + response = io.BytesIO(json.dumps(fixture.valid_response()).encode()) + with mock.patch.object(jev, "urlopen", return_value=response) as network: + result = jev.execute(spec, jev.build_request(spec), env_file=self.env_file) + network.assert_called_once() + request = network.call_args.args[0] + self.assertEqual(request.get_header("Authorization"), "Bearer private-test-secret") + self.assertNotIn("private-test-secret", json.dumps(result)) + self.assertNotIn("TYPESAFE_API_KEY", os.environ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_learning.py b/test/core/test_learning.py new file mode 100644 index 0000000..8aac274 --- /dev/null +++ b/test/core/test_learning.py @@ -0,0 +1,412 @@ +import copy +from datetime import datetime, timedelta, timezone +import hashlib +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.learning import ( + build_comparison_report, + build_learning_proposal, + evaluate_experiment, + render_learning_proposal_markdown, + validate_experiment, + validate_outcome, +) +from devsquad.store import ConflictError, Store, canonical_json, SUPPORTED_SCHEMA_VERSION + + +NOW = datetime(2026, 9, 27, 16, 0, tzinfo=timezone.utc) + + +def final_outcome(): + return { + "schema_version": 1, + "outcome_id": "outcome-final", + "kind": "final", + "verdict": "succeeded", + "selection_mode": "automatic", + "observed_at": NOW.isoformat(), + "corrects_outcome_id": None, + "summary": "A repair attempt produced the accepted result.", + "criteria": [{ + "criterion_id": "checks", + "status": "passed", + "evidence_refs": ["receipt.json#criteria/checks"], + }], + "contributions": [ + { + "attempt_id": "attempt-original", + "role": "implementer", + "result": "failed", + "independent_success": False, + "evidence_refs": ["attempt-original.json"], + }, + { + "attempt_id": "attempt-repair", + "role": "implementer", + "result": "repair", + "independent_success": False, + "evidence_refs": ["attempt-repair.json"], + }, + ], + "lead_repairs": [], + "evidence_refs": ["receipt.json"], + } + + +def experiment(project_path): + return { + "schema_version": 1, + "experiment_id": "experiment-profile-b", + "project_path": str(project_path), + "question": "Does profile B improve successful outcomes?", + "hypothesis": "Profile B is non-inferior and improves paired success.", + "evidence_availability": "tracked_fixture", + "variable": { + "kind": "profile_binding", + "alias": "review.deep", + "control_profile_id": "profile-a", + "candidate_profile_id": "profile-b", + }, + "cases": [ + { + "case_id": "eval-1", "split": "evaluation", + "control_outcome_id": "control-eval-1", + "candidate_outcome_id": "candidate-eval-1", + }, + { + "case_id": "eval-2", "split": "evaluation", + "control_outcome_id": "control-eval-2", + "candidate_outcome_id": "candidate-eval-2", + }, + { + "case_id": "hold-1", "split": "held_out", + "control_outcome_id": "control-hold-1", + "candidate_outcome_id": "candidate-hold-1", + }, + ], + "gate": { + "min_evaluation_pairs": 2, + "min_held_out_pairs": 1, + "noninferiority_margin": 0.0, + "minimum_success_gain": 0.5, + "max_candidate_escaped_defects": 0, + }, + "budget": { + "max_cases": 3, + "max_worker_invocations": 0, + "wall_seconds": 60, + }, + "rollback_target": {"profile_id": "profile-a", "binding_version": 7}, + } + + +def experimental_final(outcome_id, verdict): + value = final_outcome() + value.update({ + "outcome_id": outcome_id, + "verdict": verdict, + "selection_mode": "experimental", + "summary": f"Experimental fixture {outcome_id} was {verdict}.", + "criteria": [], + "contributions": [], + }) + return value + + +class LearningContractTest(unittest.TestCase): + def runtime_fixture(self): + from experiment_runtime_fixture import ExperimentRuntimeFixture + + temporary = tempfile.TemporaryDirectory(prefix="devsquad-learning-public-") + self.addCleanup(temporary.cleanup) + fixture = ExperimentRuntimeFixture( + Path(temporary.name), case_splits=[ + ("eval-1", "evaluation"), ("eval-2", "evaluation"), ("hold-1", "held_out"), + ], + ) + self.addCleanup(fixture.close) + # Retain the original two-evaluation/one-held-out acceptance threshold. + fixture.spec["gate"].update(min_evaluation_pairs=2, minimum_success_gain=0.5) + fixture.run_all() + return fixture + + def test_outcome_contract_rejects_false_success_and_mutation(self): + normalized = validate_outcome(final_outcome(), now=NOW) + self.assertEqual(normalized["verdict"], "succeeded") + changed = final_outcome() + changed["criteria"][0]["status"] = "failed" + with self.assertRaisesRegex(ContractError, "must all pass"): + validate_outcome(changed, now=NOW) + changed = final_outcome() + changed["contributions"][0]["independent_success"] = True + with self.assertRaisesRegex(ContractError, "only a successful"): + validate_outcome(changed, now=NOW) + changed = final_outcome() + changed["extra"] = True + with self.assertRaisesRegex(ContractError, "fields invalid"): + validate_outcome(changed, now=NOW) + changed = final_outcome() + changed["observed_at"] = (NOW + timedelta(minutes=6)).isoformat() + with self.assertRaisesRegex(ContractError, "clock skew"): + validate_outcome(changed, now=NOW) + + def test_experiment_gate_promotes_only_complete_held_out_evidence(self): + spec = experiment(Path("/tmp/experiment-project")) + validate_experiment(spec) + chains = {} + for case in spec["cases"]: + chains[case["control_outcome_id"]] = { + "final": experimental_final(case["control_outcome_id"], "failed"), + "late_corrections": [], + } + chains[case["candidate_outcome_id"]] = { + "final": experimental_final(case["candidate_outcome_id"], "succeeded"), + "late_corrections": [], + } + promoted = evaluate_experiment(spec, chains, evaluated_at=NOW.isoformat()) + self.assertEqual(promoted["verdict"], "promotion_proposal") + self.assertFalse(promoted["active_policy_changed"]) + self.assertEqual( + promoted["rollback_target"], + {"profile_id": "profile-a", "binding_version": 7}, + ) + + missing = dict(chains) + del missing["candidate-hold-1"] + no_change = evaluate_experiment(spec, missing, evaluated_at=NOW.isoformat()) + self.assertEqual(no_change["verdict"], "no_change") + self.assertIn("insufficient_held_out_pairs", no_change["reasons"]) + + escaped = copy.deepcopy(chains) + escaped["candidate-hold-1"]["late_corrections"] = [{ + "verdict": "escaped_defect", + }] + no_change = evaluate_experiment(spec, escaped, evaluated_at=NOW.isoformat()) + self.assertEqual(no_change["verdict"], "no_change") + self.assertIn( + "candidate_escaped_defect_limit_exceeded", no_change["reasons"], + ) + + invalid = experiment(Path("/tmp/experiment-project")) + invalid["rollback_target"]["profile_id"] = "profile-b" + with self.assertRaisesRegex(ContractError, "control profile"): + validate_experiment(invalid) + invalid = experiment(Path("/tmp/experiment-project")) + invalid["evidence_availability"] = [] + with self.assertRaisesRegex(ContractError, "evidence_availability"): + validate_experiment(invalid) + invalid = experiment(Path("/tmp/experiment-project")) + invalid["cases"][0]["split"] = [] + with self.assertRaisesRegex(ContractError, "case split"): + validate_experiment(invalid) + invalid_chains = copy.deepcopy(chains) + invalid_chains["candidate-hold-1"]["late_corrections"] = [None] + with self.assertRaisesRegex(ContractError, "late corrections"): + evaluate_experiment(spec, invalid_chains, evaluated_at=NOW.isoformat()) + + def test_experiment_evaluation_is_persisted_and_replay_safe(self): + fixture = self.runtime_fixture() + first = fixture.service.policy_evaluate(fixture.spec) + replay = fixture.service.policy_evaluate(fixture.spec) + self.assertEqual(first["evaluation"]["verdict"], "promotion_proposal") + self.assertFalse(first["replayed"]) + self.assertTrue(replay["replayed"]) + self.assertTrue(first["eligibility"]["eligible"]) + changed = copy.deepcopy(fixture.spec) + changed["hypothesis"] = "Mutated after evaluation." + with self.assertRaisesRegex(ConflictError, "different specification"): + fixture.service.policy_evaluate(changed) + store = fixture.store() + self.addCleanup(store.close) + inputs = store.learning_proposal_inputs(fixture.repo) + self.assertEqual(inputs["experiment"]["experiment"]["experiment_id"], fixture.spec["experiment_id"]) + + def test_learning_proposal_is_traceable_and_never_changes_policy(self): + project_path = "/tmp/experiment-project" + report = build_comparison_report( + project_id=None, + project_path=project_path, + terminal_runs=[], + outcome_records=[], + attempt_profiles={}, + generated_at=NOW.isoformat(), + ) + no_evidence = build_learning_proposal( + report, None, generated_at=NOW.isoformat(), + ) + self.assertEqual(no_evidence["verdict"], "no_change") + self.assertFalse(no_evidence["active_policy_changed"]) + self.assertEqual(no_evidence["reasons"], ["no_evaluated_experiment"]) + self.assertEqual( + no_evidence["decision"], + {"action": "retain_current_policy", "review_required": False}, + ) + + fixture = self.runtime_fixture() + evaluation = fixture.service.policy_evaluate(fixture.spec) + evaluation_sha256 = evaluation["evaluation_sha256"] + store = fixture.store() + self.addCleanup(store.close) + inputs = store.learning_proposal_inputs(fixture.repo) + report, record = inputs["report"], inputs["experiment"] + proposal = build_learning_proposal( + report, record, generated_at=datetime.now(timezone.utc).isoformat(), + ) + self.assertEqual(proposal["verdict"], "promotion_proposal") + self.assertFalse(proposal["active_policy_changed"]) + self.assertEqual(proposal["rollback_target"]["profile_id"], "profile-a") + self.assertEqual( + proposal["evidence"]["experiment"]["evaluation_sha256"], + evaluation_sha256, + ) + self.assertIn(proposal["proposal_id"], render_learning_proposal_markdown(proposal)) + tampered = copy.deepcopy(record) + tampered["evaluation_sha256"] = "0" * 64 + with self.assertRaisesRegex(ContractError, "evidence hash"): + build_learning_proposal(report, tampered, generated_at=NOW.isoformat()) + + def test_final_and_late_outcomes_are_append_only_and_attempt_bound(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + repository = path / "repo" + subprocess.run(["git", "init", "-q", str(repository)], check=True) + subprocess.run( + ["git", "-C", str(repository), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(repository), "config", "user.name", "Test"], + check=True, + ) + (repository / "README").write_text("fixture\n") + subprocess.run(["git", "-C", str(repository), "add", "README"], check=True) + subprocess.run(["git", "-C", str(repository), "commit", "-qm", "base"], check=True) + store = Store(path / "state.sqlite3", path / "artifacts") + self.addCleanup(store.close) + claim = store.claim_start(repository, "outcome-run", {}, "owner") + run = store.run(claim.run_id) + store.connection.execute( + "UPDATE runs SET state='succeeded',phase=NULL WHERE id=?", + (claim.run_id,), + ) + for attempt_id, profile_id, metadata in ( + ("attempt-original", "profile-original", {"failure": {"error": "seeded"}}), + ("attempt-repair", "profile-repair", {}), + ): + store.connection.execute( + "INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token," + "status,heartbeat_at,package_digest,output_metadata,created_at,finished_at," + "role,profile_id) VALUES(?,?,?,?,?,'finished',?,?,?,?,?,'implementer',?)", + ( + attempt_id, claim.run_id, run["project_id"], str(repository), + f"token-{attempt_id}", NOW.isoformat(), "package", + json.dumps(metadata), NOW.isoformat(), NOW.isoformat(), profile_id, + ), + ) + + false_credit = final_outcome() + false_credit["contributions"][0]["result"] = "successful" + false_credit["contributions"][0]["independent_success"] = True + with self.assertRaisesRegex(ConflictError, "failed attempt"): + store.record_outcome(claim.run_id, false_credit, now=NOW) + + first = store.record_outcome(claim.run_id, final_outcome(), now=NOW) + replay = store.record_outcome(claim.run_id, final_outcome(), now=NOW) + self.assertFalse(first["replayed"]) + self.assertTrue(replay["replayed"]) + changed = copy.deepcopy(final_outcome()) + changed["summary"] = "Changed after persistence." + with self.assertRaisesRegex(ConflictError, "different evidence"): + store.record_outcome(claim.run_id, changed, now=NOW) + + late = { + "schema_version": 1, + "outcome_id": "outcome-late-defect", + "kind": "late_correction", + "verdict": "escaped_defect", + "selection_mode": "automatic", + "observed_at": (NOW + timedelta(minutes=1)).isoformat(), + "corrects_outcome_id": "outcome-final", + "summary": "A defect escaped the original acceptance evidence.", + "criteria": [{ + "criterion_id": "checks", "status": "failed", + "evidence_refs": ["late-defect.json"], + }], + "contributions": [], + "lead_repairs": [], + "evidence_refs": ["late-defect.json"], + } + predating = copy.deepcopy(late) + predating["observed_at"] = (NOW - timedelta(seconds=1)).isoformat() + with self.assertRaisesRegex(ConflictError, "predates"): + store.record_outcome(claim.run_id, predating, now=NOW) + store.record_outcome( + claim.run_id, late, now=NOW + timedelta(minutes=1), + ) + history = store.outcomes_for_run(claim.run_id) + self.assertEqual( + [item["outcome"]["verdict"] for item in history], + ["succeeded", "escaped_defect"], + ) + self.assertFalse( + history[0]["outcome"]["contributions"][0]["independent_success"], + ) + missing = store.claim_start(repository, "missing-outcome", {}, "owner") + store.connection.execute( + "UPDATE runs SET state='failed',phase=NULL WHERE id=?", + (missing.run_id,), + ) + report = store.learning_report( + repository, now=NOW + timedelta(minutes=2), + ) + self.assertEqual(report["sample_size"], 1) + self.assertEqual(report["terminal_run_count"], 2) + self.assertEqual(report["final_successes"], 1) + self.assertEqual(report["escaped_defects"], 1) + self.assertEqual( + report["selection_modes"]["automatic"]["success_rate"], 1.0, + ) + self.assertEqual(report["profiles"]["profile-original"]["failed"], 1) + self.assertEqual(report["profiles"]["profile-repair"]["repairs"], 1) + self.assertEqual( + report["missingness"]["terminal_runs_without_final_outcome"], 1, + ) + self.assertTrue( + report["interpretation"]["final_task_success_is_not_profile_success"], + ) + + def test_current_schema_contains_outcome_ledger(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + store = Store(path / "state.sqlite3", path / "artifacts") + self.addCleanup(store.close) + self.assertEqual( + store.connection.execute( + "SELECT MAX(version) FROM schema_migrations", + ).fetchone()[0], + SUPPORTED_SCHEMA_VERSION, + ) + columns = { + row[1] for row in store.connection.execute("PRAGMA table_info(outcomes)") + } + self.assertTrue({"outcome_id", "payload_sha256", "corrects_outcome_id"} <= columns) + experiment_columns = { + row[1] + for row in store.connection.execute("PRAGMA table_info(experiments)") + } + self.assertTrue({ + "experiment_id", "spec_sha256", "evaluation_sha256", "verdict", + } <= experiment_columns) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py new file mode 100644 index 0000000..67dd901 --- /dev/null +++ b/test/core/test_lifecycle.py @@ -0,0 +1,569 @@ +import copy +from datetime import datetime, timezone +import hashlib +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import threading +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.catalog import update_last_good +from devsquad.lifecycle import ( + guarded_change_violation, + profile_template_violation, + validate_profile_template, +) +from devsquad.router import load_routing +from devsquad.service import Service +from devsquad.store import ConflictError, Store + + +NOW = datetime.now(timezone.utc) + + +def profile(profile_id, model_id, *, tools=None, pool="pool-a", billing="subscription"): + return { + "id": profile_id, + "harness": "fixture", + "model_family": "fixture-family", + "model_id": model_id, + "effort": {"value": "high", "transport": "native"}, + "required_tools": list(tools or ["read"]), + "permission_policy": "read_only", + "account_pool_id": pool, + "billing_mode": billing, + "quality_status": "proven", + "evidence_refs": ["tracked-fixture"], + } + + +def lifecycle_template(*, update_mode="guarded_auto"): + return { + "schema_version": 1, + "template_id": "template-review-deep-v1", + "alias": "review.deep", + "update_mode": update_mode, + "policy": {"id": "fixture-policy", "version": 3}, + "allowed_harnesses": ["fixture"], + "allowed_model_families": ["fixture-family"], + "allowed_account_pools": ["pool-a"], + "allowed_task_classes": ["fixture-review-small"], + "permission_policy": "read_only", + "allowed_tools": ["read", "web"], + "allowed_billing_modes": ["subscription"], + "gate": { + "min_evaluation_pairs": 1, + "min_held_out_pairs": 1, + "max_critical_defects": 0, + "max_latency_ratio": None, + "max_usage_ratio": None, + }, + } + + +def final_outcome(outcome_id, verdict): + return { + "schema_version": 1, + "outcome_id": outcome_id, + "kind": "final", + "verdict": verdict, + "selection_mode": "experimental", + "observed_at": NOW.isoformat(), + "corrects_outcome_id": None, + "summary": f"Lifecycle fixture {outcome_id} was {verdict}.", + "criteria": [], + "contributions": [], + "lead_repairs": [], + "evidence_refs": [f"{outcome_id}.json"], + } + + +def routing_policy(): + return { + "schema_version": 1, + "id": "fixture-policy", + "version": 3, + "roles": {"reviewer": [{"kind": "alias", "id": "review.deep"}]}, + "task_classes": {"fixture-review-small": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": False, + "account_pools": { + "pool-a": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 2, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + + +def review_task(repo, *, pinned_profile_id=None): + routing = { + "profiles_file": "profiles.json", + "policy_file": "policy.json", + } + if pinned_profile_id is not None: + routing["overrides"] = { + "reviewer": {"profile_id": pinned_profile_id, "fallback": "none"}, + } + return { + "schema_version": 1, + "project": { + "repo_path": str(repo), "base_ref": "HEAD", "target_ref": "HEAD", + }, + "workflow": "branch-review", + "goal": "Verify lifecycle routing.", + "task_class": "fixture-review-small", + "acceptance": [{ + "id": "routing", "description": "The expected profile is frozen.", + "evidence_kind": "review", + }], + "checks": [], + "scope": {"read_paths": ["README"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": routing, + "budget": { + "wall_seconds": 60, "max_worker_invocations": 1, + "max_revisions": 0, "max_fallbacks_per_step": 0, + }, + "origin": {"surface": "test"}, + } + + +class ProfileLifecycleTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-lifecycle-") + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], + check=True, + ) + (self.repo / "README").write_text("fixture\n") + subprocess.run(["git", "-C", str(self.repo), "add", "README"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.database = self.root / "state.sqlite3" + self.artifacts = self.root / "artifacts" + self.store = Store(self.database, self.artifacts) + self.incumbent = profile("profile-a", "model-a") + self.candidate = profile("profile-b", "model-b") + self.runtime_fixtures = [] + + def tearDown(self): + for fixture in self.runtime_fixtures: + fixture.close() + self.store.close() + self.temp.cleanup() + + def seed_experiment( + self, + *, + experiment_id="experiment-profile-b", + candidate_succeeds=True, + ): + # Positive authority comes from real workers and public completion, + # never SQL-terminalized empty runs or manually imported final outcomes. + from experiment_runtime_fixture import ExperimentRuntimeFixture + + fixture = ExperimentRuntimeFixture( + self.root / experiment_id, service=Service(self.root), repo=self.repo, + experiment_id=experiment_id, candidate_succeeds=candidate_succeeds, + ) + self.runtime_fixtures.append(fixture) + fixture.run_all() + return fixture.service.policy_evaluate(fixture.spec) + + def qualification(self, evaluation, *, qualification_id="qualification-b"): + return { + "schema_version": 1, + "qualification_id": qualification_id, + "alias": "review.deep", + "template_id": "template-review-deep-v1", + "candidate_profile": self.candidate, + "task_class": "fixture-review-small", + "experiment_id": evaluation["experiment"]["experiment_id"], + "evaluation_sha256": evaluation["evaluation_sha256"], + "source": { + "harness_version": "fixture 1.0", + "catalog_sha256": "a" * 64, + "model_revision": None, + }, + "budget": { + "max_cases": 2, + "used_cases": 2, + "max_worker_invocations": 4, + "worker_invocations": 4, + "max_wall_seconds": 60, + "wall_seconds": 5, + }, + "measured": { + "evaluation_pairs": 1, + "held_out_pairs": 1, + "critical_defects": 0, + "latency_ratio": None, + "usage_ratio": None, + }, + "verdict": "qualified", + "evidence_refs": ["experiment-profile-b", "evaluation.json"], + } + + @staticmethod + def promotion(decision_id, *, actor="human", expected=7): + return { + "schema_version": 1, + "decision_id": decision_id, + "action": "promote", + "alias": "review.deep", + "expected_binding_version": expected, + "qualification_id": "qualification-b", + "rollback_target": None, + "experiment_id": None, + "evaluation_sha256": None, + "actor": actor, + "reason": "Held-out evidence passed the reviewed gate.", + "evidence_refs": ["experiment-profile-b", "evaluation.json"], + } + + def test_template_and_guarded_authority_boundaries_are_strict(self): + template = validate_profile_template(lifecycle_template()) + self.assertIsNone(profile_template_violation(self.candidate, template)) + paid = profile("paid", "model-paid", billing="paid_api") + self.assertEqual( + profile_template_violation(paid, template), + "billing_mode_not_allowed", + ) + expanded = profile("expanded", "model-expanded", tools=["read", "web"]) + self.assertEqual( + guarded_change_violation(self.incumbent, expanded), + "guarded_tool_expansion", + ) + invalid = lifecycle_template() + invalid["gate"]["min_held_out_pairs"] = 0 + with self.assertRaisesRegex(ContractError, "min_held_out_pairs"): + validate_profile_template(invalid) + + def test_qualification_cas_promotion_and_rollback_are_replay_safe(self): + template = lifecycle_template() + registered = self.store.register_profile_template(template) + replayed = self.store.register_profile_template(template) + self.assertFalse(registered["replayed"]) + self.assertTrue(replayed["replayed"]) + baseline = self.store.bootstrap_profile_binding( + template, self.incumbent, version=7, + ) + self.assertFalse(baseline["replayed"]) + self.assertTrue(self.store.bootstrap_profile_binding( + template, self.incumbent, version=7, + )["replayed"]) + registry = { + "schema_version": 1, + "profiles": [self.incumbent], + "bindings": { + "review.deep": {"profile_id": "profile-a", "version": 7}, + }, + } + profiles_payload = json.dumps(registry, sort_keys=True) + "\n" + policy_payload = json.dumps(routing_policy(), sort_keys=True) + "\n" + frozen_input = self.store.effective_profile_registry( + profiles_payload, policy_payload, + ) + frozen_routing = load_routing( + review_task(self.repo), frozen_input["profiles_payload"], policy_payload, + ) + self.assertEqual( + frozen_routing["roles"]["reviewer"]["selected"]["profile_id"], + "profile-a", + ) + evaluation = self.seed_experiment() + qualified = self.store.record_profile_qualification( + self.qualification(evaluation), + ) + self.assertEqual(qualified["gate_failures"], []) + self.assertTrue(self.store.record_profile_qualification( + self.qualification(evaluation), + )["replayed"]) + + barrier = threading.Barrier(2) + results = [] + + def promote(decision_id): + connection = Store(self.database, self.artifacts) + try: + barrier.wait(timeout=10) + results.append(connection.change_profile_binding( + self.promotion(decision_id), + )) + except Exception as exc: + results.append(exc) + finally: + connection.close() + + threads = [ + threading.Thread(target=promote, args=(decision_id,)) + for decision_id in ("decision-promote-a", "decision-promote-b") + ] + for thread in threads: + thread.start() + for thread in threads: + thread.join() + successes = [item for item in results if isinstance(item, dict)] + conflicts = [item for item in results if isinstance(item, ConflictError)] + self.assertEqual((len(successes), len(conflicts)), (1, 1), results) + receipt = successes[0]["receipt"] + self.assertEqual(receipt["to"]["binding_version"], 8) + self.assertTrue(receipt["affects_new_runs_only"]) + self.assertEqual(self.store.profile_binding("review.deep")["profile_id"], "profile-b") + promoted_input = self.store.effective_profile_registry( + profiles_payload, policy_payload, + ) + promoted_routing = load_routing( + review_task(self.repo), promoted_input["profiles_payload"], policy_payload, + ) + self.assertEqual( + promoted_routing["roles"]["reviewer"]["selected"]["profile_id"], + "profile-b", + ) + self.assertEqual( + frozen_routing["roles"]["reviewer"]["selected"]["profile_id"], + "profile-a", + ) + pinned = load_routing( + review_task(self.repo, pinned_profile_id="profile-a"), + promoted_input["profiles_payload"], policy_payload, + ) + self.assertEqual( + pinned["roles"]["reviewer"]["selected"]["profile_id"], + "profile-a", + ) + winning_request = self.promotion(receipt["decision_id"]) + self.assertTrue(self.store.change_profile_binding( + winning_request, + )["replayed"]) + + regression = self.seed_experiment( + experiment_id="experiment-profile-b-regression", + candidate_succeeds=False, + ) + self.assertEqual(regression["evaluation"]["verdict"], "no_change") + rollback = { + "schema_version": 1, + "decision_id": "decision-rollback-a", + "action": "rollback", + "alias": "review.deep", + "expected_binding_version": 8, + "qualification_id": None, + "rollback_target": {"profile_id": "profile-a", "binding_version": 7}, + "experiment_id": "experiment-profile-b-regression", + "evaluation_sha256": regression["evaluation_sha256"], + "actor": "guarded_auto", + "reason": "Held-out regression requires the qualified predecessor.", + "evidence_refs": ["experiment-profile-b-regression", "regression.json"], + } + rolled_back = self.store.change_profile_binding(rollback) + self.assertEqual(rolled_back["receipt"]["to"]["binding_version"], 9) + self.assertEqual(self.store.profile_binding("review.deep")["profile_id"], "profile-a") + self.assertEqual(len(self.store.profile_binding_decisions("review.deep")), 2) + rollback_input = self.store.effective_profile_registry( + profiles_payload, policy_payload, + ) + rollback_routing = load_routing( + review_task(self.repo), rollback_input["profiles_payload"], policy_payload, + ) + self.assertEqual( + rollback_routing["roles"]["reviewer"]["selected"]["binding"]["version"], + 9, + ) + service = Service(self.root) + service_change = service.profile_binding_change( + self.promotion("decision-service", expected=9), + ) + self.assertEqual(service_change["receipt"]["to"]["binding_version"], 10) + for artifact in service_change["artifacts"].values(): + content = Path(artifact["path"]).read_bytes() + self.assertEqual(hashlib.sha256(content).hexdigest(), artifact["sha256"]) + status = service.profile_binding_status("review.deep") + self.assertEqual(status["binding"]["profile_id"], "profile-b") + self.assertEqual(len(status["decisions"]), 3) + + def test_insufficient_evidence_and_disabled_guarded_auto_cannot_promote(self): + reviewed = lifecycle_template(update_mode="reviewed") + self.store.bootstrap_profile_binding( + reviewed, self.incumbent, version=7, + ) + incomplete = self.qualification({ + "experiment": {"experiment_id": "unused"}, + "evaluation_sha256": "0" * 64, + }, qualification_id="qualification-incomplete") + incomplete.update({ + "experiment_id": None, + "evaluation_sha256": None, + "verdict": "incomplete", + "evidence_refs": [], + }) + saved = self.store.record_profile_qualification(incomplete) + self.assertIn("experiment_evidence_missing", saved["gate_failures"]) + with self.assertRaisesRegex(ContractError, "qualified candidate"): + self.store.change_profile_binding( + self.promotion("decision-insufficient"), + ) + + evaluation = self.seed_experiment() + self.store.record_profile_qualification( + self.qualification(evaluation), + ) + with self.assertRaisesRegex(ContractError, "not enabled"): + self.store.change_profile_binding( + self.promotion("decision-auto", actor="guarded_auto"), + ) + + def test_catalog_unavailable_incumbent_uses_only_qualified_predecessor(self): + template = lifecycle_template() + self.store.bootstrap_profile_binding( + template, self.incumbent, version=7, + ) + evaluation = self.seed_experiment() + self.store.record_profile_qualification( + self.qualification(evaluation), + ) + self.store.change_profile_binding( + self.promotion("decision-promote-catalog"), + ) + registry = { + "schema_version": 1, + "profiles": [self.incumbent], + "bindings": { + "review.deep": {"profile_id": "profile-a", "version": 7}, + }, + } + profiles_payload = json.dumps(registry, sort_keys=True) + "\n" + policy_payload = json.dumps(routing_policy(), sort_keys=True) + "\n" + frozen_before = self.store.effective_profile_registry( + profiles_payload, policy_payload, + ) + self.assertEqual( + load_routing( + review_task(self.repo), frozen_before["profiles_payload"], + policy_payload, + )["roles"]["reviewer"]["selected"]["profile_id"], + "profile-b", + ) + + catalog_path = self.root / "catalog-fallback.json" + update_last_good( + catalog_path, harness="fixture", version="1", complete=True, + models=[{"id": "model-a"}, {"id": "model-b"}], + profiles=[self.incumbent, self.candidate], + ) + catalog_change = update_last_good( + catalog_path, harness="fixture", version="1", complete=True, + models=[{"id": "model-a"}], + profiles=[self.incumbent, self.candidate], + )["catalog_change"] + request = { + "schema_version": 1, + "decision_id": "decision-catalog-fallback", + "action": "rollback", + "alias": "review.deep", + "expected_binding_version": 8, + "catalog_change": catalog_change, + "actor": "guarded_auto", + "reason": "The complete catalog removed the active model.", + "evidence_refs": ["catalog-fallback.json"], + } + service = Service(self.root) + fallback = service.profile_binding_fallback(request) + receipt = fallback["receipt"] + self.assertEqual(receipt["from"]["profile_id"], "profile-b") + self.assertEqual(receipt["to"]["profile_id"], "profile-a") + self.assertEqual(receipt["to"]["binding_version"], 9) + self.assertEqual( + receipt["rollback_evaluation"]["kind"], "catalog_unavailable", + ) + self.assertTrue(receipt["affects_new_runs_only"]) + self.assertTrue(service.profile_binding_fallback(request)["replayed"]) + for artifact in fallback["artifacts"].values(): + content = Path(artifact["path"]).read_bytes() + self.assertEqual(hashlib.sha256(content).hexdigest(), artifact["sha256"]) + + current = self.store.effective_profile_registry( + profiles_payload, policy_payload, + ) + self.assertEqual( + load_routing( + review_task(self.repo), current["profiles_payload"], + policy_payload, + )["roles"]["reviewer"]["selected"]["profile_id"], + "profile-a", + ) + self.assertEqual( + load_routing( + review_task(self.repo), frozen_before["profiles_payload"], + policy_payload, + )["roles"]["reviewer"]["selected"]["profile_id"], + "profile-b", + ) + + self.store.change_profile_binding( + self.promotion("decision-repromote-catalog", expected=9), + ) + all_removed_path = self.root / "catalog-all-removed.json" + update_last_good( + all_removed_path, harness="fixture", version="1", complete=True, + models=[{"id": "model-a"}, {"id": "model-b"}], + profiles=[self.incumbent, self.candidate], + ) + all_removed = update_last_good( + all_removed_path, harness="fixture", version="1", complete=True, + models=[], profiles=[self.incumbent, self.candidate], + )["catalog_change"] + blocked = { + **request, + "decision_id": "decision-catalog-blocked", + "expected_binding_version": 10, + "catalog_change": all_removed, + } + with self.assertRaisesRegex(ContractError, "no available qualified predecessor"): + self.store.fallback_unavailable_profile_binding(blocked) + self.assertEqual( + self.store.profile_binding("review.deep")["profile_id"], "profile-b", + ) + + def test_bootstrap_requires_a_proven_baseline(self): + trial = copy.deepcopy(self.incumbent) + trial["quality_status"] = "trial" + with self.assertRaisesRegex(ContractError, "already be proven"): + self.store.bootstrap_profile_binding( + lifecycle_template(), trial, version=1, + ) + + def test_schema_twelve_contains_lifecycle_ledger(self): + version = self.store.connection.execute( + "SELECT MAX(version) FROM schema_migrations", + ).fetchone()[0] + from devsquad.store import SUPPORTED_SCHEMA_VERSION + self.assertEqual(version, SUPPORTED_SCHEMA_VERSION) + tables = { + row[0] for row in self.store.connection.execute( + "SELECT name FROM sqlite_master WHERE type='table'", + ) + } + self.assertTrue({ + "profile_templates", "concrete_profiles", "qualification_runs", + "profile_bindings", "profile_binding_versions", "binding_decisions", + } <= tables) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_m1.py b/test/core/test_m1.py new file mode 100644 index 0000000..4e39195 --- /dev/null +++ b/test/core/test_m1.py @@ -0,0 +1,551 @@ +from __future__ import annotations + +import json +import os +import stat +import tempfile +import subprocess +import unittest +from dataclasses import replace +from pathlib import Path +from unittest.mock import patch + +CORE = Path(__file__).resolve().parents[2] / "plugin" / "core" +import sys +sys.path.insert(0, str(CORE / "src")) + +from devsquad.adapters import AdapterManifest, classify_cli, prepare_cli, prepare_native_codex, prepare_native_codex_from_catalog +from devsquad.catalog import update_last_good +from devsquad.codex_protocol import JsonLinePeer, NativeTurnState, collect_model_pages, discover_models, initialize_request, initialized_notification, model_list_request, parse_model_page, receive_response, review_start_request, thread_start_request, turn_interrupt_request, turn_start_request +from devsquad.contracts import ContractError, validate_launch_payload +from devsquad.validation import validate_policy, validate_profile, validate_task + + +class M1ContractsTest(unittest.TestCase): + def manifest(self, name: str) -> AdapterManifest: + return AdapterManifest.load(CORE / "adapters" / name / "adapter.json") + + def fake_path(self, name: str) -> tuple[tempfile.TemporaryDirectory, Path]: + temp = tempfile.TemporaryDirectory() + path = Path(temp.name) / name + if name == "codex": + path.write_text( + "#!/bin/sh\n" + "if [ \"${1:-}\" = --version ]; then\n" + " echo 'codex-cli 0.153.4'\n" + "fi\n" + "exit 0\n" + ) + else: + path.write_text("#!/bin/sh\nexit 0\n") + path.chmod(path.stat().st_mode | stat.S_IXUSR) + return temp, path + + def test_verified_binary_candidate_wins_over_an_older_path_binary(self): + temp, older = self.fake_path("codex") + self.addCleanup(temp.cleanup) + older.write_text("#!/bin/sh\necho 'codex-cli 0.135.0'\n") + verified = Path(temp.name) / "bundled-codex" + verified.write_text("#!/bin/sh\necho 'codex-cli 0.155.0-alpha.9.2'\n") + verified.chmod(verified.stat().st_mode | stat.S_IXUSR) + manifest = replace( + self.manifest("codex"), + binary_candidates=("codex", str(verified)), + ) + + with patch.dict(os.environ, {"PATH": str(older.parent)}): + self.assertEqual(manifest.resolve_binary(), str(verified)) + + def test_current_bundled_layout_and_version_are_explicitly_supported(self): + manifest = self.manifest("codex") + bundled = "/Applications/ChatGPT.app/Contents/Resources/codex-cli/bin/codex" + self.assertIn(bundled, manifest.binary_candidates) + self.assertIn("codex-cli 0.159.2", manifest.verified_versions) + registration = json.loads((CORE / "integrations/codex/registration.json").read_text()) + self.assertIn(bundled, registration["executable_paths"]) + temp, older = self.fake_path("codex") + self.addCleanup(temp.cleanup) + older.write_text("#!/bin/sh\necho 'codex-cli 0.135.0'\n") + current = Path(temp.name) / "current-codex" + current.write_text("#!/bin/sh\necho 'codex-cli 0.159.2'\n") + current.chmod(0o700) + manifest = replace(manifest, binary_candidates=("codex", str(current))) + with patch.dict(os.environ, {"PATH": str(older.parent)}): + self.assertEqual(manifest.resolve_binary(), str(current)) + spec = prepare_native_codex( + manifest.with_model_efforts({"gpt-test": ("low",)}), + cwd=temp.name, model="gpt-test", effort="low", + permission="read_only", timeout_seconds=9, + harness_version_value="codex-cli 0.159.2", + ) + self.assertEqual(spec.requested.verification, "verified") + + def test_prepare_preserves_spaces_and_tsx_prompt(self): + temp, binary = self.fake_path("agy") + self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("antigravity"), prompt="Review ui/My Card.tsx", cwd=temp.name, model="Gemini Test", effort=None, permission="read_only", timeout_seconds=9) + self.assertIn("Review ui/My Card.tsx", spec.argv) + self.assertIn("Gemini Test", spec.argv) + self.assertEqual(spec.environment, {"DEVSQUAD_WORKER": "1"}) + + def test_explicit_unsupported_effort_fails_before_launch(self): + temp, binary = self.fake_path("grok") + self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + with self.assertRaisesRegex(ContractError, "unsupported or unverified effort"): + prepare_cli(self.manifest("grok"), prompt="x", cwd=temp.name, model=None, effort="ultra", permission="read_only", timeout_seconds=9) + + def test_classifier_does_not_accept_empty_denied_or_malformed_exit_zero(self): + temp, binary = self.fake_path("grok") + self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("grok"), prompt="x", cwd=temp.name, model=None, effort=None, permission="read_only", timeout_seconds=9) + self.assertEqual(classify_cli(spec, returncode=0, stdout="", stderr="").execution_status, "malformed") + self.assertEqual(classify_cli(spec, returncode=0, stdout='{"type":"result","is_error":true,"error":"tool denied"}', stderr="").execution_status, "denied") + self.assertEqual(classify_cli(spec, returncode=0, stdout="not json", stderr="").execution_status, "malformed") + self.assertEqual(classify_cli(spec, returncode=0, stdout="42", stderr="").execution_status, "malformed") + nested_bad = '{"type":"item.completed","item":"bad"}\n{"type":"turn.completed"}' + self.assertEqual(classify_cli(spec, returncode=0, stdout=nested_bad, stderr="").execution_status, "malformed") + + def test_auth_precedes_rate_and_acceptance_is_separate(self): + temp, binary = self.fake_path("grok") + self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("grok"), prompt="x", cwd=temp.name, model=None, effort=None, permission="read_only", timeout_seconds=9) + result = classify_cli(spec, returncode=0, stdout='{"type":"result","is_error":true,"error":"401 quota rate limit"}', stderr="") + self.assertEqual(result.error_code, "AUTH_ERROR") + self.assertEqual(result.acceptance_status, "not_evaluated") + self.assertEqual(result.artifact_status, "unknown") + + def test_valid_auth_topic_is_deliverable_and_startup_only_is_not(self): + temp, binary = self.fake_path("grok") + self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("grok"), prompt="x", cwd=temp.name, model=None, effort=None, permission="read_only", timeout_seconds=9) + valid = '{"type":"result","result":"The author explains authentication.","is_error":false}' + self.assertEqual(classify_cli(spec, returncode=0, stdout=valid, stderr="").execution_status, "succeeded") + startup = '{"type":"system","subtype":"init"}' + self.assertEqual(classify_cli(spec, returncode=0, stdout=startup, stderr="").execution_status, "malformed") + + def test_codex_real_jsonl_shape_requires_agent_message(self): + temp, binary = self.fake_path("codex"); self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("codex"), prompt="x", cwd=temp.name, model=None, effort=None, permission="read_only", timeout_seconds=9) + valid = '\n'.join([json.dumps({"type":"thread.started","thread_id":"x"}), json.dumps({"type":"item.completed","item":{"type":"agent_message","text":"done"}}), json.dumps({"type":"turn.completed","usage":{"input_tokens":1}})]) + self.assertEqual(classify_cli(spec, returncode=0, stdout=valid, stderr="").execution_status, "succeeded") + partial = json.dumps({"type":"item.completed","item":{"type":"agent_message","text":"done"}}) + self.assertEqual(classify_cli(spec, returncode=0, stdout=partial, stderr="").execution_status, "malformed") + startup = json.dumps({"type":"thread.started","thread_id":"x"}) + "\n" + json.dumps({"type":"turn.completed","usage":{}}) + self.assertEqual(classify_cli(spec, returncode=0, stdout=startup, stderr="").execution_status, "malformed") + + def test_native_launch_is_preparation_only_and_version_scoped(self): + temp, binary = self.fake_path("codex"); self.addCleanup(temp.cleanup) + manifest = self.manifest("codex").with_model_efforts({"gpt-test": ("low",)}) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.153.4") + self.assertEqual(spec.transport, "native_protocol") + self.assertEqual(spec.argv[-2:], ("--listen", "stdio://")) + self.assertIn('model_reasoning_effort="low"', spec.argv) + self.assertEqual(spec.environment["DEVSQUAD_WORKER"], "1") + current = prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.155.0-alpha.9.2") + self.assertEqual(current.transport, "native_protocol") + with self.assertRaises(ContractError): + prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli future") + + def test_discovered_snapshot_feeds_native_preparation_and_rejects_drift(self): + temp, binary = self.fake_path("codex"); self.addCleanup(temp.cleanup) + manifest = self.manifest("codex") + snapshot = {"complete":True,"harness":"codex","harness_version":"codex-cli 0.153.4","models":[{"id":"gpt-test","supported_efforts":["low"]}]} + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_native_codex_from_catalog(manifest, snapshot, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.153.4") + self.assertEqual(spec.requested.model, "gpt-test") + drifted = dict(snapshot); drifted["harness_version"] = "codex-cli future" + with self.assertRaises(ContractError): + prepare_native_codex_from_catalog(manifest, drifted, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.153.4") + + def test_launch_round_trip_is_strict(self): + temp, binary = self.fake_path("grok"); self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("grok"), prompt="x", cwd=temp.name, model=None, effort=None, permission="read_only", timeout_seconds=9) + validate_launch_payload(spec.to_dict()) + invalid = spec.to_dict(); invalid["surprise"] = True + with self.assertRaises(ContractError): validate_launch_payload(invalid) + + def test_task_examples_validate_and_unknown_fields_fail(self): + root = Path(__file__).resolve().parents[2] + for name in ("branch-review.json", "issue-delivery.json"): + task = json.loads((root / "docs/plans/engineering-team/examples" / name).read_text()) + validate_task(task) + task["unknown"] = 1 + with self.assertRaises(ContractError): validate_task(task) + + def test_nested_task_override_is_strict(self): + root = Path(__file__).resolve().parents[2] + task = json.loads((root / "docs/plans/engineering-team/examples/issue-delivery.json").read_text()) + task["routing"]["overrides"] = {"reviewer":{"profile_id":"x", "fallback":"anything", "unknown":True}} + with self.assertRaises(ContractError): validate_task(task) + + def test_profile_and_policy_strict_fixtures(self): + profile = {"id":"p1","harness":"codex","model_family":"gpt","model_id":"gpt-test","effort":{"value":"low","transport":"native"},"required_tools":["read"],"permission_policy":"read_only","account_pool_id":"codex-sub","billing_mode":"subscription","quality_status":"proven","evidence_refs":["e1"]} + validate_profile(profile) + broken = dict(profile); broken["surprise"] = 1 + with self.assertRaises(ContractError): validate_profile(broken) + policy = {"schema_version":1,"id":"default","version":1,"roles":{"reviewer":[{"kind":"profile","id":"p1"}]},"task_classes":{},"require_different_model_for_review":True,"account_pools":{},"experiment_budget":{}} + validate_policy(policy) + policy["roles"]["reviewer"][0]["unknown"] = True + with self.assertRaises(ContractError): validate_policy(policy) + + +class NativeProtocolTest(unittest.TestCase): + def test_peer_partial_frame_times_out_and_buffered_second_frame_drains(self): + read_fd, write_fd = os.pipe() + reader = os.fdopen(read_fd, "r"); writer = os.fdopen(write_fd, "w") + self.addCleanup(reader.close); self.addCleanup(writer.close) + peer = JsonLinePeer(reader, writer) + os.write(write_fd, b'{') + started = __import__('time').monotonic() + with self.assertRaises(TimeoutError): peer.receive(0.05) + self.assertLess(__import__('time').monotonic() - started, 0.2) + os.write(write_fd, b'}\n{"n":2}\n') + self.assertEqual(peer.receive(0.1), {}) + self.assertEqual(peer.receive(0.1), {"n":2}) + def test_fake_app_server_protocol_conformance(self): + fake = Path(__file__).parent / "fakes/codex_app_server.py" + process = subprocess.Popen([sys.executable, str(fake)], stdin=subprocess.PIPE, stdout=subprocess.PIPE, text=True) + def cleanup(): + if process.poll() is None: process.terminate() + process.wait(timeout=2) + process.stdin.close(); process.stdout.close() + self.addCleanup(cleanup) + peer = JsonLinePeer(process.stdout, process.stdin) + peer.send(initialize_request(1)); self.assertIn("result", receive_response(peer, 1, timeout_seconds=2)); peer.send(initialized_notification()) + self.assertEqual([m["id"] for m in discover_models(peer, first_request_id=2, timeout_seconds=2)], ["gpt-fake"]) + peer.send(turn_start_request(4, thread_id="thread-1", prompt="p", model="gpt-fake", effort="low", cwd="/tmp", permission="read_only")) + response = receive_response(peer, 4, timeout_seconds=2); self.assertEqual(response["result"]["turn"]["id"], "turn-1") + state = NativeTurnState(thread_id="thread-1", turn_id="turn-1") + for _ in range(3): state.consume(peer.receive(2)) + self.assertTrue(state.terminal); self.assertEqual("".join(state.output), "ok") + def test_paginated_model_request_and_response(self): + self.assertEqual(model_list_request(2, "next")["params"]["cursor"], "next") + models, cursor = parse_model_page({"result": {"data": [{"id": "gpt-x"}], "nextCursor": "c2"}}) + self.assertEqual(models[0]["id"], "gpt-x") + self.assertEqual(cursor, "c2") + + def test_start_and_interrupt_ack_are_not_terminal(self): + state = NativeTurnState(thread_id="th1") + state.consume({"method": "turn/started", "params": {"threadId": "th1", "turn": {"id": "t1", "status":"inProgress"}}}) + state.acknowledge_interrupt({"id": 4, "result": {}}) + self.assertFalse(state.terminal) + self.assertTrue(state.interrupted_acknowledged) + state.consume({"method": "turn/completed", "params": {"threadId":"th1", "turn":{"id":"t1", "status":"interrupted"}}}) + self.assertTrue(state.terminal) + + def test_completed_agent_message_is_output_fallback_without_duplication(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + state.consume({ + "method": "item/completed", + "params": { + "threadId": "th1", "turnId": "t1", + "item": {"type": "agentMessage", "text": "fallback"}, + }, + }) + self.assertEqual(state.final_output(), "fallback") + state.consume({ + "method": "item/completed", + "params": { + "threadId": "th1", "turnId": "t1", + "item": {"type": "agentMessage", "text": "duplicate"}, + }, + }) + self.assertEqual(state.final_output(), "duplicate") + + def test_completed_agent_message_replaces_an_empty_stream(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + state.consume({ + "method": "item/agentMessage/delta", + "params": { + "threadId": "th1", "turnId": "t1", "itemId": "item-1", + "delta": "", + }, + }) + state.consume({ + "method": "item/completed", + "params": { + "threadId": "th1", "turnId": "t1", + "item": {"type": "agentMessage", "text": "structured result"}, + }, + }) + self.assertEqual(state.final_output().strip(), "structured result") + + def test_last_completed_agent_message_wins_over_multiple_delta_streams(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + for item_id, content in (("first", '{"draft":true}'), ("final", '{"ok":true}')): + state.consume({ + "method": "item/agentMessage/delta", + "params": { + "threadId": "th1", "turnId": "t1", + "itemId": item_id, "delta": content, + }, + }) + state.consume({ + "method": "item/completed", + "params": { + "threadId": "th1", "turnId": "t1", + "item": {"id": item_id, "type": "agentMessage", "text": content}, + }, + }) + state.consume({ + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": {"id": "t1", "status": "completed", "items": []}, + }, + }) + self.assertEqual("".join(state.output), '{"draft":true}{"ok":true}') + self.assertEqual(state.final_output(), '{"ok":true}') + + def test_terminal_turn_items_are_output_fallback_without_duplication(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + state.consume({ + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": { + "id": "t1", "status": "completed", + "items": [ + {"id": "reasoning", "type": "reasoning", "summary": []}, + {"id": "first", "type": "agentMessage", "text": ""}, + {"id": "final", "type": "agentMessage", "text": "structured result"}, + ], + }, + }, + }) + self.assertEqual(state.final_output(), "structured result") + self.assertTrue(state.terminal) + + streamed = NativeTurnState(thread_id="th1", turn_id="t1") + streamed.consume({ + "method": "item/agentMessage/delta", + "params": {"threadId": "th1", "turnId": "t1", "delta": "streamed"}, + }) + streamed.consume({ + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": { + "id": "t1", "status": "completed", + "items": [ + {"id": "final", "type": "agentMessage", "text": "streamed"}, + ], + }, + }, + }) + self.assertEqual(streamed.final_output(), "streamed") + + def test_terminal_turn_rejects_malformed_items(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + with self.assertRaisesRegex(ContractError, "items must be an array"): + state.consume({ + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": {"id": "t1", "status": "completed", "items": {}}, + }, + }) + + def test_output_diagnostics_are_structural_and_redacted(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + secret = "do-not-retain-this-model-text" + for message in [ + { + "method": "item/agentMessage/delta", + "params": {"threadId": "other", "turnId": "t1", "delta": secret}, + }, + { + "method": "item/completed", + "params": { + "threadId": "th1", "turnId": "t1", + "item": {"type": "agentMessage", "text": secret}, + }, + }, + { + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": { + "id": "t1", "status": "completed", + "items": [{"type": "agentMessage", "text": secret}], + }, + }, + }, + ]: + state.consume(message) + diagnostics = state.output_diagnostics() + self.assertEqual(diagnostics["event_count"], 3) + self.assertEqual(diagnostics["matching_completed_messages"], 1) + self.assertEqual( + diagnostics["matching_completed_message_bytes"], len(secret.encode()), + ) + self.assertEqual(diagnostics["matching_terminal_agent_messages"], 1) + self.assertEqual(diagnostics["uncorrelated_output_events"], 1) + self.assertNotIn(secret, json.dumps(diagnostics, sort_keys=True)) + + def test_unrelated_turn_cannot_complete_ours_and_disconnect_is_visible(self): + state = NativeTurnState(thread_id="th1", turn_id="ours") + state.consume({"method":"turn/completed", "params":{"threadId":"th1", "turn":{"id":"other", "status":"completed"}}}) + self.assertFalse(state.terminal) + state.disconnected(); self.assertEqual(state.terminal_status, "transport_disconnected") + + def test_failed_terminal_turn_preserves_typed_provider_error(self): + state = NativeTurnState(thread_id="th1", turn_id="ours") + state.consume({ + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": { + "id": "ours", "status": "failed", + "error": {"code": "model_error", "message": "bounded detail"}, + }, + }, + }) + self.assertEqual(state.terminal_status, "failed") + self.assertEqual(state.error["code"], "model_error") + + def test_typed_native_requests_match_installed_contract(self): + thread = thread_start_request(1, cwd="/tmp/repo", model="gpt-test", permission="read_only") + self.assertEqual(thread["params"]["sandbox"], "read-only") + self.assertFalse(thread["params"]["ephemeral"]) + self.assertTrue(thread_start_request( + 5, cwd="/tmp/repo", model="gpt-test", permission="read_only", + ephemeral=True, + )["params"]["ephemeral"]) + turn = turn_start_request(2, thread_id="th", prompt="p", model="gpt-test", effort="low", cwd="/tmp/repo", permission="read_only", output_schema={"type":"object"}) + self.assertEqual(turn["params"]["sandboxPolicy"], {"type":"readOnly", "networkAccess":False}) + self.assertEqual(turn["params"]["outputSchema"]["type"], "object") + self.assertEqual(turn_interrupt_request(3, thread_id="th", turn_id="tu")["params"]["turnId"], "tu") + self.assertEqual(review_start_request(4, thread_id="th", target={"type":"commit", "sha":"abc"})["method"], "review/start") + + def test_disconnect_or_malformed_page_is_visible(self): + with self.assertRaises(ContractError): + parse_model_page({"result": {"nextCursor": "never"}}) + + def test_complete_pagination_empty_page_and_repeated_cursor(self): + pages = {None:{"result":{"data":[{"id":"a"}],"nextCursor":"c"}}, "c":{"result":{"data":[],"nextCursor":None}}} + self.assertEqual([m["id"] for m in collect_model_pages(lambda c: pages[c])], ["a"]) + with self.assertRaises(ContractError): + collect_model_pages(lambda c: {"result":{"data":[],"nextCursor":"same"}}) + + +class CatalogTest(unittest.TestCase): + @staticmethod + def profile(profile_id, model_id, *, harness="codex"): + return { + "id": profile_id, + "harness": harness, + "model_family": "gpt", + "model_id": model_id, + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": f"{harness}-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["catalog-fixture"], + } + + def test_incomplete_refresh_retains_last_good(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "catalog.json" + first = update_last_good(path, harness="codex", version="1", models=[{"id": "gpt-a", "supportedReasoningEfforts": ["low"]}], complete=True) + retained = update_last_good(path, harness="codex", version="2", models=None, complete=False, error="timeout") + self.assertEqual(retained["models"], first["models"]) + self.assertEqual(retained["last_refresh"]["status"], "error") + + def test_discovered_model_is_unqualified_and_unknown_family_stays_unknown(self): + with tempfile.TemporaryDirectory() as tmp: + value = update_last_good(Path(tmp) / "catalog.json", harness="codex", version="1", models=[{"id": "surprise-9"}], complete=True) + self.assertEqual(value["models"][0]["qualification"], "unqualified") + self.assertIsNone(value["models"][0]["family"]) + + def test_changed_effort_revalidates_only_affected_profiles(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "catalog.json" + profiles = [ + self.profile("profile-a", "gpt-a"), + self.profile("profile-b", "gpt-b"), + self.profile("profile-other", "gpt-a", harness="other"), + ] + update_last_good( + path, harness="codex", version="1", complete=True, + models=[ + {"id": "gpt-a", "modelRevision": "a1", "supportedReasoningEfforts": ["low"]}, + {"id": "gpt-b", "modelRevision": "b1", "supportedReasoningEfforts": ["low"]}, + ], + profiles=profiles, + ) + changed = update_last_good( + path, harness="codex", version="1", complete=True, + models=[ + {"id": "gpt-a", "modelRevision": "a2", "supportedReasoningEfforts": ["low", "high"]}, + {"id": "gpt-b", "modelRevision": "b1", "supportedReasoningEfforts": ["low"]}, + ], + profiles=profiles, + )["catalog_change"] + self.assertEqual(changed["changed_model_ids"], ["gpt-a"]) + self.assertEqual(changed["affected_profile_ids"], ["profile-a"]) + self.assertEqual(changed["same_id_revision_unknown"], []) + self.assertEqual(changed["unavailable_profile_ids"], []) + self.assertFalse(changed["binding_changes_applied"]) + + def test_same_id_unknown_revision_added_and_removed_models_stay_safe(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "catalog.json" + profiles = [ + self.profile("profile-a", "gpt-a"), + self.profile("profile-b", "gpt-b"), + ] + update_last_good( + path, harness="codex", version="1", complete=True, + models=[ + {"id": "gpt-a", "supportedReasoningEfforts": ["low"]}, + {"id": "gpt-b", "supportedReasoningEfforts": ["low"]}, + ], + profiles=profiles, + ) + changed = update_last_good( + path, harness="codex", version="1", complete=True, + models=[ + {"id": "gpt-a", "supportedReasoningEfforts": ["low", "high"]}, + {"id": "gpt-new", "supportedReasoningEfforts": ["low"]}, + ], + profiles=profiles, + ) + drift = changed["catalog_change"] + self.assertEqual(drift["changed_model_ids"], ["gpt-a"]) + self.assertEqual(drift["same_id_revision_unknown"], ["gpt-a"]) + self.assertEqual(drift["added_model_ids"], ["gpt-new"]) + self.assertEqual(drift["unqualified_candidate_ids"], ["gpt-new"]) + self.assertEqual(drift["removed_model_ids"], ["gpt-b"]) + self.assertEqual(drift["affected_profile_ids"], ["profile-a", "profile-b"]) + self.assertEqual(drift["unavailable_profile_ids"], ["profile-b"]) + self.assertEqual( + next(model for model in changed["models"] if model["id"] == "gpt-new")["qualification"], + "unqualified", + ) + self.assertFalse(drift["binding_changes_applied"]) + + def test_duplicate_catalog_model_ids_are_rejected(self): + with tempfile.TemporaryDirectory() as tmp: + with self.assertRaisesRegex(ContractError, "unique"): + update_last_good( + Path(tmp) / "catalog.json", + harness="codex", + version="1", + complete=True, + models=[{"id": "gpt-a"}, {"model": "gpt-a"}], + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_m1_gate_review.py b/test/core/test_m1_gate_review.py new file mode 100644 index 0000000..de2eb0a --- /dev/null +++ b/test/core/test_m1_gate_review.py @@ -0,0 +1,181 @@ +"""Independent M1 review regressions for externally supplied data and framing. + +These cases came from reviewing the first implementation, rather than from +its internal structure. They run without provider access or a core install. +""" +from __future__ import annotations + +import copy +import json +import os +from pathlib import Path +import sys +import threading +import time +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) + +from devsquad.adapters import classify_cli +from devsquad.codex_protocol import JsonLinePeer, NativeTurnState +from devsquad.contracts import ( + ContractError, ExecutionIdentity, LaunchSpec, validate_launch_payload, +) +from devsquad.validation import validate_task + + +def launch() -> LaunchSpec: + return LaunchSpec( + 1, "codex", "cli_exec", ("fake-codex",), "/tmp", None, 2, + ExecutionIdentity("codex", None, None, None, None, None), + ) + + +class UntrustedInputReview(unittest.TestCase): + def test_launch_rejects_invalid_types_without_coercion(self): + cases = [ + ("schema boolean", {"schema_version": True}), + ("timeout boolean", {"timeout_seconds": True}), + ("timeout NaN", {"timeout_seconds": float("nan")}), + ("timeout infinity", {"timeout_seconds": float("inf")}), + ("timeout fraction", {"timeout_seconds": 1.5}), + ("string argv", {"argv": "fake-codex"}), + ("object argv", {"argv": {"fake-codex": True}}), + ("empty argv", {"argv": []}), + ("relative cwd", {"cwd": "relative"}), + ("arbitrary environment", {"environment": {"UNDECLARED_FLAG": "1"}}), + ] + for name, change in cases: + with self.subTest(name=name): + value = launch().to_dict() + value.update(change) + with self.assertRaises(ContractError): + validate_launch_payload(value) + + def test_task_rejects_invalid_nested_values(self): + original = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + cases = [ + ("base ref", ("project", "base_ref"), {}), + ("criterion id", ("acceptance", 0, "id"), {}), + ("criterion description", ("acceptance", 0, "description"), []), + ("check id", ("checks", 0, "id"), 123), + ("check argv", ("checks", 0, "argv"), "echo hello"), + ("check boolean", ("checks", 0, "required_to_pass"), "false"), + ("session reference", ("origin", "session_ref"), {}), + ("scope list", ("scope", "read_paths"), "src"), + ("finite budget", ("budget", "wall_seconds"), float("nan")), + ("fallback", ("routing", "overrides"), { + "reviewer": {"profile_id": "fixture", "fallback": "anything"}, + }), + ] + for name, path, value in cases: + with self.subTest(name=name): + task = copy.deepcopy(original) + target = task + for key in path[:-1]: + target = target[key] + target[path[-1]] = value + with self.assertRaises(ContractError): + validate_task(task) + + def test_malformed_provider_frames_return_a_verdict(self): + for frame in (42, None, ["bad"], {"type": "item.completed", "item": "bad"}, + {"type": "item.completed", "item": 7}): + with self.subTest(frame=frame): + result = classify_cli(launch(), returncode=0, stdout=json.dumps(frame), stderr="") + self.assertEqual(result.error_code, "CLI_ERROR") + self.assertNotEqual(result.execution_status, "succeeded") + + def test_partial_native_message_is_not_completion(self): + frame = {"type": "item.completed", "item": {"type": "agent_message", "text": "partial"}} + result = classify_cli(launch(), returncode=0, stdout=json.dumps(frame), stderr="") + self.assertEqual(result.error_code, "CLI_ERROR") + self.assertNotEqual(result.execution_status, "succeeded") + + +class NativeFramingReview(unittest.TestCase): + def test_native_state_rejects_malformed_notification_values(self): + cases = [ + {"method": "turn/completed", "params": 42}, + {"method": "turn/completed", "params": {"turn": "invalid"}}, + {"method": "item/agentMessage/delta", "params": { + "threadId": "expected-thread", "turnId": "expected-turn", "delta": 42, + }}, + ] + for message in cases: + with self.subTest(message=message): + state = NativeTurnState(thread_id="expected-thread", turn_id="expected-turn") + with self.assertRaises(ContractError): + state.consume(message) + + def test_native_state_keeps_bound_identity_and_requires_correlation(self): + state = NativeTurnState(thread_id="expected-thread", turn_id="expected-turn") + for message in [ + {"method": "thread/started", "params": {"thread": {"id": "unrelated"}}}, + {"method": "item/agentMessage/delta", "params": {"delta": "no identities"}}, + {"method": "item/agentMessage/delta", "params": { + "threadId": "unrelated", "turnId": "expected-turn", "delta": "wrong thread", + }}, + {"method": "turn/completed", "params": { + "turn": {"id": "expected-turn", "status": "completed"}, + }}, + ]: + state.consume(message) + self.assertEqual(state.thread_id, "expected-thread") + self.assertEqual(state.turn_id, "expected-turn") + self.assertEqual(state.output, []) + self.assertFalse(state.terminal) + state.consume({"method": "item/agentMessage/delta", "params": { + "threadId": "expected-thread", "turnId": "expected-turn", "delta": "valid", + }}) + state.consume({"method": "turn/completed", "params": { + "threadId": "expected-thread", "turn": {"id": "expected-turn", "status": "completed"}, + }}) + self.assertEqual(state.output, ["valid"]) + self.assertEqual(state.terminal_status, "completed") + self.assertTrue(state.terminal) + + def test_two_frames_in_one_write_are_both_available(self): + read_fd, write_fd = os.pipe() + with os.fdopen(read_fd, "r") as reader, os.fdopen(write_fd, "w") as writer: + peer = JsonLinePeer(reader, writer) + writer.write('{"n":1}\n{"n":2}\n') + writer.flush() + self.assertEqual(peer.receive(0.2), {"n": 1}) + self.assertEqual(peer.receive(0.2), {"n": 2}) + + def test_incomplete_frame_respects_receive_deadline(self): + read_fd, write_fd = os.pipe() + with os.fdopen(read_fd, "r") as reader, os.fdopen(write_fd, "w") as writer: + peer = JsonLinePeer(reader, writer) + writer.write("{") + writer.flush() + outcomes = [] + + def receive(): + try: + outcomes.append(peer.receive(0.05)) + except Exception as exc: + outcomes.append(exc) + + worker = threading.Thread(target=receive, daemon=True) + started = time.monotonic() + worker.start() + worker.join(0.4) + exceeded_deadline = worker.is_alive() + # Release a buggy blocking readline before asserting, so a failed + # regression does not leave a test thread or pipe behind. + if exceeded_deadline: + writer.write('"late":true}\n') + writer.flush() + worker.join(1) + self.assertFalse(exceeded_deadline, "receive ignored its deadline") + self.assertLess(time.monotonic() - started, 0.4) + self.assertIsInstance(outcomes[0], TimeoutError) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_m2_cross_process.py b/test/core/test_m2_cross_process.py new file mode 100644 index 0000000..cb27764 --- /dev/null +++ b/test/core/test_m2_cross_process.py @@ -0,0 +1,312 @@ +"""Independent-process races for the public M2 service and writer fences.""" +from __future__ import annotations + +import json +import multiprocessing +import os +from pathlib import Path +import signal +import subprocess +import sys +import tempfile +import time +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) + +from devsquad.service import Service +from devsquad.reports import TERMINAL_REPORT_NAMES +from devsquad.store import Store, request_hash +from devsquad.supervisor import inspect_process +from devsquad_test_fixtures import branch_review_routing_documents + + +def service_start(runtime, task, key, barrier, results): + try: + barrier.wait(timeout=10) + value = Service(Path(runtime)).start(task, key) + results.put(("ok", value["run_id"], value["created"], value["state"])) + except Exception as exc: + results.put(("error", type(exc).__name__, str(exc))) + + +def service_cancel(runtime, run_id, barrier, results): + try: + barrier.wait(timeout=10) + value = Service(Path(runtime)).cancel(run_id) + results.put(("ok", value["state"], value["version"])) + except Exception as exc: + results.put(("error", type(exc).__name__, str(exc))) + + +def service_handoff_claim(runtime, run_id, version, owner, barrier, results): + try: + barrier.wait(timeout=10) + value = Service(Path(runtime)).handoff_claim(run_id, version, owner) + results.put(( + "ok", value["claim"]["owner"], value["claim"]["fencing_token"], + )) + except Exception as exc: + results.put(("error", type(exc).__name__, str(exc))) + + +def service_handoff_complete(runtime, run_id, claim, decision, barrier, results): + try: + barrier.wait(timeout=10) + value = Service(Path(runtime)).handoff_complete(run_id, claim, decision) + results.put(( + "ok", value["replayed"], value["recorded_run_version"], + )) + except Exception as exc: + results.put(("error", type(exc).__name__, str(exc))) + + +def reserve_writer(database, artifacts, run_id, version, owner, barrier, results): + store = None + try: + store = Store(Path(database), Path(artifacts)) + barrier.wait(timeout=10) + reservation = store.reserve_attempt(run_id, version, owner, "package") + results.put(("ok", run_id, reservation.attempt_id)) + except Exception as exc: + results.put(("error", run_id, type(exc).__name__, str(exc))) + finally: + if store is not None: + store.close() + + +class CrossProcessServiceTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-process-races-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run(["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], check=True) + subprocess.run(["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True) + profiles_json, policy_json = branch_review_routing_documents() + (self.repo / "profiles.json").write_text(profiles_json) + (self.repo / "policy.json").write_text(policy_json) + subprocess.run(["git", "-C", str(self.repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) + self.task["project"] = { + "repo_path": str(self.repo), "base_ref": "HEAD", "target_ref": "HEAD", + } + self.task["routing"]["profiles_file"] = "profiles.json" + self.task["routing"]["policy_file"] = "policy.json" + self.context = multiprocessing.get_context("spawn") + + def run_processes(self, targets): + results = self.context.Queue() + barrier = self.context.Barrier(len(targets)) + processes = [self.context.Process(target=target, args=(*args, barrier, results)) for target,args in targets] + try: + for process in processes: + process.start() + outcomes = [results.get(timeout=20) for _ in processes] + for process in processes: + process.join(timeout=5) + self.assertTrue(all(process.exitcode == 0 for process in processes), processes) + return outcomes + finally: + for process in processes: + if process.is_alive(): + process.terminate() + process.join(timeout=2) + results.close() + results.join_thread() + + def waiting_handoff(self, key): + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + claim = store.claim_start(self.repo, key, {"task": key}, "preflight") + version = store.complete_preparation( + claim.run_id, + claim.fencing_token, + {"head": "fixed"}, + package_path="/frozen/package", + package_digest="package-digest", + ) + reservation = store.reserve_attempt( + claim.run_id, version, "supervisor", "package-digest", + ) + version = store.mark_attempt_running( + reservation, 101, 101, "process-start-id", + ) + handoff = store.publish_handoff( + claim.run_id, + version, + reservation.attempt_token, + reservation.supervisor_token, + {"schema_version": 1, "candidate_sha256": "c" * 64}, + ) + return claim.run_id, handoff.run_version + finally: + store.close() + + def wait_state(self, run_id, expected, timeout=8): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status = Service(self.runtime).status(run_id) + if status["state"] in expected: + return status + time.sleep(0.05) + self.fail(f"run did not reach {expected}: {Service(self.runtime).status(run_id)}") + + def test_identical_public_starts_share_one_run_across_processes(self): + args = (str(self.runtime), self.task, "same-process-key") + outcomes = self.run_processes([(service_start,args),(service_start,args)]) + self.assertTrue(all(outcome[0] == "ok" for outcome in outcomes), outcomes) + self.assertEqual(len({outcome[1] for outcome in outcomes}), 1) + self.assertEqual(sorted(outcome[2] for outcome in outcomes), [False, True]) + run_id = outcomes[0][1] + result = Service(self.runtime).result(run_id) + self.assertTrue(result["ready"]) + self.assertEqual( + {item["name"] for item in result["artifacts"]}, + set(TERMINAL_REPORT_NAMES), + ) + + def test_changed_body_conflicts_with_same_key_across_processes(self): + changed = json.loads(json.dumps(self.task)) + changed["goal"] += " changed" + outcomes = self.run_processes([ + (service_start,(str(self.runtime),self.task,"conflicting-process-key")), + (service_start,(str(self.runtime),changed,"conflicting-process-key")), + ]) + self.assertEqual(sorted(outcome[0] for outcome in outcomes), ["error", "ok"]) + error = next(outcome for outcome in outcomes if outcome[0] == "error") + self.assertEqual(error[1], "ConflictError") + + def test_two_processes_cannot_reserve_two_worktree_writers(self): + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + ready = [] + for key in ("writer-a", "writer-b"): + claim = store.claim_start(self.repo, key, {"task": key}, "preflight") + version = store.complete_preparation(claim.run_id, claim.fencing_token, {"head": key}) + ready.append((claim.run_id, version)) + finally: + store.close() + common = (str(self.runtime / "state.sqlite3"), str(self.runtime / "artifacts")) + outcomes = self.run_processes([ + (reserve_writer,(*common,*ready[0],"owner-a")), + (reserve_writer,(*common,*ready[1],"owner-b")), + ]) + self.assertEqual(sorted(outcome[0] for outcome in outcomes), ["error", "ok"]) + self.assertEqual(next(item for item in outcomes if item[0] == "error")[2], "ConflictError") + winner = next(item for item in outcomes if item[0] == "ok")[1] + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + store.cancel_launching(winner) + finally: + store.close() + + def test_repeated_cancel_is_idempotent_across_processes(self): + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + claim = store.claim_start(self.repo, "cancel-race", {"task": 1}, "preflight") + store.complete_preparation(claim.run_id, claim.fencing_token, {"head": "fixed"}) + finally: + store.close() + args = (str(self.runtime), claim.run_id) + outcomes = self.run_processes([(service_cancel,args),(service_cancel,args)]) + self.assertTrue(all(outcome[0:2] == ("ok", "cancelled") for outcome in outcomes), outcomes) + self.assertEqual(len({outcome[2] for outcome in outcomes}), 1) + result = Service(self.runtime).result(claim.run_id) + self.assertEqual([item["name"] for item in result["artifacts"]], ["result-receipt.json"]) + + def test_running_cancel_race_reaps_runner_and_worker_once(self): + service = Service(self.runtime) + started = service.start( + self.task, "running-cancel-race", _internal_fake_delay=30, + ) + self.wait_state(started["run_id"], {"running"}) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + child = None + try: + attempt = store.attempt(started["run_id"]) + child_path = Path(attempt["child_record"]) + deadline = time.monotonic() + 5 + while not child_path.is_file() and time.monotonic() < deadline: + time.sleep(0.02) + self.assertTrue(child_path.is_file()) + child = json.loads(child_path.read_text()) + finally: + store.close() + try: + args = (str(self.runtime), started["run_id"]) + outcomes = self.run_processes([(service_cancel, args), (service_cancel, args)]) + self.assertTrue(all(outcome[0] == "ok" for outcome in outcomes), outcomes) + self.assertTrue(all(outcome[1] in {"cancelling", "cancelled"} for outcome in outcomes)) + self.wait_state(started["run_id"], {"cancelled"}) + + deadline = time.monotonic() + 5 + while (inspect_process( + attempt["pid"], attempt["pgid"], attempt["process_start_id"], + ) != "dead" and time.monotonic() < deadline): + time.sleep(0.02) + self.assertEqual( + inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]), + "dead", + ) + self.assertEqual( + inspect_process(child["pid"], child["pgid"], child["process_start_id"]), + "dead", + ) + result = service.result(started["run_id"]) + self.assertTrue(result["ready"]) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + types = [ + event["type"] + for event in store.events_page(started["run_id"], limit=1000)["events"] + ] + self.assertEqual(types.count("run.cancelling"), 1) + self.assertEqual(types.count("run.cancelled"), 1) + finally: + store.close() + finally: + if child and inspect_process( + child["pid"], child["pgid"], child["process_start_id"], + ) == "live": + os.killpg(child["pgid"], signal.SIGKILL) + + def test_two_processes_cannot_claim_one_host_handoff(self): + run_id, version = self.waiting_handoff("handoff-claim-race") + outcomes = self.run_processes([ + (service_handoff_claim, (str(self.runtime), run_id, version, "host-a")), + (service_handoff_claim, (str(self.runtime), run_id, version, "host-b")), + ]) + self.assertEqual(sorted(outcome[0] for outcome in outcomes), ["error", "ok"]) + self.assertEqual( + next(outcome for outcome in outcomes if outcome[0] == "error")[1], + "ConflictError", + ) + + def test_identical_handoff_completions_replay_across_processes(self): + run_id, version = self.waiting_handoff("handoff-complete-race") + acquired = Service(self.runtime).handoff_claim(run_id, version, "host-a") + body = { + "schema_version": 1, + "submission_id": "submission-1", + "disposition": "accept", + "reason": "accepted", + "evidence_refs": [], + } + decision = {**body, "submission_hash": request_hash(body)} + common = (str(self.runtime), run_id, acquired["claim"], decision) + outcomes = self.run_processes([ + (service_handoff_complete, common), + (service_handoff_complete, common), + ]) + self.assertTrue(all(outcome[0] == "ok" for outcome in outcomes), outcomes) + self.assertEqual(sorted(outcome[1] for outcome in outcomes), [False, True]) + self.assertEqual(len({outcome[2] for outcome in outcomes}), 1) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_m2_gate_review.py b/test/core/test_m2_gate_review.py new file mode 100644 index 0000000..ef1f538 --- /dev/null +++ b/test/core/test_m2_gate_review.py @@ -0,0 +1,125 @@ +"""Independent M2 checks for persisted ownership and artifact integrity.""" +from __future__ import annotations + +import json +import multiprocessing +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) + +from devsquad.contracts import ContractError +from devsquad.store import ConflictError, Store + + +def concurrent_open(database, artifacts, barrier, results): + store = None + try: + barrier.wait(timeout=10) + store = Store(Path(database), Path(artifacts)) + results.put(None) + except Exception as exc: + results.put(f"{type(exc).__name__}: {exc}") + finally: + if store is not None: + store.close() + + +class StoreIntegrityReview(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-store-review-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + self.store = Store(self.root / "state.sqlite3", self.root / "artifacts") + self.addCleanup(self.store.close) + self.task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) + self.task["project"]["repo_path"] = str(self.repo) + self.claim = self.store.claim_start(self.repo, "review-key", self.task, "first-owner") + + def test_duplicate_artifact_never_changes_existing_receipt_content(self): + artifact_id = self.store.store_artifact(self.claim.run_id, "receipt.json", b"original") + row = self.store.connection.execute("SELECT path,sha256 FROM artifacts WHERE id=?", (artifact_id,)).fetchone() + path, original_digest = Path(row["path"]), row["sha256"] + try: + self.store.store_artifact(self.claim.run_id, "receipt.json", b"replacement") + except ContractError: + pass + self.assertEqual(path.read_bytes(), b"original") + row = self.store.connection.execute("SELECT sha256 FROM artifacts WHERE id=?", (artifact_id,)).fetchone() + self.assertEqual(row["sha256"], original_digest) + + def test_artifact_finalization_cannot_escape_with_run_id(self): + outside = self.root / "outside" + for run_id in (str(outside), "../outside"): + with self.subTest(run_id=run_id), self.assertRaises(ContractError): + self.store.finalize_artifact(run_id, "receipt.json", b"unowned") + self.assertFalse(outside.exists()) + + def test_queued_preparation_cannot_be_made_runnable_by_generic_event(self): + run = self.store.run(self.claim.run_id) + self.assertEqual(run["state"], "queued") + self.assertEqual(run["phase"], "preparing") + with self.assertRaises(ConflictError): + self.store.append_event(self.claim.run_id, run["version"], "run.started", {}, state="running") + self.assertEqual(self.store.run(self.claim.run_id)["phase"], "preparing") + + def test_cancelled_preparation_cannot_publish_late_snapshot(self): + version = self.store.cancel_preparing(self.claim.run_id) + with self.assertRaises(ConflictError): + self.store.complete_preparation(self.claim.run_id, self.claim.fencing_token, {"base": "late"}) + self.assertEqual(self.store.run(self.claim.run_id)["state"], "cancelled") + self.assertEqual(self.store.cancel_preparing(self.claim.run_id), version) + + def test_artifact_reference_is_versioned_and_terminal_run_is_immutable(self): + before = self.store.run(self.claim.run_id)["version"] + self.store.store_artifact(self.claim.run_id, "input.json", b"frozen input") + after = self.store.run(self.claim.run_id)["version"] + self.assertEqual(after, before + 1) + event = self.store.connection.execute( + "SELECT run_version FROM events WHERE run_id=? ORDER BY id DESC LIMIT 1", + (self.claim.run_id,), + ).fetchone() + self.assertEqual(event["run_version"], after) + terminal_version = self.store.cancel_preparing(self.claim.run_id) + with self.assertRaises(ConflictError): + self.store.store_artifact(self.claim.run_id, "late.json", b"late write") + self.assertEqual(self.store.run(self.claim.run_id)["version"], terminal_version) + count = self.store.connection.execute( + "SELECT COUNT(*) FROM artifacts WHERE run_id=?", (self.claim.run_id,), + ).fetchone()[0] + self.assertEqual(count, 2) + self.assertIsNotNone(self.store.artifact_named(self.claim.run_id, "result-receipt.json")) + + +class StoreInitializationReview(unittest.TestCase): + def test_independent_processes_can_open_one_new_database(self): + context = multiprocessing.get_context("spawn") + with tempfile.TemporaryDirectory(prefix="devsquad-store-race-") as directory: + root = Path(directory) + barrier, results = context.Barrier(4), context.Queue() + processes = [context.Process(target=concurrent_open, args=(str(root / "db"), str(root / "artifacts"), barrier, results)) for _ in range(4)] + try: + for process in processes: + process.start() + outcomes = [results.get(timeout=15) for _ in processes] + for process in processes: + process.join(timeout=2) + self.assertEqual(outcomes, [None] * 4) + self.assertTrue(all(process.exitcode == 0 for process in processes)) + finally: + for process in processes: + if process.is_alive(): + process.terminate() + process.join(timeout=2) + results.close() + results.join_thread() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_m2_supervisor_gate.py b/test/core/test_m2_supervisor_gate.py new file mode 100644 index 0000000..d304458 --- /dev/null +++ b/test/core/test_m2_supervisor_gate.py @@ -0,0 +1,239 @@ +"""Independent adversarial gates for M2 process supervision.""" +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import signal +import shutil +import subprocess +import sys +import tempfile +import time +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) + +from devsquad.contracts import ContractError, ExecutionIdentity, LaunchSpec +from devsquad.store import ConflictError, Store +from devsquad.supervisor import Supervisor, inspect_process, process_start_identity + + +class SupervisorGateReview(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-supervisor-gate-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + self.store = Store(self.root / "runtime.sqlite3", self.root / "artifacts") + self.addCleanup(self.store.close) + self.supervisor = Supervisor(self.store, output_limit=32, grace_seconds=0.1) + + def _ready_run(self, key: str): + claim = self.store.claim_start(self.repo, key, {"task": key}, "preflight") + version = self.store.complete_preparation(claim.run_id, claim.fencing_token, {"head": "fixed"}) + return claim.run_id, version + + def _spec(self, *argv: str, stdin_path: str | None = None) -> LaunchSpec: + return LaunchSpec( + 1, "fake", "cli_exec", tuple(argv), str(self.repo), stdin_path, 5, + ExecutionIdentity("fake", "1", "fixture", "fixture", "fixture", "low"), + ) + + @staticmethod + def _force_cleanup(handle) -> None: + if handle.process.poll() is None: + try: + os.killpg(handle.process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + handle.process.wait(timeout=2) + for drain in (handle.stdout, handle.stderr): + drain.thread.join(timeout=2) + drain.stream.close() + + def test_current_process_has_stable_strong_identity(self): + first = process_start_identity(os.getpid()) + second = process_start_identity(os.getpid()) + self.assertIsInstance(first, str) + self.assertEqual(first, second) + self.assertEqual(inspect_process(os.getpid(), os.getpgid(os.getpid()), first), "live") + + def test_fast_success_is_a_valid_attempt(self): + run_id, version = self._ready_run("fast") + true_binary = shutil.which("true") + self.assertIsNotNone(true_binary) + handle = self.supervisor.launch(run_id, version, self._spec(true_binary), "owner", "package") + self.assertEqual(self.supervisor.wait(handle, 2), 0) + self.assertEqual(self.store.run(run_id)["state"], "succeeded") + + def test_spawn_failure_is_fenced_and_releases_worktree_writer(self): + run_id, version = self._ready_run("missing") + with self.assertRaises(FileNotFoundError): + self.supervisor.launch(run_id, version, self._spec("/definitely/missing/devsquad-worker"), "owner", "package") + self.assertEqual(self.store.run(run_id)["state"], "blocked") + self.assertIsNone(self.store.active_attempt(run_id)) + second, second_version = self._ready_run("after-missing") + reservation = self.store.reserve_attempt(second, second_version, "other-owner", "package") + self.assertTrue(reservation.attempt_token) + + def test_durable_outer_gate_failure_requeues_without_running_command(self): + run_id, version = self._ready_run("outer-gate-failure") + marker = self.root / "GATE_COMMAND_EXECUTED" + code = f"from pathlib import Path;Path({str(marker)!r}).write_text('executed')" + source = str(ROOT / "plugin" / "core" / "src") + with mock.patch.dict(os.environ, {"PYTHONPATH": source}), \ + mock.patch.object( + self.supervisor, + "_release_runner_gate", + side_effect=OSError("synthetic gate failure"), + ): + with self.assertRaisesRegex(OSError, "synthetic gate failure"): + self.supervisor.launch_durable( + run_id, + version, + self._spec(sys.executable, "-c", code), + "owner", + "package", + ) + run = self.store.run(run_id) + self.assertEqual((run["state"], run["phase"]), ("queued", None)) + self.assertEqual(self.store.attempt(run_id)["status"], "recovery_required") + self.assertFalse(marker.exists()) + + def test_output_is_bounded_but_full_stream_is_accounted(self): + run_id, version = self._ready_run("output") + code = "import sys;sys.stdout.write('o'*1000);sys.stderr.write('e'*2000)" + handle = self.supervisor.launch(run_id, version, self._spec(sys.executable, "-c", code), "owner", "package") + self.assertEqual(self.supervisor.wait(handle, 3), 0) + attempt = self.store.connection.execute("SELECT * FROM attempts WHERE run_id=?", (run_id,)).fetchone() + metadata = json.loads(attempt["output_metadata"]) + self.assertEqual(metadata["stdout"], { + "total_bytes": 1000, "captured_bytes": 32, "truncated": True, + "full_sha256": hashlib.sha256(b"o" * 1000).hexdigest(), + }) + self.assertEqual(metadata["stderr"]["total_bytes"], 2000) + self.assertEqual(metadata["stderr"]["captured_bytes"], 32) + for column in ("stdout_artifact_id", "stderr_artifact_id"): + artifact = self.store.connection.execute( + "SELECT path,byte_size FROM artifacts WHERE id=?", (attempt[column],), + ).fetchone() + self.assertEqual(artifact["byte_size"], 32) + self.assertEqual(Path(artifact["path"]).stat().st_size, 32) + + def test_database_fences_second_writer_for_same_worktree(self): + first_run, first_version = self._ready_run("writer-one") + handle = self.supervisor.launch( + first_run, first_version, self._spec(sys.executable, "-c", "import time;time.sleep(30)"), + "owner-one", "package", + ) + self.addCleanup(self._force_cleanup, handle) + second_run, second_version = self._ready_run("writer-two") + with self.assertRaises(ConflictError): + self.store.reserve_attempt(second_run, second_version, "owner-two", "package") + self.supervisor.cancel(first_run, handle) + self.assertEqual(self.store.run(first_run)["state"], "cancelled") + + def test_durable_timeout_is_failure_when_term_handler_exits_zero(self): + run_id, version = self._ready_run("timeout-zero") + code = ( + "import signal,sys,time;" + "signal.signal(signal.SIGTERM,lambda *_:sys.exit(0));" + "time.sleep(30)" + ) + spec = LaunchSpec( + 1, "fake", "cli_exec", (sys.executable, "-c", code), str(self.repo), None, 1, + ExecutionIdentity("fake", "1", "fixture", "fixture", "fixture", "low"), + ) + source = str(ROOT / "plugin" / "core" / "src") + with mock.patch.dict(os.environ, {"PYTHONPATH": source}): + handle = self.supervisor.launch_durable(run_id, version, spec, "owner", "package") + self.assertEqual(self.supervisor.wait_durable(handle, 1), 124) + self.assertEqual(self.store.run(run_id)["state"], "failed") + receipt_artifact = self.store.artifact_named(run_id, "result-receipt.json") + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertTrue(receipt["timed_out"]) + self.assertEqual(receipt["returncode"], 0) + self.assertEqual(receipt["error"], "TIMEOUT") + terminal = self.store.connection.execute( + "SELECT payload FROM events WHERE run_id=? AND type='run.failed'", (run_id,), + ).fetchone() + self.assertEqual(json.loads(terminal["payload"])["error"], "TIMEOUT") + self.assertEqual(self.store.attempt(run_id)["status"], "finished") + + def test_durable_launch_passes_an_opened_regular_stdin_artifact(self): + run_id,version=self._ready_run("durable-stdin") + stdin_path=self.root/"request input.json" + payload=b'{"request":"exact bytes"}\n' + stdin_path.write_bytes(payload) + code="import sys;sys.stdout.buffer.write(sys.stdin.buffer.read())" + spec=self._spec(sys.executable,"-c",code,stdin_path=str(stdin_path)) + source=str(ROOT/"plugin"/"core"/"src") + with mock.patch.dict(os.environ,{"PYTHONPATH":source}): + handle=self.supervisor.launch_durable(run_id,version,spec,"owner","package") + self.assertEqual(self.supervisor.wait_durable(handle,2),0) + attempt=self.store.attempt(run_id) + artifact=self.store.connection.execute( + "SELECT path FROM artifacts WHERE id=?",(attempt["stdout_artifact_id"],), + ).fetchone() + self.assertEqual(Path(artifact["path"]).read_bytes(),payload) + + linked_run,linked_version=self._ready_run("durable-stdin-symlink") + linked=self.root/"linked-input" + linked.symlink_to(stdin_path) + linked_spec=self._spec(sys.executable,"-c",code,stdin_path=str(linked)) + with self.assertRaises(ContractError): + self.supervisor.launch_durable( + linked_run,linked_version,linked_spec,"owner","package", + ) + self.assertEqual(self.store.run(linked_run)["state"],"blocked") + version = self.supervisor.cancel_orphan(linked_run) + self.assertEqual(self.store.run(linked_run)["state"], "cancelled") + self.assertEqual(self.store.run(linked_run)["version"], version) + self.assertIsNotNone(self.store.artifact_named(linked_run, "result-receipt.json")) + + def test_recovery_never_signals_an_ambiguous_identity(self): + run_id, version = self._ready_run("ambiguous") + handle = self.supervisor.launch( + run_id, version, self._spec(sys.executable, "-c", "import time;time.sleep(30)"), + "owner", "package", + ) + self.addCleanup(self._force_cleanup, handle) + with mock.patch("devsquad.supervisor.process_start_identity", return_value="different-start"), \ + mock.patch("devsquad.supervisor.os.killpg") as killpg: + self.assertEqual(self.supervisor.recover(run_id), "RECOVERY_REQUIRED") + killpg.assert_not_called() + self.assertEqual(self.store.run(run_id)["state"], "blocked") + attempt = self.store.connection.execute( + "SELECT status FROM attempts WHERE run_id=?", (run_id,), + ).fetchone() + self.assertEqual(attempt["status"], "ownership_ambiguous") + second, second_version = self._ready_run("after-ambiguous") + with self.assertRaises(ConflictError): + self.store.reserve_attempt(second, second_version, "other-owner", "package") + + def test_confirmed_dead_recovery_releases_writer_fence(self): + run_id, version = self._ready_run("dead") + handle = self.supervisor.launch( + run_id, version, self._spec(sys.executable, "-c", "pass"), "owner", "package", + ) + self.addCleanup(self._force_cleanup, handle) + handle.process.wait(timeout=2) + handle.stdout.thread.join(timeout=2) + handle.stderr.thread.join(timeout=2) + self.assertEqual(self.supervisor.recover(run_id), "RECOVERY_REQUIRED") + attempt = self.store.connection.execute( + "SELECT status FROM attempts WHERE run_id=?", (run_id,), + ).fetchone() + self.assertEqual(attempt["status"], "recovery_required") + second, second_version = self._ready_run("after-dead") + reservation = self.store.reserve_attempt(second, second_version, "other-owner", "package") + self.assertTrue(reservation.attempt_token) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py new file mode 100644 index 0000000..63635ab --- /dev/null +++ b/test/core/test_mcp.py @@ -0,0 +1,900 @@ +import contextlib +import asyncio +import io +import importlib.util +import json +import os +from pathlib import Path +import shlex +import shutil +import subprocess +import sys +import tempfile +import time +import tomllib +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +CORE = ROOT / "plugin/core" +sys.path.insert(0, str(CORE / "src")) + +from devsquad import cli, diagnostics, mcp_server +from devsquad.contracts import ContractError +from devsquad.integrations import ( + IntegrationTemplate, + LocalIntegrationManager, + load_integrations, +) +from devsquad.service import Service +from devsquad.store import ConflictError +from devsquad_test_fixtures import branch_review_routing_documents + + +class MCPDependencyBoundaryTest(unittest.TestCase): + def test_supported_sdk_is_an_exact_optional_dependency(self): + project = tomllib.loads((CORE / "pyproject.toml").read_text())["project"] + self.assertEqual(project["requires-python"], ">=3.11") + self.assertEqual(project["dependencies"], []) + self.assertEqual(project["optional-dependencies"]["mcp"], ["mcp==2.2.0"]) + self.assertEqual(mcp_server.MCP_SDK_REQUIREMENT, "mcp==2.2.0") + locked = { + line.strip().lower() + for line in (CORE / "requirements-mcp.lock").read_text().splitlines() + if line.strip() and not line.startswith("#") + } + self.assertIn("mcp==2.2.0", locked) + + def test_core_cli_import_does_not_import_optional_sdk(self): + probe = """ +import sys +sys.path.insert(0, {source!r}) +import devsquad.cli +assert not any(name == 'mcp' or name.startswith('mcp.') for name in sys.modules) +""".format(source=str(CORE / "src")) + subprocess.run([sys.executable, "-P", "-c", probe], check=True) + + def test_mcp_serve_dispatches_without_writing_protocol_stdout(self): + runtime = Path("/tmp/devsquad-mcp-boundary") + stdout, stderr = io.StringIO(), io.StringIO() + with mock.patch.object(mcp_server, "serve_stdio") as serve: + with contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): + code = cli.main(["mcp", "serve", "--runtime-dir", str(runtime)]) + self.assertEqual((code, stdout.getvalue(), stderr.getvalue()), (0, "", "")) + serve.assert_called_once_with( + runtime, caller_surface=None, caller_session_ref=None, + ) + + def test_missing_sdk_is_actionable_and_keeps_stdout_clean(self): + stdout, stderr = io.StringIO(), io.StringIO() + missing = mcp_server.MCPDependencyUnavailable("install devsquad-core[mcp]") + with mock.patch.object(mcp_server, "serve_stdio", side_effect=missing): + with contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): + code = cli.main(["mcp", "serve"]) + self.assertEqual(code, 69) + self.assertEqual(stdout.getvalue(), "") + self.assertIn("devsquad-core[mcp]", stderr.getvalue()) + + +class MCPIntegrationTemplateTest(unittest.TestCase): + def test_four_host_templates_render_absolute_argv_without_a_shell(self): + templates = {template.id: template for template in load_integrations()} + self.assertEqual(set(templates), { + "codex", "claude-code", "antigravity", "grok", + }) + host = Path(sys.executable) + squad = CORE / "bin/squad" + prefixes = { + "codex": ("mcp", "add", "devsquad", "--"), + "claude-code": ( + "mcp", "add", "--scope", "user", "devsquad", "--", + ), + "antigravity": ("mcp", "add", "devsquad", "--"), + "grok": ( + "mcp", "add", "--scope", "user", "devsquad", "--", + ), + } + for integration_id, template in templates.items(): + with self.subTest(integration=integration_id): + command = template.registration_command(host, squad) + resolved_host = str(host.resolve(strict=True)) + resolved_squad = str(squad.resolve(strict=True)) + self.assertEqual(command[0], resolved_host) + self.assertEqual(command[1:1 + len(prefixes[integration_id])], prefixes[integration_id]) + squad_index = command.index(resolved_squad) + self.assertEqual(command[squad_index + 1:squad_index + 4], ( + "mcp", "serve", "--surface", + )) + self.assertEqual(command[squad_index + 4], template.surface) + self.assertFalse(any("{" in argument for argument in command)) + inspection = template.inspection_command(host, squad) + self.assertEqual(inspection[0], resolved_host) + self.assertIn("mcp", inspection) + removal = template.removal_command(host) + if integration_id == "claude-code": + self.assertIsNotNone(removal) + self.assertIn("remove", removal) + else: + self.assertIsNone(removal) + + def test_template_schema_rejects_unknown_placeholders_and_fields(self): + with tempfile.TemporaryDirectory(prefix="devsquad-template-") as directory: + path = Path(directory) / "registration.json" + template = { + "schema_version": 1, + "id": "bad", + "display_name": "Bad", + "executable_paths": [], + "executable_names": ["bad"], + "server_name": "devsquad", + "surface": "bad", + "register_argv": ["{unknown}"], + "remove_argv": None, + "inspect_argv": ["{host_executable}"], + "inspect_format": "text", + } + path.write_text(json.dumps(template)) + with self.assertRaisesRegex(ContractError, "unknown integration placeholders"): + IntegrationTemplate.load(path) + template["unexpected"] = True + path.write_text(json.dumps(template)) + with self.assertRaisesRegex(ContractError, "fields differ"): + IntegrationTemplate.load(path) + + +class FakeMCPHost: + def __init__(self, template, home): + self.template = template + self.home = home + self.loaded = None + self.registration_calls = 0 + + def _config_path(self): + return { + "codex": self.home / ".codex/config.toml", + "claude-code": self.home / ".claude.json", + "antigravity": self.home / ".gemini/config/mcp_config.json", + "grok": self.home / ".grok/config.toml", + }[self.template.id] + + def seed_unrelated_config(self): + path = self._config_path() + path.parent.mkdir(parents=True, exist_ok=True) + if path.suffix == ".json": + path.write_text(json.dumps({"unrelated": {"credential": "preserve-me"}})) + else: + path.write_text('unrelated = "preserve-me"\n') + + def _save_registration(self, command, args): + path = self._config_path() + path.parent.mkdir(parents=True, exist_ok=True) + if path.suffix == ".json": + value = json.loads(path.read_text()) if path.exists() else {} + value.setdefault("mcpServers", {})["devsquad"] = { + "command": command, + "args": args, + } + if self.template.id == "antigravity": + value["mcpServers"]["devsquad"]["disabled"] = False + path.write_text(json.dumps(value)) + else: + existing = path.read_text() if path.exists() else "" + if "[mcp_servers.devsquad]" not in existing: + enabled = "enabled = true\n" if self.template.id == "grok" else "" + path.write_text( + existing + + "\n[mcp_servers.devsquad]\n" + + f"command = {json.dumps(command)}\n" + + f"args = {json.dumps(args)}\n" + + enabled + ) + + def _inspection_result(self, argv): + if self.loaded is None: + if self.template.id == "grok": + return subprocess.CompletedProcess(argv, 0, "[]\n", "") + if self.template.id == "antigravity": + return subprocess.CompletedProcess( + argv, 0, "NAME TYPE STATUS COMMAND/URL\n", "", + ) + return subprocess.CompletedProcess(argv, 1, "", "not found") + command, args = self.loaded + if self.template.id == "codex": + stdout = json.dumps({ + "name": "devsquad", + "enabled": True, + "transport": {"command": command, "args": args, "env": None}, + }) + elif self.template.id == "grok": + stdout = json.dumps([{ + "name": "devsquad", "enabled": True, "scope": "user", + "command": command, "args": args, + }]) + elif self.template.id == "claude-code": + stdout = ( + "devsquad:\n" + " Scope: User config (available in all your projects)\n" + " Status: ✓ Connected\n" + " Type: stdio\n" + f" Command: {command}\n" + f" Args: {shlex.join(args)}\n" + " Environment: PRIVATE_TOKEN=not-reported\n" + ) + else: + stdout = ( + "NAME TYPE STATUS COMMAND/URL\n" + f"devsquad stdio enabled {shlex.join([command, *args])}\n" + ) + return subprocess.CompletedProcess(argv, 0, stdout, "") + + def __call__(self, argv, **_): + argv = tuple(argv) + if "remove" in argv: + self.loaded = None + path = self._config_path() + value = json.loads(path.read_text()) + value.get("mcpServers", {}).pop("devsquad", None) + path.write_text(json.dumps(value)) + return subprocess.CompletedProcess(argv, 0, "removed\n", "") + if "add" not in argv: + return self._inspection_result(argv) + self.registration_calls += 1 + delimiter = argv.index("--") + command = argv[delimiter + 1] + args = list(argv[delimiter + 2:]) + self.loaded = (command, args) + self._save_registration(command, args) + return subprocess.CompletedProcess(argv, 0, "registered\n", "") + + +class LocalMCPRegistrationTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-local-mcp-") + self.root = Path(self.temp.name) + self.home = self.root / "home" + self.project = self.root / "project" + self.home.mkdir() + self.project.mkdir() + self.squad = CORE / "bin/squad" + + def tearDown(self): + self.temp.cleanup() + + def manager(self, fake): + return LocalIntegrationManager( + project=self.project, + home=self.home, + squad_executable=self.squad, + which=lambda _: sys.executable, + runner=fake, + mcp_sdk_available=True, + mcp_sdk_version="2.2.0", + ) + + def test_setup_is_idempotent_for_every_host_and_preserves_unrelated_config(self): + for template in load_integrations(): + with self.subTest(host=template.id): + fake = FakeMCPHost(template, self.home) + fake.seed_unrelated_config() + manager = self.manager(fake) + + first = manager.setup(template) + second = manager.setup(template) + + self.assertEqual(first["action"], "added") + self.assertTrue(first["ready"]) + self.assertEqual(second["action"], "unchanged") + self.assertTrue(second["ready"]) + self.assertEqual(fake.registration_calls, 1) + self.assertEqual(len(second["sources"]), 1) + self.assertEqual(second["loaded"]["args"], [ + "mcp", "serve", "--surface", template.surface, + ]) + self.assertNotIn("PRIVATE_TOKEN", json.dumps(second)) + self.assertIn("preserve-me", fake._config_path().read_text()) + + fake._config_path().unlink() + + def test_duplicate_and_inherited_registrations_fail_closed_without_mutation(self): + template = next(item for item in load_integrations() if item.id == "codex") + expected = ( + str(self.squad.resolve()), + ["mcp", "serve", "--surface", template.surface], + ) + fake = FakeMCPHost(template, self.home) + fake.loaded = expected + fake._save_registration(*expected) + project_config = self.project / ".codex/config.toml" + project_config.parent.mkdir(parents=True) + project_config.write_text( + "[mcp_servers.devsquad]\n" + f"command = {json.dumps(expected[0])}\n" + f"args = {json.dumps(expected[1])}\n" + ) + duplicate = self.manager(fake).setup(template) + self.assertEqual(duplicate["status"], "duplicate") + self.assertEqual(duplicate["action"], "blocked_duplicate") + self.assertEqual(fake.registration_calls, 0) + fake._config_path().unlink() + project_config.unlink() + inherited = self.manager(fake).setup(template) + self.assertEqual(inherited["status"], "inherited") + self.assertEqual(inherited["action"], "blocked_inherited") + self.assertEqual(fake.registration_calls, 0) + + fake._save_registration(*expected) + fake.loaded = ("/inherited/override", ["mcp", "serve"]) + overlaid = self.manager(fake).setup(template) + self.assertEqual(overlaid["status"], "duplicate") + self.assertEqual(overlaid["action"], "blocked_duplicate") + self.assertEqual(fake.registration_calls, 0) + + def test_doctor_data_redacts_drifted_arguments_and_malformed_config_content(self): + template = next( + item for item in load_integrations() if item.id == "claude-code" + ) + fake = FakeMCPHost(template, self.home) + fake.loaded = (str(self.squad.resolve()), ["--api-key", "super-secret"]) + fake._save_registration(*fake.loaded) + drifted = self.manager(fake).inspect(template) + encoded = json.dumps(drifted) + self.assertEqual(drifted["status"], "drifted") + self.assertIsNone(drifted["loaded"]["args"]) + self.assertIsNone(drifted["sources"][0]["args"]) + self.assertNotIn("super-secret", encoded) + + updated = self.manager(fake).setup(template) + self.assertEqual(updated["action"], "updated") + self.assertTrue(updated["ready"]) + self.assertEqual(updated["removal_exit_code"], 0) + self.assertEqual(fake.registration_calls, 1) + + fake._config_path().write_text('{"private":"do-not-report"') + fake.loaded = None + malformed = self.manager(fake).inspect(template) + self.assertEqual(malformed["status"], "invalid_config") + self.assertNotIn("do-not-report", json.dumps(malformed)) + + def test_setup_requires_the_exact_supported_optional_sdk(self): + template = next(item for item in load_integrations() if item.id == "grok") + fake = FakeMCPHost(template, self.home) + missing = LocalIntegrationManager( + project=self.project, + home=self.home, + squad_executable=self.squad, + which=lambda _: sys.executable, + runner=fake, + mcp_sdk_available=False, + ).setup(template) + self.assertEqual(missing["action"], "blocked_missing_mcp_sdk") + unsupported = LocalIntegrationManager( + project=self.project, + home=self.home, + squad_executable=self.squad, + which=lambda _: sys.executable, + runner=fake, + mcp_sdk_available=True, + mcp_sdk_version="2.1.0", + ).setup(template) + self.assertEqual(unsupported["action"], "blocked_unsupported_mcp_sdk") + self.assertEqual(fake.registration_calls, 0) + + def test_explicit_missing_launcher_does_not_silently_fall_back(self): + template = next(item for item in load_integrations() if item.id == "codex") + fake = FakeMCPHost(template, self.home) + manager = LocalIntegrationManager( + project=self.project, + home=self.home, + squad_executable=self.root / "missing-squad", + which=lambda _: sys.executable, + runner=fake, + mcp_sdk_available=True, + mcp_sdk_version="2.2.0", + ) + result = manager.setup(template) + self.assertEqual(result["status"], "unstable_launcher") + self.assertEqual(result["action"], "blocked_unstable_launcher") + self.assertIsNone(result["expected"]["command"]) + self.assertEqual(fake.registration_calls, 0) + + +class MCPDoctorReportTest(unittest.TestCase): + def test_installed_app_drift_controls_readiness_but_unavailable_apps_do_not(self): + manager = mock.Mock( + mcp_sdk_available=True, + mcp_sdk_supported=True, + mcp_sdk_version="2.2.0", + squad_executable=CORE / "bin/squad", + launcher_error=None, + ) + rows = [ + {"id": "codex", "installed": True, "ready": True}, + {"id": "claude-code", "installed": False, "ready": False}, + ] + manager.inspect.side_effect = rows + templates = (mock.Mock(id="codex"), mock.Mock(id="claude-code")) + adapters = [{ + "adapter": "codex", "status": "supported", "installed": True, + "supported": True, "authenticated": True, "ready": True, + "operation_verified": None, + }] + with ( + mock.patch.object(diagnostics, "_adapter_rows", return_value=adapters), + mock.patch.object(diagnostics, "load_integrations", return_value=templates), + ): + report = diagnostics.build_doctor_report(project=ROOT, manager=manager) + self.assertTrue(report["ready"]) + self.assertEqual(report["local_app_access"]["installed_count"], 1) + self.assertEqual(report["local_app_access"]["configured_count"], 1) + + manager.inspect.side_effect = [ + {"id": "codex", "installed": True, "ready": False}, rows[1], + ] + with ( + mock.patch.object(diagnostics, "_adapter_rows", return_value=adapters), + mock.patch.object(diagnostics, "load_integrations", return_value=templates), + ): + drifted = diagnostics.build_doctor_report(project=ROOT, manager=manager) + self.assertFalse(drifted["ready"]) + self.assertFalse(drifted["local_app_access"]["ready"]) + + +class MCPBridgeTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-mcp-bridge-") + self.runtime = Path(self.temp.name) / "runtime" + (self.runtime / "artifacts").mkdir(parents=True) + self.service = mock.Mock() + self.bridge = mcp_server.MCPBridge( + self.runtime, self.service, environment={}, + ) + + def tearDown(self): + self.temp.cleanup() + + def assert_success(self, payload, data): + self.assertEqual(payload, { + "schema_version": 1, + "ok": True, + "data": data, + "error": None, + }) + + def test_operations_map_directly_to_the_saved_run_service(self): + task = {"schema_version": 1} + recovery = {"attempt_id": "attempt-1", "disposition": "confirm_dead"} + claim = {"run_id": "run-1", "fencing_token": 3} + decision = {"disposition": "accept"} + calls = [ + ( + lambda: self.bridge.start(task, "key-1", "run-0"), + "start", (task, "key-1", "run-0"), + {"run_id": "run-1", "state": "queued"}, + ), + ( + lambda: self.bridge.status("run-1"), + "status", ("run-1",), {"run_id": "run-1", "state": "running"}, + ), + ( + lambda: self.bridge.events("run-1", 4, 25), + "events", ("run-1", 4, 25), {"events": [], "next_cursor": 4}, + ), + ( + lambda: self.bridge.cancel("run-1"), + "cancel", ("run-1",), {"run_id": "run-1", "state": "cancelling"}, + ), + ( + lambda: self.bridge.resume("run-1", recovery), + "resume", ("run-1", recovery), {"run_id": "run-1", "state": "running"}, + ), + ( + lambda: self.bridge.handoff_claim("run-1", 7, "claude", claim), + "handoff_claim", ("run-1", 7, "claude", claim), + {"run_id": "run-1", "action": "renewed"}, + ), + ( + lambda: self.bridge.handoff_complete("run-1", claim, decision), + "handoff_complete", ("run-1", claim, decision), + {"run_id": "run-1", "state": "succeeded"}, + ), + ] + for invoke, method_name, expected_args, response in calls: + with self.subTest(operation=method_name): + method = getattr(self.service, method_name) + method.return_value = response + self.assert_success(invoke(), response) + method.assert_called_once_with(*expected_args) + method.reset_mock() + + def test_doctor_uses_the_shared_read_only_report(self): + report = {"core_version": "0.1.0", "ready": True, "local_apps": []} + with mock.patch.object(mcp_server, "build_doctor_report", return_value=report) as doctor: + self.assert_success(self.bridge.doctor(), report) + doctor.assert_called_once_with(project=Path.cwd().resolve()) + + def test_contract_conflict_and_internal_failures_keep_machine_envelopes(self): + cases = [ + (ContractError("bad request"), "INPUT_INVALID"), + (ConflictError("stale claim"), "CONFLICT"), + (OSError("disk unavailable"), "INTERNAL_ERROR"), + ] + for failure, code in cases: + with self.subTest(code=code): + self.service.status.side_effect = failure + payload = self.bridge.status("run-1") + self.assertEqual(payload["schema_version"], 1) + self.assertFalse(payload["ok"]) + self.assertIsNone(payload["data"]) + self.assertEqual(payload["error"]["code"], code) + self.assertEqual(payload["error"]["message"], str(failure)) + + def test_result_caps_preview_bytes_across_saved_artifacts(self): + first = self.runtime / "artifacts" / "receipt.json" + second = self.runtime / "artifacts" / "events.jsonl" + first.write_text("abcdef") + second.write_text("ghijkl") + artifacts = [ + {"id": "a-1", "name": first.name, "path": str(first), "sha256": "1" * 64, "byte_size": 6}, + {"id": "a-2", "name": second.name, "path": str(second), "sha256": "2" * 64, "byte_size": 6}, + ] + self.service.result.return_value = { + "run_id": "run-1", "ready": True, "state": "succeeded", "artifacts": artifacts, + } + payload = self.bridge.result("run-1", preview_bytes=5) + self.assertTrue(payload["ok"]) + result = payload["data"] + self.assertEqual(result["preview_bytes"], 5) + self.assertEqual(result["artifacts"][0]["preview_text"], "abcde") + self.assertTrue(result["artifacts"][0]["preview_truncated"]) + self.assertIsNone(result["artifacts"][1]["preview_text"]) + self.assertTrue(result["artifacts"][1]["preview_truncated"]) + self.assertEqual( + {key: result["artifacts"][0][key] for key in ("id", "path", "sha256")}, + {"id": "a-1", "path": str(first), "sha256": "1" * 64}, + ) + + too_large = self.bridge.result( + "run-1", preview_bytes=mcp_server.MAX_ARTIFACT_PREVIEW_BYTES + 1, + ) + self.assertFalse(too_large["ok"]) + self.assertEqual(too_large["error"]["code"], "INPUT_INVALID") + self.assertEqual(self.service.result.call_count, 1) + + def test_result_rejects_preview_path_outside_the_runtime(self): + outside = Path(self.temp.name) / "outside.txt" + outside.write_text("not a saved runtime artifact") + self.service.result.return_value = { + "run_id": "run-1", + "ready": True, + "state": "succeeded", + "artifacts": [{ + "id": "a-1", "name": outside.name, "path": str(outside), + "sha256": "1" * 64, "byte_size": outside.stat().st_size, + }], + } + payload = self.bridge.result("run-1") + self.assertFalse(payload["ok"]) + self.assertEqual(payload["error"]["code"], "CONFLICT") + self.assertIn("escapes", payload["error"]["message"]) + + def test_configured_origin_is_saved_as_provenance_not_authorization(self): + task = {"schema_version": 1, "origin": {"surface": "user-label"}} + self.service.start.return_value = { + "run_id": "run-1", "state": "queued", "created": True, + } + bridge = mcp_server.MCPBridge( + self.runtime, + self.service, + caller_surface="codex-app", + caller_session_ref="thread-7", + environment={}, + ) + payload = bridge.start(task, "key-1") + self.assertTrue(payload["ok"]) + submitted = self.service.start.call_args.args[0] + self.assertEqual(submitted["origin"], { + "surface": "codex-app", "session_ref": "thread-7", + }) + self.assertEqual(task["origin"], {"surface": "user-label"}) + + self.service.reset_mock() + user_label_only = mcp_server.MCPBridge( + self.runtime, + self.service, + caller_surface="worker", + environment={}, + ) + self.service.start.return_value = { + "run_id": "run-2", "state": "queued", "created": True, + } + self.assertTrue(user_label_only.start(task, "key-2")["ok"]) + self.service.start.assert_called_once() + + def test_worker_environment_rejects_every_mutation_but_allows_inspection(self): + worker_bridge = mcp_server.MCPBridge( + self.runtime, + self.service, + environment={ + "DEVSQUAD_WORKER": "1", + "DEVSQUAD_RUN_ID": "run-1", + "DEVSQUAD_DELEGATION_DEPTH": "1", + }, + ) + mutations = [ + lambda: worker_bridge.start({"schema_version": 1}, "key-1"), + lambda: worker_bridge.cancel("run-1"), + lambda: worker_bridge.resume("run-1"), + lambda: worker_bridge.handoff_claim("run-1", 2, "host"), + lambda: worker_bridge.handoff_complete("run-1", {}, {}), + ] + for mutate in mutations: + with self.subTest(mutation=mutate): + payload = mutate() + self.assertFalse(payload["ok"]) + self.assertEqual(payload["error"]["code"], "POLICY_DENIED") + for method in ( + self.service.start, + self.service.cancel, + self.service.resume, + self.service.handoff_claim, + self.service.handoff_complete, + ): + method.assert_not_called() + + self.service.status.return_value = { + "run_id": "run-1", "state": "running", "version": 2, + } + self.assertTrue(worker_bridge.status("run-1")["ok"]) + self.service.status.assert_called_once_with("run-1") + + +@unittest.skipUnless(importlib.util.find_spec("mcp"), "optional MCP SDK is not installed") +class OfficialSDKConformanceTest(unittest.TestCase): + def test_tool_schemas_and_calls_use_the_official_in_memory_transport(self): + from mcp import Client + + service = mock.Mock() + service.status.return_value = {"run_id": "run-1", "state": "running", "version": 2} + service.start.return_value = {"run_id": "run-1", "state": "queued", "created": True} + service.cancel.return_value = {"run_id": "run-1", "state": "cancelling", "version": 3} + with tempfile.TemporaryDirectory(prefix="devsquad-sdk-server-") as directory: + server = mcp_server.build_server( + Path(directory), service, environment={}, + ) + + async def probe(): + async with Client(server) as client: + listing = await client.list_tools() + tools = {tool.name: tool for tool in listing.tools} + self.assertEqual(set(tools), { + "squad_doctor", "squad_start", "squad_status", "squad_events", "squad_result", + "squad_cancel", "squad_resume", "squad_handoff_claim", + "squad_handoff_complete", + }) + self.assertEqual( + tools["squad_status"].input_schema["required"], ["run_id"], + ) + status_annotations = tools["squad_status"].annotations.model_dump( + by_alias=True, + ) + self.assertEqual(status_annotations, { + "title": None, + "readOnlyHint": True, + "destructiveHint": False, + "idempotentHint": True, + "openWorldHint": False, + }) + cancel_annotations = tools["squad_cancel"].annotations.model_dump( + by_alias=True, + ) + self.assertTrue(cancel_annotations["destructiveHint"]) + self.assertFalse(cancel_annotations["readOnlyHint"]) + status = await client.call_tool("squad_status", {"run_id": "run-1"}) + self.assertFalse(status.is_error) + self.assertEqual(status.structured_content["data"]["version"], 2) + started = await client.call_tool("squad_start", { + "task": {"schema_version": 1}, "idempotency_key": "key-1", + }) + self.assertFalse(started.is_error) + self.assertEqual(started.structured_content["data"]["run_id"], "run-1") + cancelled = await client.call_tool("squad_cancel", {"run_id": "run-1"}) + self.assertFalse(cancelled.is_error) + self.assertEqual(cancelled.structured_content["data"]["state"], "cancelling") + malformed = await client.call_tool("squad_events", { + "run_id": "run-1", "after": 0, "limit": "not-an-integer", + }) + self.assertTrue(malformed.is_error) + + asyncio.run(probe()) + + def test_closing_a_real_stdio_client_does_not_cancel_the_detached_worker(self): + from mcp import Client, StdioServerParameters + + with tempfile.TemporaryDirectory(prefix="devsquad-sdk-disconnect-") as directory: + root = Path(directory) + repo = root / "repo" + runtime = root / "runtime" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + subprocess.run( + ["git", "-C", str(repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(repo), "config", "user.name", "Test"], + check=True, + ) + (repo / "src").mkdir() + (repo / "tests").mkdir() + (repo / "src/app.py").write_text("VALUE = 'fixture'\n") + (repo / "tests/test_app.py").write_text("# fixture\n") + profiles, policy = branch_review_routing_documents() + (repo / "profiles.json").write_text(profiles) + (repo / "policy.json").write_text(policy) + subprocess.run(["git", "-C", str(repo), "add", "."], check=True) + subprocess.run( + ["git", "-C", str(repo), "commit", "-qm", "fixture"], + check=True, + ) + task = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + task["project"] = { + "repo_path": str(repo), "base_ref": "HEAD", "target_ref": "HEAD", + } + task["routing"] = { + "profiles_file": "profiles.json", "policy_file": "policy.json", + } + service = Service(runtime) + started = service.start( + task, "stdio-client-disconnect", _internal_fake_delay=2, + ) + + async def observe_then_disconnect(): + parameters = StdioServerParameters( + command=sys.executable, + args=[ + str(CORE / "bin/squad"), "mcp", "serve", + "--runtime-dir", str(runtime), "--surface", "codex-app", + ], + cwd=ROOT, + ) + async with Client(parameters) as client: + status = await client.call_tool( + "squad_status", {"run_id": started["run_id"]}, + ) + self.assertFalse(status.is_error) + self.assertEqual( + status.structured_content["data"]["run_id"], started["run_id"], + ) + self.assertIn( + status.structured_content["data"]["state"], {"queued", "running"}, + ) + + asyncio.run(observe_then_disconnect()) + deadline = time.monotonic() + 8 + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + if status["state"] == "succeeded": + break + time.sleep(0.05) + self.assertEqual(service.status(started["run_id"])["state"], "succeeded") + + +class InstalledWheelMCPBoundaryTest(unittest.TestCase): + @staticmethod + def build_python(): + candidates = [ + os.environ.get("DEVSQUAD_BUILD_PYTHON"), + sys.executable, + str(Path.home() / ".cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3"), + shutil.which("python3.13"), + shutil.which("python3.12"), + shutil.which("python3.11"), + ] + for candidate in dict.fromkeys(value for value in candidates if value): + try: + result = subprocess.run( + [candidate, "-c", "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68"], + text=True, + capture_output=True, + ) + except OSError: + continue + if result.returncode == 0: + return candidate + return None + + def test_build_python_skips_missing_first_candidate_for_supported_interpreter(self): + missing = str(Path(tempfile.mkdtemp(prefix="devsquad-missing-")) / "no-such-python") + probed = [] + + def fake_run(argv, **kwargs): + probed.append(argv[0]) + if argv[0] == missing: + raise FileNotFoundError(argv[0]) + return subprocess.CompletedProcess(argv, 0) + + with ( + mock.patch.dict(os.environ, {"DEVSQUAD_BUILD_PYTHON": missing}), + mock.patch("subprocess.run", side_effect=fake_run), + ): + result = self.build_python() + self.assertEqual(probed[0], missing) + self.assertEqual(result, sys.executable) + + def test_plain_installed_wheel_keeps_cli_usable_without_mcp(self): + build_python = self.build_python() + if build_python is None: + self.skipTest("offline wheel gate requires setuptools>=68 and wheel") + with tempfile.TemporaryDirectory(prefix="devsquad-mcp-wheel-") as directory: + root = Path(directory) + source = root / "core" + shutil.copytree(CORE, source) + wheels = root / "wheels" + wheels.mkdir() + subprocess.run( + [ + build_python, "-m", "pip", "wheel", str(source), + "--wheel-dir", str(wheels), "--no-index", "--no-deps", "--no-build-isolation", + ], + check=True, + text=True, + capture_output=True, + ) + wheel = next(wheels.glob("devsquad_core-*.whl")) + environment = os.environ.copy() + environment.pop("PYTHONPATH", None) + venv = root / "venv" + subprocess.run([build_python, "-m", "venv", str(venv)], check=True, env=environment) + python = venv / ("Scripts/python.exe" if os.name == "nt" else "bin/python") + squad = venv / ("Scripts/squad.exe" if os.name == "nt" else "bin/squad") + subprocess.run( + [str(python), "-m", "pip", "install", "--no-index", "--no-deps", str(wheel)], + check=True, + text=True, + capture_output=True, + env=environment, + ) + version = subprocess.run( + [str(squad), "--version"], check=True, text=True, capture_output=True, env=environment, + ) + self.assertEqual(version.stdout.strip(), "squad 0.1.0") + integrations = subprocess.run( + [ + str(python), "-P", "-c", + "from devsquad.integrations import load_integrations; " + "print(','.join(sorted(item.id for item in load_integrations())))", + ], + check=True, + text=True, + capture_output=True, + cwd=root, + env=environment, + ) + self.assertEqual( + integrations.stdout.strip(), + "antigravity,claude-code,codex,grok", + ) + schemas = subprocess.run( + [ + str(python), "-P", "-c", + "from pathlib import Path; import sys; " + "print(','.join(sorted(path.name for path in " + "(Path(sys.prefix) / 'share/devsquad/schemas').glob('*.schema.json'))))", + ], + check=True, text=True, capture_output=True, cwd=root, env=environment, + ) + self.assertEqual( + schemas.stdout.strip(), + ",".join(sorted(path.name for path in (CORE / "schemas").glob("*.schema.json"))), + ) + missing = subprocess.run( + [str(squad), "mcp", "serve"], text=True, capture_output=True, env=environment, + ) + self.assertEqual(missing.returncode, 69) + self.assertEqual(missing.stdout, "") + self.assertIn("devsquad-core[mcp]", missing.stderr) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_native_catalog.py b/test/core/test_native_catalog.py new file mode 100644 index 0000000..e30e003 --- /dev/null +++ b/test/core/test_native_catalog.py @@ -0,0 +1,112 @@ +import json +from datetime import datetime, timedelta, timezone +from pathlib import Path +import sys +import tempfile +import threading +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.native_catalog import NativeCatalogCache, native_account_pool, native_scope, normalize_codex_limits +from devsquad.capacity import derive_pool_capacity + + +NOW = datetime(2026, 10, 2, tzinfo=timezone.utc) +MODELS = [{"id": "gpt-test", "supportedReasoningEfforts": ["low"]}] +TARGET = {"harness": "codex", "model_family": "gpt", "model_id": "gpt-test"} + + +class NativeCatalogTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.cache = NativeCatalogCache(Path(self.temp.name), "scope-a", "v-test") + + def test_last_good_ttl_failure_backoff_and_drift(self): + first = self.cache.refresh(lambda: MODELS, now=NOW) + self.assertTrue(first["complete"]) + self.assertEqual(self.cache.refresh(lambda: self.fail("fresh cache queried"), now=NOW), first) + def failure(): + raise ContractError("AUTH_ERROR private provider diagnostics") + later = NOW + timedelta(days=2) + failed = self.cache.refresh(failure, now=later) + self.assertEqual(failed["models"], first["models"]) + self.assertEqual(failed["last_refresh"]["status"], "error") + self.assertNotIn("private provider", json.dumps(failed)) + self.cache.refresh(lambda: self.fail("backoff ignored"), now=later + timedelta(seconds=10)) + changed = self.cache.refresh(lambda: [], now=later + timedelta(minutes=3)) + self.assertEqual(changed["catalog_change"]["removed_model_ids"], ["gpt-test"]) + + def test_concurrent_refresh_and_dead_owner_release(self): + self.cache.refresh(lambda: MODELS, now=NOW) + entered, release = threading.Event(), threading.Event() + result = [] + def slow(): + entered.set() + self.assertTrue(release.wait(5)) + return MODELS + owner = threading.Thread(target=lambda: result.append(self.cache.refresh(slow, now=NOW + timedelta(days=2)))) + owner.start() + self.assertTrue(entered.wait(5)) + try: + cached = self.cache.refresh(lambda: self.fail("second refresh owner"), now=NOW + timedelta(days=2)) + self.assertEqual(cached["models"][0]["id"], "gpt-test") + finally: + release.set() + owner.join(5) + self.assertEqual(len(result), 1) + def interrupted(): + raise KeyboardInterrupt() + with self.assertRaises(KeyboardInterrupt): + self.cache.refresh(interrupted, now=NOW + timedelta(days=4)) + self.assertTrue(self.cache.refresh(lambda: MODELS, now=NOW + timedelta(days=4))["complete"]) + + def test_scope_is_private_and_changes_for_account_config_binary_version(self): + account = {"account": {"type": "chatgpt", "email": "private@example.invalid", "planType": "plus"}} + scope = native_scope(account, {"provider": "native"}, "/binary/a", "v1") + pool = native_account_pool(account) + self.assertEqual(pool, native_account_pool({"account": {**account["account"], "planType": "pro"}})) + self.assertNotEqual(pool, native_account_pool({"account": {"type": "chatgpt", "email": "other@example.invalid"}})) + for a, c, b, v in ( + ({"account": {"type": "chatgpt", "email": "other@example.invalid"}}, {}, "/binary/a", "v1"), + (account, {"provider": "changed"}, "/binary/a", "v1"), + (account, {"provider": "native"}, "/binary/b", "v1"), + (account, {"provider": "native"}, "/binary/a", "v2"), + ): + self.assertNotEqual(scope, native_scope(a, c, b, v)) + self.assertNotIn("private", scope) + for account in ({"account": None}, {"account": {"type": "apiKey"}}, {"account": {"type": "chatgpt"}}): + with self.assertRaises(ContractError): + native_scope(account, {}, "/binary/a", "v1") + other = NativeCatalogCache(Path(self.temp.name), "scope-b", "v-test") + with self.assertRaises(ContractError): + other.refresh(lambda: (_ for _ in ()).throw(TimeoutError()), now=NOW) + with self.assertRaisesRegex(ContractError, "backing off"): + other.refresh(lambda: self.fail("initial failure backoff ignored"), now=NOW + timedelta(seconds=1)) + + def test_provider_default_hint_does_not_change_capability_fingerprint(self): + first = self.cache.refresh(lambda: [{**MODELS[0], "isDefault": True}], now=NOW) + changed = self.cache.refresh(lambda: [{**MODELS[0], "isDefault": False}], now=NOW + timedelta(days=2)) + self.assertEqual(first["models"][0]["fingerprint"], changed["models"][0]["fingerprint"]) + self.assertEqual(changed["catalog_change"]["changed_model_ids"], []) + + def test_weekly_limit_blocks_available_primary_and_null_is_unknown(self): + payload = {"rateLimitsByLimitId": {"codex": { + "primary": {"usedPercent": 10, "windowDurationMins": 300, "resetsAt": int((NOW + timedelta(hours=5)).timestamp())}, + "secondary": {"usedPercent": 100, "windowDurationMins": 10080, "resetsAt": int((NOW + timedelta(days=5)).timestamp())}, + }}} + observations = normalize_codex_limits(payload, "pool", now=NOW) + self.assertEqual(derive_pool_capacity("pool", observations, target=TARGET, now=NOW)["status"], "exhausted") + self.assertEqual(derive_pool_capacity("pool", observations, target=TARGET, now=NOW + timedelta(minutes=2))["status"], "unknown") + unknown = normalize_codex_limits({"rateLimitsByLimitId": {}, "rateLimits": payload["rateLimitsByLimitId"]["codex"]}, "pool", now=NOW) + self.assertEqual(derive_pool_capacity("pool", unknown, target=TARGET, now=NOW)["status"], "unknown") + self.assertTrue(all(o["used"] is None for o in unknown)) + malformed = normalize_codex_limits({"rateLimits": {"primary": {"usedPercent": True}}}, "pool", now=NOW) + self.assertEqual(derive_pool_capacity("pool", malformed, target=TARGET, now=NOW)["status"], "unknown") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_objective_outcomes.py b/test/core/test_objective_outcomes.py new file mode 100644 index 0000000..021bf94 --- /dev/null +++ b/test/core/test_objective_outcomes.py @@ -0,0 +1,161 @@ +"""Public terminal-origin regressions: no manually imported final outcomes.""" +import copy +from datetime import datetime, timezone +from pathlib import Path +import sys +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path[:0] = [str(ROOT / "plugin/core/src"), str(ROOT / "test/core")] + +from devsquad.store import ConflictError, Store +import test_review_runtime as review_fixtures + + +class ObjectiveOutcomeTest(unittest.TestCase): + def setUp(self): + self.fixture = review_fixtures.DurableBranchReviewTest() + self.fixture.setUp() + self.addCleanup(self.fixture.doCleanups) + self.service = self.fixture.service + self.fixture.task["checks"][0]["argv"] = [sys.executable, "-c", "print('verified')"] + + def outcome(self, run_id): + store = self.service._store() + try: + rows = store.outcomes_for_run(run_id) + self.assertEqual(len(rows), 1, "public terminal run must project exactly one final outcome") + return rows[0]["outcome"] + finally: + store.close() + + def accept(self, run_id, waiting): + claimed = self.service.handoff_claim(run_id, waiting["version"], "objective-fixture-host") + decision = self.fixture.decision(claimed["handoff"]["packet"], "objective-accept", "accept", "Verified objective fixture evidence.") + return self.service.handoff_complete(run_id, claimed["claim"], decision) + + def test_preparation_failure_projects_missing_evidence_truthfully(self): + task = copy.deepcopy(self.fixture.task) + task["project"]["target_ref"] = "nonexistent-objective-target" + started = self.service.start(task, "objective-preparation-failure") + self.assertEqual(started["state"], "failed") + outcome = self.outcome(started["run_id"]) + self.assertEqual(outcome["verdict"], "failed") + self.assertEqual(outcome["contributions"], []) + self.assertTrue(all(c["status"] == "unknown" and not c["evidence_refs"] for c in outcome["criteria"])) + + def test_prelaunch_cancel_has_no_completed_exposure(self): + with mock.patch.object(self.service, "_spawn_daemon", return_value=0): + started = self.service.start(self.fixture.task, "objective-prelaunch-cancel", _internal_review_fixture=self.fixture.fixture) + self.service.cancel(started["run_id"]) + outcome = self.outcome(started["run_id"]) + self.assertEqual(outcome["verdict"], "cancelled") + self.assertEqual(outcome["contributions"], []) + self.assertEqual(self.outcome(started["run_id"]), outcome) + + def test_worker_failure_is_not_independent_success(self): + self.fixture.task["checks"][0]["cwd"] = "missing-check-directory" + started = self.service.start(self.fixture.task, "objective-worker-failure", _internal_review_fixture=self.fixture.fixture) + self.fixture.wait_state(started["run_id"], {"failed"}) + outcome = self.outcome(started["run_id"]) + self.assertEqual(outcome["verdict"], "failed") + self.assertTrue(outcome["contributions"]) + self.assertTrue(all(c["result"] == "failed" and not c["independent_success"] for c in outcome["contributions"])) + + def test_host_completion_and_late_correction_are_append_only(self): + run_id, waiting = self.fixture.start_waiting("objective-host") + self.assertEqual(self.accept(run_id, waiting)["state"], "succeeded") + final = self.outcome(run_id) + self.assertEqual(final["verdict"], "succeeded") + self.assertTrue(final["contributions"]) + correction = {**final, "outcome_id": "objective-late-correction", "kind": "late_correction", "verdict": "escaped_defect", + "corrects_outcome_id": final["outcome_id"], "observed_at": datetime.now(timezone.utc).isoformat(), "summary": "Explicit later escaped-defect evidence."} + self.service.outcome_add(run_id, correction) + self.service.result(run_id) + store = self.service._store() + try: + rows = store.outcomes_for_run(run_id) + self.assertEqual(len(rows), 2) + self.assertEqual(rows[0]["outcome"], final) + finally: + store.close() + + def test_headless_completion_projects_without_manual_import(self): + self.fixture.configure_fixture_headless() + started = self.service.start(self.fixture.task, "objective-headless", _internal_review_fixture=self.fixture.fixture, + _internal_lead_fixture={"disposition": "accept", "reason": "Verified fixture evidence."}) + completed = self.fixture.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "succeeded") + outcome = self.outcome(started["run_id"]) + self.assertEqual({c["role"] for c in outcome["contributions"]}, {"reviewer", "lead"}) + + def test_projection_crash_replays_one_outcome_after_terminal_commit(self): + run_id, waiting = self.fixture.start_waiting("objective-crash") + with mock.patch.object(Store, "project_final_outcome", create=True, side_effect=RuntimeError("projection crash")): + with self.assertRaisesRegex(RuntimeError, "projection crash"): + self.accept(run_id, waiting) + self.assertEqual(self.service.status(run_id)["state"], "succeeded") + final = self.outcome(run_id) + self.service.result(run_id) + self.assertEqual(self.outcome(run_id), final) + + def test_repaired_failed_attempt_never_gets_independent_credit(self): + self.fixture.configure_reviewer_fallback() + run_id, waiting = self.fixture.start_waiting("objective-repaired") + self.accept(run_id, waiting) + contributions = self.outcome(run_id)["contributions"] + self.assertEqual([c["result"] for c in contributions], ["failed", "repair"]) + self.assertTrue(all(not c["independent_success"] for c in contributions)) + + def test_report_repairs_a_projection_crash_without_manual_outcome_import(self): + run_id, waiting = self.fixture.start_waiting("objective-report-crash") + with mock.patch.object(Store, "project_final_outcome", side_effect=RuntimeError("projection crash")): + with self.assertRaisesRegex(RuntimeError, "projection crash"): + self.accept(run_id, waiting) + report = self.service.learning_report(self.fixture.repo) + self.assertEqual(report["sample_size"], 1) + self.assertEqual(report["missingness"]["terminal_runs_without_final_outcome"], 0) + final = self.outcome(run_id) + with self.assertRaisesRegex(ConflictError, "objective projections"): + self.service.outcome_add(run_id, {**final, "outcome_id": "manual-replacement-final"}) + + def test_corrupt_pending_projection_does_not_block_other_runs(self): + run_id, waiting = self.fixture.start_waiting("objective-corrupt-pending") + with mock.patch.object(Store, "project_final_outcome", side_effect=RuntimeError("projection crash")): + with self.assertRaises(RuntimeError): + self.accept(run_id, waiting) + store = self.service._store() + try: + artifact = next(a for a in store.artifacts_for_run(run_id) if a["name"] == "receipt.json") + Path(artifact["path"]).write_bytes(b"corrupt fixture receipt") + finally: + store.close() + with self.assertRaisesRegex(ConflictError, "integrity"): + self.service.status(run_id) + good_id, good_waiting = self.fixture.start_waiting("objective-unrelated-good") + self.assertEqual(self.accept(good_id, good_waiting)["state"], "succeeded") + self.assertEqual(self.service.status(good_id)["state"], "succeeded") + self.assertEqual(self.outcome(good_id)["verdict"], "succeeded") + + def test_proposal_repairs_pending_outcome_before_its_consistent_read(self): + run_id, waiting = self.fixture.start_waiting("objective-proposal-crash") + with mock.patch.object(Store, "project_final_outcome", side_effect=RuntimeError("projection crash")): + with self.assertRaises(RuntimeError): + self.accept(run_id, waiting) + proposed = self.service.learning_propose(self.fixture.repo) + self.assertEqual(proposed["proposal"]["sample_sizes"]["final_outcomes"], 1) + self.assertEqual(self.outcome(run_id)["verdict"], "succeeded") + + def test_generic_worker_timeout_preserves_native_exit_and_projects_failure(self): + task = copy.deepcopy(self.fixture.task) + task["budget"]["wall_seconds"] = 2 + started = self.service.start(task, "objective-generic-timeout", _internal_fake_delay=4) + self.fixture.wait_state(started["run_id"], {"failed"}) + final = self.outcome(started["run_id"]) + self.assertEqual(final["verdict"], "failed") + self.assertEqual([c["result"] for c in final["contributions"]], ["failed"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_public_trials.py b/test/core/test_public_trials.py new file mode 100644 index 0000000..0f5c5b4 --- /dev/null +++ b/test/core/test_public_trials.py @@ -0,0 +1,207 @@ +"""Public opt-in trials: declaration, shared budgets and lifecycle chain.""" +import copy +import json +from pathlib import Path +import tempfile +import threading +import time +import unittest +from unittest.mock import patch + +from experiment_runtime_fixture import ExperimentRuntimeFixture +import test_lifecycle as lifecycle_fixtures +from devsquad.contracts import ContractError + + +class PublicTrialTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-public-trial-") + self.addCleanup(self.temporary.cleanup) + self.fixture = ExperimentRuntimeFixture(Path(self.temporary.name)) + self.addCleanup(self.fixture.close) + + def start(self, case="eval-1", arm="candidate", *, fixture=None): + fixture = fixture or self.fixture + started = fixture.service.trial_start( + fixture.spec, case, arm, fixture.task(case, arm), f"public-trial-{case}-{arm}", + _internal_review_fixture={"verdict": "clean", "summary": "Explicit offline trial review.", "findings": []}, + ) + fixture.runs[(case, arm)] = started["run_id"] + return started + + def test_opt_in_declaration_is_immutable_and_exact_replay_does_not_launch_again(self): + with patch.object(self.fixture.service, "_spawn_daemon", return_value=0) as spawn: + first = self.start() + replay = self.start() + self.assertEqual(replay["run_id"], first["run_id"]) + self.assertFalse(replay["created"]) + spawn.assert_called_once() + original = copy.deepcopy(self.fixture.spec) + self.fixture.spec["hypothesis"] = "Changed after one arm was frozen." + changed = self.start(arm="control") + self.assertEqual(changed["state"], "failed") + store = self.fixture.store() + try: + self.assertEqual(json.loads(store.connection.execute("SELECT spec_json FROM experiment_specs").fetchone()[0]), original) + self.assertEqual(store.attempts_for_run(changed["run_id"]), []) + self.assertEqual(len(store.outcomes_for_run(changed["run_id"])), 1) + finally: + store.close() + + def test_shared_experiment_budget_is_not_replenished_for_another_arm(self): + self.fixture.spec["budget"]["max_worker_invocations"] = 1 + self.fixture.run_arm("eval-1", "control") + second = self.start() + self.assertEqual(self.fixture.wait(second["run_id"])["state"], "failed") + store = self.fixture.store() + try: + self.assertEqual(store.attempts_for_run(second["run_id"]), []) + self.assertEqual(store.outcomes_for_run(second["run_id"])[0]["outcome"]["contributions"], []) + self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM attempts").fetchone()[0], 1) + finally: + store.close() + evaluation = self.fixture.service.policy_evaluate(self.fixture.spec)["evaluation"] + self.assertEqual(evaluation["metrics"]["evaluation"]["available_pairs"], 0) + + def test_concurrent_arms_cannot_overbook_the_same_trial_budget(self): + self.fixture.spec["budget"]["max_worker_invocations"] = 1 + with patch.object(self.fixture.service, "_spawn_daemon", return_value=0): + runs = [self.start(arm=arm)["run_id"] for arm in ("control", "candidate")] + barrier, errors = threading.Barrier(2), [] + def resume(run_id): + try: + barrier.wait(timeout=5) + self.fixture.service.resume(run_id) + except Exception as exc: + errors.append(exc) + workers = [threading.Thread(target=resume, args=(run_id,)) for run_id in runs] + for worker in workers: + worker.start() + for worker in workers: + worker.join(timeout=10) + self.assertFalse(worker.is_alive()) + self.assertEqual(errors, []) + states = [self.fixture.wait(run_id)["state"] for run_id in runs] + self.assertCountEqual(states, ["awaiting_host", "failed"]) + store = self.fixture.store() + try: + self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM attempts").fetchone()[0], 1) + finally: + store.close() + + def test_fallback_cannot_spend_a_second_slot_after_the_trial_budget(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "fallback", with_fallback=True) + self.addCleanup(fixture.close) + fixture.spec["budget"]["max_worker_invocations"] = 1 + started = self.start(fixture=fixture) + self.assertEqual(fixture.wait(started["run_id"])["state"], "failed") + store = fixture.store() + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual(len(attempts), 1) + self.assertEqual(attempts[0]["profile_index"], 0) + final = store.outcomes_for_run(started["run_id"])[0]["outcome"] + self.assertEqual([c["result"] for c in final["contributions"]], ["failed"]) + finally: + store.close() + + def test_experiment_deadline_blocks_later_launch_without_invented_exposure(self): + self.fixture.spec["budget"]["wall_seconds"] = 1 + with patch.object(self.fixture.service, "_spawn_daemon", return_value=0): + started = self.start() + time.sleep(1.05) + self.fixture.service.resume(started["run_id"]) + self.assertEqual(self.fixture.wait(started["run_id"])["state"], "failed") + store = self.fixture.store() + try: + self.assertEqual(store.attempts_for_run(started["run_id"]), []) + self.assertEqual(store.outcomes_for_run(started["run_id"])[0]["outcome"]["contributions"], []) + finally: + store.close() + + def test_experiment_deadline_also_stops_an_already_running_worker(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "active-deadline", workflow="issue-delivery") + self.addCleanup(fixture.close) + fixture.spec["budget"]["wall_seconds"] = 5 + before = time.monotonic() + started = fixture.service.trial_start( + fixture.spec, "eval-1", "candidate", fixture.task("eval-1", "candidate"), "public-active-deadline", + _internal_implementation_fixture={"writes": [{"path": "README", "content": "fixed eval-1\n"}], "delay_seconds": 10}, + _internal_review_fixture={"verdict": "clean", "summary": "Active deadline fixture.", "findings": []}, + ) + fixture.runs[("eval-1", "candidate")] = started["run_id"] + self.assertEqual(fixture.wait(started["run_id"])["state"], "failed") + self.assertLess(time.monotonic() - before, 8, "worker must not run for its separate 120-second task budget") + store = fixture.store() + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual(len(attempts), 1) + self.assertIsNotNone(attempts[0]["pid"], "this must exercise active work, not a prelaunch failure") + final = store.outcomes_for_run(started["run_id"])[0]["outcome"] + self.assertEqual([c["result"] for c in final["contributions"]], ["failed"]) + finally: + store.close() + + def test_public_controller_rejects_delivery_reviewer_and_unbounded_requests(self): + task = self.fixture.task("eval-1", "control") + task.update(workflow="issue-delivery") + task["scope"]["write_paths"] = ["README"] + with self.assertRaisesRegex(ContractError, "frozen review"): + self.fixture.service.trial_start(self.fixture.spec, "eval-1", "control", task, "invalid-delivery-reviewer") + experiment = copy.deepcopy(self.fixture.spec) + experiment["budget"]["wall_seconds"] = 3601 + with self.assertRaisesRegex(ContractError, "bounded controller"): + self.fixture.service.trial_start(experiment, "eval-1", "control", self.fixture.task("eval-1", "control"), "unbounded-trial") + + def test_public_outcomes_evaluate_qualify_promote_new_run_and_roll_back(self): + service = self.fixture.service + service.profile_binding_bootstrap({"template": lifecycle_fixtures.lifecycle_template(update_mode="reviewed"), + "profile": self.fixture.profiles["control"], "version": 7}) + self.fixture.run_all() + self.assertEqual(service.learning_report(self.fixture.repo)["sample_size"], 4) + evaluation = service.policy_evaluate(self.fixture.spec) + self.assertTrue(evaluation["eligibility"]["eligible"]) + helper = lifecycle_fixtures.ProfileLifecycleTest() + helper.candidate = self.fixture.profiles["candidate"] + qualification = helper.qualification(evaluation) + self.assertEqual(service.profile_qualification_add(qualification)["gate_failures"], []) + promoted = service.profile_binding_change(helper.promotion("public-chain-promote")) + self.assertEqual(promoted["receipt"]["to"]["binding_version"], 8) + # Normal automatic routing of a NEW run observes the promoted alias; + # completed experimental runs retain their exact prelaunch bindings. + task = self.fixture.task("hold-1", "candidate") + task["routing"].pop("overrides") + with patch.object(service, "_spawn_daemon", return_value=0): + automatic = service.start(task, "public-promoted-new-run", + _internal_review_fixture={"verdict": "clean", "summary": "New-run binding verification.", "findings": []}) + try: + store = self.fixture.store() + try: + snapshot = json.loads(store.run(automatic["run_id"])["mutable_snapshot"]) + self.assertEqual(snapshot["routing"]["roles"]["reviewer"]["selected"]["profile_id"], "profile-b") + self.assertEqual(snapshot["routing"]["roles"]["reviewer"]["selected"]["binding"]["version"], 8) + old = json.loads(store.run(self.fixture.runs[("eval-1", "control")])["mutable_snapshot"]) + self.assertEqual(old["routing"]["roles"]["reviewer"]["selected"]["profile_id"], "profile-a") + finally: + store.close() + finally: + service.cancel(automatic["run_id"]) + regression = ExperimentRuntimeFixture(self.fixture.root / "regression", service=service, repo=self.fixture.repo, + experiment_id="public-post-promotion-regression", candidate_succeeds=False) + self.addCleanup(regression.close) + regression.run_all() + evaluated = service.policy_evaluate(regression.spec) + self.assertEqual(evaluated["evaluation"]["verdict"], "no_change") + rollback = {"schema_version": 1, "decision_id": "public-chain-rollback", "action": "rollback", "alias": "review.deep", + "expected_binding_version": 8, "qualification_id": None, + "rollback_target": {"profile_id": "profile-a", "binding_version": 7}, + "experiment_id": regression.spec["experiment_id"], "evaluation_sha256": evaluated["evaluation_sha256"], + "actor": "human", "reason": "Predeclared evaluation and held-out regression favor the prior incumbent.", + "evidence_refs": ["evaluation.json"]} + reverted = service.profile_binding_change(rollback) + self.assertEqual(reverted["receipt"]["to"]["binding_version"], 9) + self.assertEqual(service.profile_binding_status("review.deep")["binding"]["profile_id"], "profile-a") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py new file mode 100644 index 0000000..d532ab8 --- /dev/null +++ b/test/core/test_review_runtime.py @@ -0,0 +1,1286 @@ +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import time +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.capacity import derive_pool_capacity +from devsquad.contracts import ContractError +from devsquad.reports import TERMINAL_REPORT_NAMES, build_handoff_reports +from devsquad.review_worker import _run_check +from devsquad.service import Service +from devsquad.store import ConflictError, Store, request_hash +from devsquad_test_fixtures import branch_review_routing_documents + + +class DurableBranchReviewTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-review-runtime-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], + check=True, + ) + (self.repo / "src").mkdir() + (self.repo / "tests").mkdir() + (self.repo / "devsquad").mkdir() + (self.repo / "src/app.py").write_text("VALUE = 'base'\n") + (self.repo / "tests/test_app.py").write_text("# fixture\n") + profiles, policy = branch_review_routing_documents() + (self.repo / "devsquad/profiles.json").write_text(profiles) + (self.repo / "devsquad/policy.json").write_text(policy) + subprocess.run(["git", "-C", str(self.repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.base = self.git_text("rev-parse", "HEAD").strip() + (self.repo / "src/app.py").write_text("VALUE = 'candidate'\n") + subprocess.run(["git", "-C", str(self.repo), "add", "src/app.py"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "candidate"], + check=True, + ) + self.target = self.git_text("rev-parse", "HEAD").strip() + self.task = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + self.task["project"] = { + "repo_path": str(self.repo), + "base_ref": self.base, + "target_ref": self.target, + } + self.task["checks"] = [{ + "id": "fixture-tests", + "argv": [sys.executable, "-c", "print('reported failure'); raise SystemExit(7)"], + "cwd": ".", + "timeout_seconds": 10, + "required_to_pass": False, + }] + self.service = Service(self.runtime) + self.fixture = { + "verdict": "findings", + "summary": "The candidate changes the configured value.", + "findings": [{ + "id": "F-1", + "severity": "medium", + "title": "Changed behavior needs confirmation", + "description": "The new value differs from the baseline.", + "path": "src/app.py", + "start_line": 1, + "end_line": 1, + "evidence": "The target contains VALUE = 'candidate'.", + }], + } + + def git_bytes(self, *arguments): + return subprocess.run( + ["git", "-C", str(self.repo), *arguments], + check=True, + capture_output=True, + ).stdout + + def git_text(self, *arguments): + return self.git_bytes(*arguments).decode() + + def wait_state(self, run_id, states, timeout=10): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status = self.service.status(run_id) + if status["state"] in states: + return status + time.sleep(0.05) + log = self.runtime / "private-logs" / f"{run_id}.supervisor.log" + detail = log.read_text() if log.exists() else "no supervisor log" + self.fail(f"run did not reach {states}: {self.service.status(run_id)}\n{detail}") + + @staticmethod + def decision(packet, submission_id, disposition, reason): + body = { + "schema_version": 1, + "submission_id": submission_id, + "disposition": disposition, + "reason": reason, + "evidence_refs": [ + { + "artifact_id": reference["artifact_id"], + "sha256": reference["sha256"], + } + for reference in packet["artifacts"] + ], + } + return {**body, "submission_hash": request_hash(body)} + + def start_waiting(self, key): + started = self.service.start( + self.task, + key, + _internal_review_fixture=self.fixture, + ) + self.assertTrue(started["created"]) + waiting = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + return started["run_id"], waiting + + def test_each_trusted_check_gets_a_fresh_isolated_home(self): + inherited_home = self.root / "inherited-home" + inherited_home.mkdir() + (inherited_home / "credential-marker").write_text("private\n") + script = ( + "from pathlib import Path; import os,sys; " + "home=Path(os.environ['HOME']); inherited=Path(sys.argv[1]); " + "assert home.is_dir(); assert home != inherited; " + "assert not (home/'credential-marker').exists(); " + "assert not (home/'prior-check-marker').exists(); " + "(home/sys.argv[2]).write_text('created\\n'); print(home)" + ) + results = [] + with patch.dict(os.environ, {"HOME": str(inherited_home)}): + for check_id, marker in ( + ("first-home-check", "prior-check-marker"), + ("second-home-check", "second-check-marker"), + ): + results.append(_run_check( + { + "id": check_id, + "argv": [sys.executable, "-c", script, str(inherited_home), marker], + "cwd": ".", + "timeout_seconds": 10, + "required_to_pass": True, + }, + self.repo.resolve(), + "candidate-sha256", + self.target, + )) + + self.assertEqual([result["status"] for result in results], ["passed", "passed"]) + homes = [result["stdout"]["preview"].strip() for result in results] + self.assertNotEqual(homes[0], homes[1]) + self.assertTrue(all(home != str(inherited_home) for home in homes)) + self.assertTrue(all(not Path(home).exists() for home in homes)) + self.assertEqual((inherited_home / "credential-marker").read_text(), "private\n") + + def test_trusted_python_check_does_not_write_bytecode_into_candidate(self): + (self.repo / "checked_module.py").write_text("VALUE = 7\n") + with patch.dict(os.environ, {"PYTHONDONTWRITEBYTECODE": "0"}): + result = _run_check({ + "id": "import-candidate", "argv": [sys.executable, "-c", "import checked_module; assert checked_module.VALUE == 7"], + "cwd": ".", "timeout_seconds": 10, "required_to_pass": True, + }, self.repo.resolve(), "candidate-sha256", self.target) + self.assertEqual(result["status"], "passed") + self.assertFalse((self.repo / "__pycache__").exists()) + + def configure_fixture_headless(self): + profiles = json.loads((self.repo / "devsquad/profiles.json").read_text()) + lead = dict(profiles["profiles"][0]) + lead.update({ + "id": "fixture-lead", + "model_family": "fixture-family-lead", + "model_id": "fixture-lead-model", + }) + profiles["profiles"].append(lead) + profiles["bindings"]["lead.primary"] = { + "profile_id": "fixture-lead", "version": 1, + } + policy = json.loads((self.repo / "devsquad/policy.json").read_text()) + policy["roles"]["lead"] = [{"kind": "alias", "id": "lead.primary"}] + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "headless routing"], + check=True, + ) + self.task["project"]["target_ref"] = self.git_text("rev-parse", "HEAD").strip() + self.task["lead"] = {"mode": "headless"} + self.task["budget"]["max_worker_invocations"] = 2 + + def configure_reviewer_fallback(self): + profiles = json.loads((self.repo / "devsquad/profiles.json").read_text()) + failing = profiles["profiles"][0] + failing["id"] = "reviewer-fixture-fail" + failing["model_id"] = "fixture-model-fail" + fallback = dict(failing) + fallback.update({ + "id": "reviewer-fallback", + "model_family": "fixture-family-b", + "model_id": "fixture-model-fallback", + }) + profiles["profiles"].append(fallback) + profiles["bindings"]["review.deep"] = { + "profile_id": failing["id"], "version": 2, + } + policy = json.loads((self.repo / "devsquad/policy.json").read_text()) + policy["roles"]["reviewer"] = [ + {"kind": "alias", "id": "review.deep"}, + {"kind": "profile", "id": fallback["id"]}, + ] + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "fallback routing"], + check=True, + ) + self.task["project"]["target_ref"] = self.git_text("rev-parse", "HEAD").strip() + self.task["budget"]["max_fallbacks_per_step"] = 1 + + def configure_headless_lead_fallback(self): + self.configure_fixture_headless() + profiles = json.loads((self.repo / "devsquad/profiles.json").read_text()) + failing = next(profile for profile in profiles["profiles"] + if profile["id"] == "fixture-lead") + failing.update({ + "id": "lead-fixture-fail", + "model_id": "fixture-lead-model-fail", + }) + fallback = dict(failing) + fallback.update({ + "id": "lead-fallback", + "model_family": "fixture-family-lead-fallback", + "model_id": "fixture-lead-model-fallback", + }) + profiles["profiles"].append(fallback) + profiles["bindings"]["lead.primary"] = { + "profile_id": failing["id"], "version": 2, + } + policy = json.loads((self.repo / "devsquad/policy.json").read_text()) + policy["roles"]["lead"] = [ + {"kind": "alias", "id": "lead.primary"}, + {"kind": "profile", "id": fallback["id"]}, + ] + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "lead fallback routing"], + check=True, + ) + self.task["project"]["target_ref"] = self.git_text("rev-parse", "HEAD").strip() + self.task["budget"]["max_worker_invocations"] = 3 + self.task["budget"]["max_fallbacks_per_step"] = 1 + + def test_detached_review_imports_bound_evidence_and_publishes_host_handoff(self): + (self.repo / "notes.txt").write_text("unrelated local work\n") + before_head = self.git_bytes("rev-parse", "HEAD") + before_status = self.git_bytes("status", "--porcelain=v1", "-z") + before_index = hashlib.sha256((self.repo / ".git/index").read_bytes()).hexdigest() + + started = self.service.start( + self.task, + "durable-review", + _internal_review_fixture=self.fixture, + ) + self.assertTrue(started["created"]) + waiting = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + self.assertEqual(waiting["next_action"], "claim_handoff") + self.assertEqual( + (self.git_bytes("rev-parse", "HEAD"), + self.git_bytes("status", "--porcelain=v1", "-z"), + hashlib.sha256((self.repo / ".git/index").read_bytes()).hexdigest()), + (before_head, before_status, before_index), + ) + + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + run = store.run(started["run_id"]) + snapshot = json.loads(run["mutable_snapshot"]) + self.assertEqual(run["worktree_path"], snapshot["workspace"]["path"]) + self.assertNotEqual( + snapshot["workspace"]["path"], snapshot["check_workspace"]["path"], + ) + attempts = store.connection.execute( + "SELECT status FROM attempts WHERE run_id=?", (started["run_id"],), + ).fetchall() + self.assertEqual([row["status"] for row in attempts], ["finished"]) + artifacts = store.result_snapshot(started["run_id"])[1] + names = {item["name"] for item in artifacts} + attempt_id = store.attempt(started["run_id"])["id"] + self.assertTrue({ + f"review-{attempt_id}.json", + f"checks-{attempt_id}.json", + f"evaluation-{attempt_id}.json", + f"review-attempt-{attempt_id}.json", + "handoff.json", + "handoff.md", + } <= names) + self.assertNotIn("result-receipt.json", names) + handoff = store.handoff_snapshot(started["run_id"]) + finally: + store.close() + + packet = handoff.packet + self.assertEqual(packet["attempt_id"], attempt_id) + self.assertEqual(packet["candidate_sha256"], snapshot["workspace"]["candidate_sha256"]) + self.assertEqual(packet["review"]["findings"][0]["id"], "F-1") + self.assertEqual(packet["checks"][0]["status"], "failed") + self.assertFalse(packet["checks"][0]["required_to_pass"]) + self.assertTrue(packet["evaluation"]["accept_allowed"]) + self.assertEqual(packet["evaluation"]["report_only_failures"], ["fixture-tests"]) + self.assertEqual( + len(packet["artifacts"]), + 4, + ) + self.assertTrue(all(reference["artifact_id"] for reference in packet["artifacts"])) + + claimed = self.service.handoff_claim( + started["run_id"], waiting["version"], "host-terminal", + ) + self.assertEqual(claimed["handoff"]["packet"], packet) + self.assertFalse(self.service.result(started["run_id"])["ready"]) + + decision = self.decision(packet, "accept-review", "accept", "Evidence accepted.") + completed = self.service.handoff_complete( + started["run_id"], claimed["claim"], decision, + ) + self.assertEqual((completed["state"], completed["phase"]), ("succeeded", None)) + self.assertEqual(completed["continuation"]["action"], "terminal") + self.assertFalse(completed["launched"]) + result = self.service.result(started["run_id"]) + self.assertTrue(result["ready"]) + names = {artifact["name"] for artifact in result["artifacts"]} + self.assertTrue({ + "receipt.json", "receipt.md", "events.jsonl", + "artifact-manifest.json", "result-receipt.json", + } <= names) + receipt_artifact = next( + artifact for artifact in result["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["state"], "succeeded") + self.assertEqual(receipt["candidate"]["sha256"], packet["candidate_sha256"]) + self.assertEqual(receipt["evaluation"]["report_only_failures"], ["fixture-tests"]) + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertIsNone(receipt["accounting"]["native_model_requests"]) + self.assertFalse(receipt["accounting"]["host_usage_measured"]) + self.assertEqual(receipt["lead"]["usage"]["source"], "unavailable") + self.assertFalse(receipt["events_export"]["includes_terminal_event"]) + self.assertIn("offline fixture", receipt["limitations"][0]) + replay = self.service.handoff_complete( + started["run_id"], claimed["claim"], decision, + ) + self.assertTrue(replay["replayed"]) + self.assertEqual(replay["state"], "succeeded") + self.assertEqual( + (self.git_bytes("rev-parse", "HEAD"), + self.git_bytes("status", "--porcelain=v1", "-z"), + hashlib.sha256((self.repo / ".git/index").read_bytes()).hexdigest()), + (before_head, before_status, before_index), + ) + + def test_required_failure_blocks_accept_before_record_then_allows_reject(self): + self.task["checks"][0]["required_to_pass"] = True + run_id, waiting = self.start_waiting("required-failure") + claimed = self.service.handoff_claim(run_id, waiting["version"], "host-required") + packet = claimed["handoff"]["packet"] + accept = self.decision(packet, "blocked-accept", "accept", "Accept anyway.") + with self.assertRaisesRegex(ContractError, "blocked by required evidence"): + self.service.handoff_complete(run_id, claimed["claim"], accept) + status = self.service.status(run_id) + self.assertEqual((status["state"], status["phase"]), ("awaiting_host", None)) + + reject = self.decision(packet, "required-reject", "reject", "Required check failed.") + completed = self.service.handoff_complete(run_id, claimed["claim"], reject) + self.assertEqual(completed["state"], "failed") + result = self.service.result(run_id) + receipt_artifact = next( + artifact for artifact in result["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "REVIEW_REJECTED") + self.assertEqual(receipt["evaluation"]["required_failures"], ["fixture-tests"]) + + def test_decision_must_bind_every_presented_evidence_artifact(self): + run_id, waiting = self.start_waiting("missing-evidence") + claimed = self.service.handoff_claim(run_id, waiting["version"], "host-evidence") + packet = claimed["handoff"]["packet"] + decision = self.decision(packet, "missing-ref", "accept", "Incomplete evidence.") + decision["evidence_refs"].pop() + body = {key: value for key, value in decision.items() if key != "submission_hash"} + decision["submission_hash"] = request_hash(body) + with self.assertRaisesRegex(ContractError, "bind every presented"): + self.service.handoff_complete(run_id, claimed["claim"], decision) + self.assertEqual(self.service.status(run_id)["handoff"]["status"], "open") + + def test_revision_rechecks_clean_workspace_and_preserves_both_attempts(self): + self.task["budget"]["max_revisions"] = 1 + self.task["budget"]["max_worker_invocations"] = 3 + self.task["checks"] = [{ + "id": "dirtying-check", + "argv": [ + sys.executable, + "-c", + "from pathlib import Path; p=Path('generated.tmp'); " + "assert not p.exists(); p.write_text('generated')", + ], + "cwd": ".", + "timeout_seconds": 10, + "required_to_pass": True, + "output_paths": ["generated.tmp"], + }] + run_id, first_wait = self.start_waiting("one-revision") + first_claim = self.service.handoff_claim( + run_id, first_wait["version"], "host-revision-one", + ) + first_packet = first_claim["handoff"]["packet"] + revise = self.decision( + first_packet, "request-revision", "revise", "Repeat the frozen review.", + ) + requeued = self.service.handoff_complete(run_id, first_claim["claim"], revise) + self.assertEqual(requeued["continuation"]["action"], "requeued") + self.assertTrue(requeued["launched"]) + + second_wait = self.wait_state(run_id, {"awaiting_host", "failed"}) + self.assertEqual(second_wait["state"], "awaiting_host") + self.assertEqual(second_wait["handoff"]["sequence"], 2) + second_claim = self.service.handoff_claim( + run_id, second_wait["version"], "host-revision-two", + ) + second_packet = second_claim["handoff"]["packet"] + self.assertNotEqual(first_packet["attempt_id"], second_packet["attempt_id"]) + self.assertTrue(all(result["status"] == "passed" for result in second_packet["checks"])) + self.assertTrue( + {reference["name"] for reference in first_packet["artifacts"]}.isdisjoint( + {reference["name"] for reference in second_packet["artifacts"]} + ) + ) + accept = self.decision( + second_packet, "accept-revision", "accept", "Second review accepted.", + ) + completed = self.service.handoff_complete(run_id, second_claim["claim"], accept) + self.assertEqual(completed["state"], "succeeded") + receipt_artifact = next( + artifact for artifact in self.service.result(run_id)["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(len(receipt["attempts"]), 2) + self.assertEqual( + [item["disposition"] for item in receipt["dispositions"]], + ["revise", "accept"], + ) + self.assertEqual(receipt["revisions"]["executed"], 1) + late = self.decision( + first_packet, "late-first-host", "accept", "This claim is stale.", + ) + with self.assertRaises(ConflictError): + self.service.handoff_complete(run_id, first_claim["claim"], late) + + def test_zero_revision_budget_turns_revise_into_terminal_failure(self): + self.task["budget"]["max_revisions"] = 0 + run_id, waiting = self.start_waiting("no-revisions") + claimed = self.service.handoff_claim(run_id, waiting["version"], "host-no-revision") + packet = claimed["handoff"]["packet"] + revise = self.decision(packet, "revise-exhausted", "revise", "Try again.") + completed = self.service.handoff_complete(run_id, claimed["claim"], revise) + self.assertEqual(completed["state"], "failed") + receipt_artifact = next( + artifact for artifact in self.service.result(run_id)["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "BUDGET_EXHAUSTED") + self.assertEqual(receipt["lead"]["disposition"], "revise") + + def test_wall_budget_exhaustion_between_attempts_preserves_review_history(self): + self.task["budget"]["max_revisions"] = 1 + self.task["budget"]["max_worker_invocations"] = 2 + run_id, waiting = self.start_waiting("wall-budget-between-attempts") + claimed = self.service.handoff_claim( + run_id, waiting["version"], "host-wall-budget", + ) + packet = claimed["handoff"]["packet"] + revise = self.decision( + packet, "wall-budget-revise", "revise", "Repeat once.", + ) + with patch.object(self.service, "_spawn_daemon", return_value=12345): + requeued = self.service.handoff_complete( + run_id, claimed["claim"], revise, + ) + self.assertEqual(requeued["state"], "queued") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + attempt = store.attempt(run_id) + old = "2026-09-18T00:00:00+00:00" + finished = "2026-09-18T00:10:00+00:00" + store.connection.execute( + "UPDATE attempts SET created_at=?,finished_at=? WHERE id=?", + (old, finished, attempt["id"]), + ) + finally: + store.close() + terminal = self.service.fail_budget_exhausted( + run_id, requeued["version"], + ) + self.assertEqual(terminal["state"], "failed") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(run_id)["artifacts"] + } + self.assertTrue(TERMINAL_REPORT_NAMES <= set(artifacts)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "BUDGET_EXHAUSTED") + self.assertEqual(len(receipt["attempts"]), 1) + self.assertEqual( + [item["disposition"] for item in receipt["dispositions"]], ["revise"], + ) + + def test_resume_finishes_submission_recorded_before_continuation(self): + run_id, waiting = self.start_waiting("resume-submission") + claimed = self.service.handoff_claim(run_id, waiting["version"], "host-crash") + packet = claimed["handoff"]["packet"] + decision = self.decision(packet, "saved-before-crash", "accept", "Accept evidence.") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + store.record_handoff_submission( + run_id, + self.service._decode_claim(claimed["claim"]), + decision, + ) + finally: + store.close() + self.assertEqual(self.service.status(run_id)["phase"], "handoff_submitted") + resumed = self.service.resume(run_id) + self.assertEqual((resumed["state"], resumed["disposition"]), ("succeeded", "terminal")) + self.assertFalse(resumed["launched"]) + self.assertTrue(self.service.result(run_id)["ready"]) + + def test_public_native_codex_driver_verifies_identity_usage_and_output(self): + profiles = { + "schema_version": 1, + "profiles": [{ + "id": "native-codex-reviewer", + "harness": "codex", + "model_family": "gpt-fixture", + "model_id": "gpt-fake-review", + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "codex-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["native-fixture"], + }], + "bindings": { + "review.deep": {"profile_id": "native-codex-reviewer", "version": 1}, + }, + } + policy_document = { + "schema_version": 1, + "id": "native-codex-policy", + "version": 1, + "roles": {"reviewer": [{"kind": "alias", "id": "review.deep"}]}, + "task_classes": {"fixture-review-small": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "codex-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy_document, sort_keys=True) + "\n" + ) + subprocess.run( + ["git", "-C", str(self.repo), "add", "devsquad"], check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "native review config"], + check=True, + ) + self.target = self.git_text("rev-parse", "HEAD").strip() + self.task["project"]["target_ref"] = self.target + fake_bin = self.root / "fake-bin" + fake_bin.mkdir() + (fake_bin / "codex").symlink_to( + ROOT / "test/core/fakes/codex_review_cli.py" + ) + fake_home = self.root / "fake-codex-home" + fake_home.mkdir() + (fake_home / "auth.json").write_text("{}\n") + (fake_home / "auth.json").chmod(0o600) + environment = { + "PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", + "CODEX_HOME": str(fake_home), + } + with patch.dict(os.environ, environment, clear=False): + started = self.service.start(self.task, "native-codex-review") + self.assertTrue(started["created"]) + waiting = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + claimed = self.service.handoff_claim( + started["run_id"], waiting["version"], "native-host", + ) + packet = claimed["handoff"]["packet"] + observed = packet["attempt"]["observed_identity"] + self.assertEqual( + (observed["harness"], observed["harness_version"], observed["model_id"]), + ("codex", "codex-cli 0.153.4", "gpt-fake-review"), + ) + self.assertEqual(packet["attempt"]["usage"], { + "input_tokens": 120, + "output_tokens": 40, + "total_tokens": 160, + "source": "native_reported", + }) + self.assertEqual(packet["review"]["verdict"], "clean") + decision = self.decision(packet, "accept-native", "accept", "Native review accepted.") + completed = self.service.handoff_complete( + started["run_id"], claimed["claim"], decision, + ) + self.assertEqual(completed["state"], "succeeded") + receipt_artifact = next( + artifact for artifact in self.service.result(started["run_id"])["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["accounting"]["attempt_usage"][0]["total_tokens"], 160) + self.assertEqual(receipt["limitations"], []) + + def test_native_codex_faults_never_become_valid_reviews(self): + profiles_text, policy_text = branch_review_routing_documents() + profiles = json.loads(profiles_text) + profile = profiles["profiles"][0] + profile.update({ + "harness": "codex", + "model_family": "gpt-fixture", + "required_tools": ["read"], + "permission_policy": "read_only", + }) + fake_bin = self.root / "fault-bin" + fake_bin.mkdir() + (fake_bin / "codex").symlink_to( + ROOT / "test/core/fakes/codex_review_cli.py" + ) + fake_home = self.root / "fault-codex-home" + fake_home.mkdir() + (fake_home / "auth.json").write_text("{}\n") + (fake_home / "auth.json").chmod(0o600) + environment = { + "PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", + "CODEX_HOME": str(fake_home), + } + for mode in ("empty", "malformed", "denied", "disconnect", "identity-drift"): + with self.subTest(mode=mode): + profile["model_id"] = f"gpt-fake-{mode}" + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text(policy_text) + subprocess.run( + ["git", "-C", str(self.repo), "add", "devsquad"], check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", f"fault {mode}"], + check=True, + ) + target = self.git_text("rev-parse", "HEAD").strip() + self.task["project"]["target_ref"] = target + with patch.dict(os.environ, environment, clear=False): + started = self.service.start(self.task, f"native-fault-{mode}") + self.assertEqual(started["state"], "queued") + failed = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(failed["state"], "failed") + self.assertIsNone(failed["handoff"]) + result = self.service.result(started["run_id"]) + self.assertTrue(result["ready"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + self.assertTrue(TERMINAL_REPORT_NAMES <= set(artifacts)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["state"], "failed") + self.assertEqual(receipt["lead"]["status"], "not_reached") + self.assertEqual(receipt["error"]["error"], "REVIEW_WORKER_FAILED") + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertEqual(len(receipt["attempts"]), 1) + self.assertEqual(receipt["attempts"][0]["status"], "failed") + self.assertEqual( + {item["name"] for item in receipt["attempts"][0]["output_artifacts"]}, + { + f"{receipt['attempts'][0]['id']}.stdout", + f"{receipt['attempts'][0]['id']}.stderr", + }, + ) + self.assertEqual( + Path(artifacts["receipt.json"]["path"]).read_bytes(), + Path(artifacts["result-receipt.json"]["path"]).read_bytes(), + ) + if mode == "empty": + stderr_artifact = next( + artifact for name, artifact in artifacts.items() + if name.endswith(".stderr") + ) + stderr_text = Path(stderr_artifact["path"]).read_text() + self.assertIn("completed without output", stderr_text) + self.assertIn("protocol_summary=", stderr_text) + self.assertIn('"matching_deltas":2', stderr_text) + manifest = json.loads( + Path(artifacts["artifact-manifest.json"]["path"]).read_text() + ) + for entry in manifest["artifacts"]: + saved = Path(artifacts[entry["name"]]["path"]).read_bytes() + self.assertEqual(hashlib.sha256(saved).hexdigest(), entry["sha256"]) + self.assertEqual(len(saved), entry["byte_size"]) + + def test_native_codex_headless_lead_verifies_its_own_identity_and_usage(self): + profiles = { + "schema_version": 1, + "profiles": [ + { + "id": profile_id, + "harness": "codex", + "model_family": "gpt-fixture", + "model_id": model_id, + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "codex-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["native-fixture"], + } + for profile_id, model_id in ( + ("native-reviewer", "gpt-fake-review"), + ("native-lead", "gpt-fake-lead"), + ) + ], + "bindings": {}, + } + policy = { + "schema_version": 1, + "id": "native-headless-policy", + "version": 1, + "roles": { + "reviewer": [{"kind": "profile", "id": "native-reviewer"}], + "lead": [{"kind": "profile", "id": "native-lead"}], + }, + "task_classes": {"fixture-review-small": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "codex-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "native headless config"], + check=True, + ) + self.task["project"]["target_ref"] = self.git_text("rev-parse", "HEAD").strip() + self.task["lead"] = {"mode": "headless"} + self.task["budget"]["max_worker_invocations"] = 2 + fake_bin = self.root / "native-headless-bin" + fake_bin.mkdir() + (fake_bin / "codex").symlink_to(ROOT / "test/core/fakes/codex_review_cli.py") + fake_home = self.root / "native-headless-home" + fake_home.mkdir() + (fake_home / "auth.json").write_text("{}\n") + (fake_home / "auth.json").chmod(0o600) + with patch.dict(os.environ, { + "PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", + "CODEX_HOME": str(fake_home), + }, clear=False): + started = self.service.start(self.task, "native-headless") + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "succeeded") + result = self.service.result(started["run_id"]) + receipt_artifact = next( + artifact for artifact in result["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["lead"]["mode"], "headless") + self.assertEqual( + receipt["lead"]["attempts"][0]["observed_identity"]["model_id"], + "gpt-fake-lead", + ) + self.assertEqual(receipt["lead"]["usage"]["total_tokens"], 160) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual(receipt["accounting"]["attempt_usage"][1]["total_tokens"], 160) + + def test_check_worker_failure_before_handoff_gets_full_terminal_reports(self): + self.task["checks"][0]["cwd"] = "missing-check-directory" + started = self.service.start( + self.task, + "check-worker-failure", + _internal_review_fixture=self.fixture, + ) + failed = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(failed["state"], "failed") + self.assertIsNone(failed["handoff"]) + result = self.service.result(started["run_id"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + self.assertTrue(TERMINAL_REPORT_NAMES <= set(artifacts)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "REVIEW_WORKER_FAILED") + self.assertIsNone(receipt["review"]) + self.assertEqual(receipt["checks"], []) + self.assertEqual(receipt["lead"]["status"], "not_reached") + + def test_waiting_handoff_report_builder_binds_packet_and_hash(self): + run_id, _ = self.start_waiting("handoff-report-builder") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + handoff = store.handoff_snapshot(run_id) + self.assertIsNotNone(handoff) + reports = build_handoff_reports( + run_id=run_id, + handoff_id=handoff.handoff_id, + sequence=handoff.sequence, + packet=handoff.packet, + packet_sha256=handoff.packet_sha256, + created_at=handoff.created_at, + ) + finally: + store.close() + self.assertEqual(set(reports), {"handoff.json", "handoff.md"}) + document = json.loads(reports["handoff.json"]) + self.assertEqual(document["state"], "awaiting_host") + self.assertEqual(document["packet"], handoff.packet) + self.assertEqual(document["packet_sha256"], handoff.packet_sha256) + self.assertIn(b"claim this saved handoff", reports["handoff.md"]) + + def test_cancel_waiting_review_publishes_full_terminal_reports(self): + run_id, _ = self.start_waiting("cancel-waiting-review") + cancelled = self.service.cancel(run_id) + self.assertEqual(cancelled["state"], "cancelled") + result = self.service.result(run_id) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + self.assertTrue(TERMINAL_REPORT_NAMES <= set(artifacts)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["state"], "cancelled") + self.assertEqual(receipt["phase"], "awaiting_host") + self.assertEqual(receipt["lead"]["status"], "cancelled") + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertEqual(len(receipt["attempts"]), 1) + + def test_public_preflight_observes_live_shared_pool_reservations(self): + def full_unknown_pool(_store, pool_id, *, target=None, now=None): + return derive_pool_capacity( + pool_id, [], target=target, in_flight=1, now=now, + ) + + with patch.object(Store, "capacity_snapshot", full_unknown_pool): + started = self.service.start( + self.task, + "pool-full-before-launch", + _internal_review_fixture=self.fixture, + ) + self.assertEqual(started["state"], "failed") + self.assertEqual(started["error"]["error"], "CAPABILITY_UNAVAILABLE") + self.assertIn("no currently available profile", started["error"]["message"]) + result = self.service.result(started["run_id"]) + self.assertTrue(result["ready"]) + self.assertTrue( + TERMINAL_REPORT_NAMES + <= {artifact["name"] for artifact in result["artifacts"]} + ) + + def test_failed_reviewer_uses_one_frozen_fallback_and_keeps_both_attempts(self): + self.configure_reviewer_fallback() + self.task["budget"]["max_worker_invocations"] = 2 + run_id, waiting = self.start_waiting("reviewer-fallback") + claimed = self.service.handoff_claim( + run_id, waiting["version"], "host-fallback", + ) + packet = claimed["handoff"]["packet"] + self.assertEqual( + packet["attempt"]["selected_profile"]["profile_id"], + "reviewer-fallback", + ) + accepted = self.decision( + packet, "accept-fallback", "accept", "Fallback review accepted.", + ) + completed = self.service.handoff_complete( + run_id, claimed["claim"], accepted, + ) + self.assertEqual(completed["state"], "succeeded") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(run_id)["artifacts"] + } + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual( + [attempt["selected_profile"]["profile_id"] + for attempt in receipt["attempts"]], + ["reviewer-fixture-fail", "reviewer-fallback"], + ) + self.assertEqual(receipt["attempts"][0]["status"], "failed") + self.assertEqual(receipt["attempts"][1]["role"], "reviewer") + + def test_recovered_prelaunch_reservation_reuses_the_same_profile_and_budget(self): + self.task["budget"]["max_worker_invocations"] = 1 + with patch.object(self.service, "_spawn_daemon"): + started = self.service.start( + self.task, + "recover-review-reservation", + _internal_review_fixture=self.fixture, + ) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + run = store.run(started["run_id"]) + snapshot = json.loads(run["mutable_snapshot"]) + selected = snapshot["routing"]["roles"]["reviewer"]["selected"] + abandoned = store.reserve_attempt( + started["run_id"], + run["version"], + "abandoned-review-supervisor", + run["package_digest"], + "reviewer", + account_pool_id=selected["profile"]["account_pool_id"], + profile_id=selected["profile_id"], + profile_index=0, + ) + recovered_version = store.recover_launching( + started["run_id"], abandoned.version, + ) + self.assertEqual(store.worker_invocations(started["run_id"]), 0) + finally: + store.close() + + resumed = self.service.resume(started["run_id"]) + self.assertEqual(resumed["disposition"], "continued") + self.assertTrue(resumed["launched"]) + self.assertGreater(recovered_version, abandoned.version) + waiting = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual( + [(attempt["status"], attempt["profile_index"]) + for attempt in attempts], + [("recovery_required", 0), ("finished", 0)], + ) + self.assertEqual(store.worker_invocations(started["run_id"]), 1) + finally: + store.close() + + def test_fallback_none_does_not_retry_a_failed_reviewer(self): + self.configure_reviewer_fallback() + self.task["routing"]["overrides"] = { + "reviewer": { + "profile_id": "reviewer-fixture-fail", + "fallback": "none", + }, + } + self.task["budget"]["max_worker_invocations"] = 3 + started = self.service.start( + self.task, "reviewer-no-fallback", _internal_review_fixture=self.fixture, + ) + completed = self.wait_state(started["run_id"], {"failed"}) + self.assertEqual(completed["state"], "failed") + result = self.service.result(started["run_id"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["routing"]["roles"]["reviewer"]["fallbacks"], []) + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertEqual(len(receipt["attempts"]), 1) + self.assertEqual( + receipt["attempts"][0]["selected_profile"]["profile_id"], + "reviewer-fixture-fail", + ) + self.assertNotIn( + "run.fallback_queued", + {event["type"] for event in self.service.events(started["run_id"])["events"]}, + ) + + def test_worker_invocation_budget_blocks_a_frozen_reviewer_fallback(self): + self.configure_reviewer_fallback() + self.task["budget"]["max_worker_invocations"] = 1 + started = self.service.start( + self.task, "reviewer-budget-no-fallback", + _internal_review_fixture=self.fixture, + ) + completed = self.wait_state(started["run_id"], {"failed"}) + self.assertEqual(completed["state"], "failed") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(started["run_id"])["artifacts"] + } + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual( + [item["profile_id"] for item in + receipt["routing"]["roles"]["reviewer"]["fallbacks"]], + ["reviewer-fallback"], + ) + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertEqual(len(receipt["attempts"]), 1) + + def test_failed_headless_lead_uses_its_own_frozen_fallback(self): + self.configure_headless_lead_fallback() + started = self.service.start( + self.task, + "headless-lead-fallback", + _internal_review_fixture=self.fixture, + _internal_lead_fixture={ + "disposition": "accept", + "reason": "The frozen review evidence is sufficient.", + }, + ) + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "succeeded") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(started["run_id"])["artifacts"] + } + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["accounting"]["worker_invocations"], 3) + self.assertEqual( + [attempt["selected_profile"]["profile_id"] + for attempt in receipt["lead"]["attempts"]], + ["lead-fixture-fail", "lead-fallback"], + ) + self.assertEqual(receipt["lead"]["attempts"][0]["status"], "failed") + + def test_reviewer_fallback_exhausts_headless_lead_budget_terminally(self): + self.configure_reviewer_fallback() + profiles = json.loads((self.repo / "devsquad/profiles.json").read_text()) + lead = dict(profiles["profiles"][-1]) + lead.update({ + "id": "budgeted-headless-lead", + "model_family": "fixture-lead-family", + "model_id": "fixture-lead-model", + }) + profiles["profiles"].append(lead) + policy = json.loads((self.repo / "devsquad/policy.json").read_text()) + policy["roles"]["lead"] = [ + {"kind": "profile", "id": lead["id"]}, + ] + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "budgeted lead"], + check=True, + ) + self.task["project"]["target_ref"] = self.git_text("rev-parse", "HEAD").strip() + self.task["lead"] = {"mode": "headless"} + self.task["budget"]["max_worker_invocations"] = 2 + started = self.service.start( + self.task, + "reviewer-fallback-exhausts-lead", + _internal_review_fixture=self.fixture, + _internal_lead_fixture={ + "disposition": "accept", + "reason": "This lead must not launch beyond the budget.", + }, + ) + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "failed") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(started["run_id"])["artifacts"] + } + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "BUDGET_EXHAUSTED") + self.assertEqual(receipt["phase"], "lead") + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual(len(receipt["attempts"]), 2) + self.assertEqual(receipt["lead"]["status"], "failed") + + def test_cancelled_later_revision_keeps_prior_attempt_and_disposition(self): + marker = self.root / "slow-second-review" + self.task["budget"]["max_revisions"] = 1 + self.task["budget"]["max_worker_invocations"] = 2 + self.task["checks"] = [{ + "id": "slow-second-check", + "argv": [ + sys.executable, + "-c", + "from pathlib import Path; import sys,time; p=Path(sys.argv[1]); " + "time.sleep(30) if p.exists() else p.write_text('first')", + str(marker), + ], + "cwd": ".", + "timeout_seconds": 40, + "required_to_pass": False, + }] + run_id, first_wait = self.start_waiting("cancel-second-review") + first_claim = self.service.handoff_claim( + run_id, first_wait["version"], "host-first-revision", + ) + revise = self.decision( + first_claim["handoff"]["packet"], + "revise-before-cancel", + "revise", + "Run the frozen review one more time.", + ) + requeued = self.service.handoff_complete( + run_id, first_claim["claim"], revise, + ) + self.assertTrue(requeued["launched"]) + self.wait_state(run_id, {"running"}) + cancelled = self.service.cancel(run_id) + self.assertIn(cancelled["state"], {"cancelling", "cancelled"}) + self.assertEqual(self.wait_state(run_id, {"cancelled"})["state"], "cancelled") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(run_id)["artifacts"] + } + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual(len(receipt["attempts"]), 2) + self.assertEqual( + [item["disposition"] for item in receipt["dispositions"]], + ["revise"], + ) + self.assertEqual(receipt["attempts"][0]["status"], "succeeded") + self.assertEqual(receipt["attempts"][1]["status"], "cancelled") + + def test_headless_lead_is_a_second_fenced_attempt_and_terminalizes_automatically(self): + self.configure_fixture_headless() + started = self.service.start( + self.task, + "headless-accept", + _internal_review_fixture=self.fixture, + _internal_lead_fixture={ + "disposition": "accept", + "reason": "The frozen review evidence is sufficient.", + }, + ) + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "succeeded") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + attempts = store.attempts_for_run(started["run_id"]) + finally: + store.close() + self.assertEqual([attempt["role"] for attempt in attempts], ["reviewer", "lead"]) + self.assertTrue(all(attempt["status"] == "finished" for attempt in attempts)) + result = self.service.result(started["run_id"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + self.assertIn("handoff.json", artifacts) + self.assertIn("handoff.md", artifacts) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["lead"]["mode"], "headless") + self.assertEqual(receipt["lead"]["disposition"], "accept") + self.assertEqual(len(receipt["lead"]["attempts"]), 1) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertIsNone(receipt["accounting"]["host_usage_measured"]) + + def test_invalid_headless_accept_is_a_failed_attempt_not_invented_success(self): + self.configure_fixture_headless() + self.task["checks"][0]["required_to_pass"] = True + started = self.service.start( + self.task, + "headless-invalid-accept", + _internal_review_fixture=self.fixture, + _internal_lead_fixture={ + "disposition": "accept", + "reason": "Attempt to override a required check.", + }, + ) + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "failed") + result = self.service.result(started["run_id"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["phase"], "lead") + self.assertEqual(receipt["lead"]["status"], "failed") + self.assertEqual(receipt["error"]["error"], "HEADLESS_LEAD_FAILED") + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual( + [attempt["role"] for attempt in receipt["attempts"]], + ["reviewer", "lead"], + ) + + def test_headless_revise_repeats_review_and_stops_at_both_budgets(self): + self.configure_fixture_headless() + self.task["budget"]["max_revisions"] = 1 + self.task["budget"]["max_worker_invocations"] = 4 + started = self.service.start( + self.task, + "headless-revise", + _internal_review_fixture=self.fixture, + _internal_lead_fixture={ + "disposition": "revise", + "reason": "Repeat the review against the same frozen candidate.", + }, + ) + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "failed") + result = self.service.result(started["run_id"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "BUDGET_EXHAUSTED") + self.assertEqual(receipt["accounting"]["worker_invocations"], 4) + self.assertEqual(len(receipt["attempts"]), 2) + self.assertEqual(len(receipt["lead"]["attempts"]), 2) + self.assertEqual(receipt["revisions"]["executed"], 1) + self.assertIn("handoff-2.json", artifacts) + self.assertIn("handoff-2.md", artifacts) + + def test_invalid_internal_review_fails_before_launch_with_a_receipt(self): + invalid = dict(self.fixture, verdict="clean") + started = self.service.start( + self.task, + "invalid-review", + _internal_review_fixture=invalid, + ) + self.assertEqual(started["state"], "failed") + self.assertEqual(started["error"]["error"], "PREPARATION_FAILED") + self.assertIn("verdict and findings disagree", started["error"]["message"]) + result = self.service.result(started["run_id"]) + self.assertTrue(result["ready"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + self.assertEqual(set(artifacts), set(TERMINAL_REPORT_NAMES)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["phase"], "preparing") + self.assertEqual(receipt["accounting"]["worker_invocations"], 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_review_workflow.py b/test/core/test_review_workflow.py new file mode 100644 index 0000000..0697bff --- /dev/null +++ b/test/core/test_review_workflow.py @@ -0,0 +1,360 @@ +import copy +import hashlib +import json +from pathlib import Path +import sys +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.workflows import ( + apply_lead_disposition, + build_review_prompt, + decode_review_document, + decode_branch_review_evidence, + evaluate_branch_review, + make_branch_review_evidence, + review_output_schema, + require_check_integrity, + validate_branch_review_evidence, + validate_check_results, + validate_review_document, +) + + +class BranchReviewWorkflowTest(unittest.TestCase): + def setUp(self): + self.task = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + self.task["project"]["repo_path"] = "/tmp/fixture-repository" + self.workspace = { + "candidate_sha256": "c" * 64, + "base_oid": "a" * 40, + "target_oid": "b" * 40, + } + self.review = { + "schema_version": 1, + "candidate_sha256": "c" * 64, + "base_oid": "a" * 40, + "target_oid": "b" * 40, + "review_mode": "standard", + "verdict": "findings", + "summary": "One supported defect was found.", + "findings": [{ + "id": "F-1", + "severity": "high", + "title": "Incorrect boundary", + "description": "The candidate accepts one value beyond the limit.", + "path": "src/example.py", + "start_line": 12, + "end_line": 13, + "evidence": "The comparison uses <= where the contract requires <.", + }], + } + self.check = self.check_result("failed", returncode=1) + self.snapshot = { + "task": self.task, + "workspace": self.workspace, + "routing": { + "roles": { + "reviewer": { + "selected": { + "profile_id": "fixture-reviewer", + "profile_sha256": "1" * 64, + "profile": {"harness": "fixture"}, + "reference": {"kind": "profile", "id": "fixture-reviewer"}, + "binding": None, + }, + }, + }, + }, + } + + def stream(self, content=""): + encoded = content.encode() + return { + "preview": content, + "captured_bytes": len(encoded), + "total_bytes": len(encoded), + "truncated": False, + "full_sha256": hashlib.sha256(encoded).hexdigest(), + } + + def test_native_output_schema_types_every_property(self): + schema = review_output_schema() + + def visit(value): + if "properties" in value: + self.assertEqual(set(value["required"]), set(value["properties"])) + self.assertFalse(value["additionalProperties"]) + for child in value["properties"].values(): + self.assertIn("type", child) + visit(child) + if isinstance(value.get("items"), dict): + visit(value["items"]) + + visit(schema) + self.assertEqual(schema["properties"]["schema_version"]["enum"], [1]) + + def check_result(self, status, *, returncode=None, error_code=None): + configured = self.task["checks"][0] + return { + "schema_version": 1, + "candidate_sha256": "c" * 64, + "target_oid": "b" * 40, + "id": configured["id"], + "argv": configured["argv"], + "cwd": configured["cwd"], + "required_to_pass": configured["required_to_pass"], + "status": status, + "returncode": returncode, + "error_code": error_code, + "duration_ms": 12, + "stdout": self.stream("check output\n"), + "stderr": self.stream(), + } + + def test_review_is_strict_and_bound_to_candidate_commits_and_mode(self): + normalized = validate_review_document(self.review, self.task, self.workspace) + self.assertEqual(normalized, self.review) + for field, replacement, message in ( + ("candidate_sha256", "d" * 64, "different candidate"), + ("base_oid", "e" * 40, "different base"), + ("target_oid", "f" * 40, "different target"), + ("review_mode", "adversarial", "mode does not match"), + ): + invalid = copy.deepcopy(self.review) + invalid[field] = replacement + with self.assertRaisesRegex(ContractError, message): + validate_review_document(invalid, self.task, self.workspace) + + def test_review_rejects_unknown_duplicate_nonfinite_and_oversized_output(self): + unknown = copy.deepcopy(self.review) + unknown["extra"] = True + with self.assertRaisesRegex(ContractError, "fields invalid"): + validate_review_document(unknown, self.task, self.workspace) + duplicate = json.dumps(self.review)[:-1] + ',"summary":"replacement"}' + with self.assertRaisesRegex(ContractError, "duplicate key"): + decode_review_document(duplicate, self.task, self.workspace) + nonfinite = json.dumps(self.review).replace('"schema_version": 1', '"schema_version": NaN') + with self.assertRaisesRegex(ContractError, "non-finite"): + decode_review_document(nonfinite, self.task, self.workspace) + with self.assertRaisesRegex(ContractError, "byte limit"): + decode_review_document(b" " * (512 * 1024 + 1), self.task, self.workspace) + + def test_clean_and_findings_verdicts_cannot_contradict_payload(self): + clean = copy.deepcopy(self.review) + clean["verdict"], clean["findings"] = "clean", [] + self.assertEqual( + decode_review_document(json.dumps(clean), self.task, self.workspace), clean, + ) + for verdict, findings in (("clean", self.review["findings"]), ("findings", [])): + invalid = copy.deepcopy(self.review) + invalid["verdict"], invalid["findings"] = verdict, findings + with self.assertRaisesRegex(ContractError, "verdict and findings disagree"): + validate_review_document(invalid, self.task, self.workspace) + + def test_findings_require_actionable_canonical_locations_and_unique_ids(self): + for field, value in ( + ("path", "../outside.py"), + ("start_line", 0), + ("end_line", 11), + ("severity", "urgent"), + ): + invalid = copy.deepcopy(self.review) + invalid["findings"][0][field] = value + with self.assertRaises(ContractError): + validate_review_document(invalid, self.task, self.workspace) + outside_scope = copy.deepcopy(self.review) + outside_scope["findings"][0]["path"] = "docs/unreviewed.md" + with self.assertRaisesRegex(ContractError, "outside the declared read scope"): + validate_review_document(outside_scope, self.task, self.workspace) + duplicate = copy.deepcopy(self.review) + duplicate["findings"].append(copy.deepcopy(duplicate["findings"][0])) + with self.assertRaisesRegex(ContractError, "duplicated"): + validate_review_document(duplicate, self.task, self.workspace) + + def test_check_results_cannot_change_host_supplied_commands_or_candidate(self): + self.assertEqual( + validate_check_results([self.check], self.task, self.workspace), [self.check], + ) + for field, value in ( + ("id", "other"), + ("argv", ["sh", "-c", "echo widened"]), + ("cwd", "src"), + ("required_to_pass", True), + ): + invalid = copy.deepcopy(self.check) + invalid[field] = value + with self.assertRaisesRegex(ContractError, "changes declared field"): + validate_check_results([invalid], self.task, self.workspace) + invalid = copy.deepcopy(self.check) + invalid["candidate_sha256"] = "d" * 64 + with self.assertRaisesRegex(ContractError, "different candidate"): + validate_check_results([invalid], self.task, self.workspace) + + def test_check_status_and_bounded_stream_metadata_are_consistent(self): + passing = self.check_result("passed", returncode=0) + timed_out = self.check_result("timed_out", error_code="TIMEOUT") + launch_failed = self.check_result("launch_failed", error_code="CLI_ERROR") + for result in (passing, self.check, timed_out, launch_failed): + self.assertEqual( + validate_check_results([result], self.task, self.workspace), [result], + ) + invalid = copy.deepcopy(passing) + invalid["returncode"] = 1 + with self.assertRaisesRegex(ContractError, "passing check"): + validate_check_results([invalid], self.task, self.workspace) + invalid = copy.deepcopy(self.check) + invalid["stdout"]["truncated"] = True + with self.assertRaisesRegex(ContractError, "truncation metadata"): + validate_check_results([invalid], self.task, self.workspace) + + def test_integrity_is_non_overridable_and_strictly_bound_to_frozen_outputs(self): + check = self.check_result("invalidated", returncode=0, error_code="CLI_ERROR") + check.update(schema_version=2, output_paths=[], integrity={ + "status": "violated", "reasons": ["check:tracked_inputs_changed"], + "before_state_sha256": "1" * 64, "after_state_sha256": "2" * 64, + "changes": [], "changes_truncated": False, + }) + evaluation = evaluate_branch_review(self.task, self.workspace, self.review, [check]) + self.assertTrue(evaluation["required_checks_passed"]) + self.assertFalse(evaluation["accept_allowed"]) + with self.assertRaises(ContractError): + apply_lead_disposition(evaluation, "accept", revisions_used=0, max_revisions=1) + for field, value in (("status", "passed"), ("output_paths", ["src"])): + tampered = copy.deepcopy(check) + tampered[field] = value + with self.assertRaises(ContractError): + validate_check_results([tampered], self.task, self.workspace) + check["integrity"]["reasons"] = [] + with self.assertRaisesRegex(ContractError, "reasons"): + validate_check_results([check], self.task, self.workspace) + + def test_legacy_checks_remain_readable_but_cannot_be_imported_or_accepted_anew(self): + evidence = make_branch_review_evidence(self.snapshot, self.review, [self.check]) + self.assertEqual(validate_branch_review_evidence(evidence, self.snapshot), evidence) + with self.assertRaisesRegex(ContractError, "integrity verification"): + decode_branch_review_evidence(json.dumps(evidence), self.snapshot) + with self.assertRaisesRegex(ContractError, "start a new run"): + require_check_integrity(evidence["checks"]) + + def test_report_only_failure_is_visible_but_does_not_block_delivery(self): + evaluation = evaluate_branch_review( + self.task, self.workspace, self.review, [self.check], + ) + self.assertTrue(evaluation["accept_allowed"]) + self.assertEqual(evaluation["required_failures"], []) + self.assertEqual(evaluation["report_only_failures"], ["fixture-tests"]) + self.assertEqual( + [item["status"] for item in evaluation["criteria"]], + ["evidence_available", "evidence_available"], + ) + accepted = apply_lead_disposition( + evaluation, "accept", revisions_used=0, max_revisions=0, + ) + self.assertEqual( + (accepted["action"], accepted["terminal_state"]), + ("complete", "succeeded"), + ) + + def test_required_failure_cannot_be_overridden_by_lead_prose(self): + required = copy.deepcopy(self.check) + required["required_to_pass"] = True + task = copy.deepcopy(self.task) + task["checks"][0]["required_to_pass"] = True + evaluation = evaluate_branch_review(task, self.workspace, self.review, [required]) + self.assertFalse(evaluation["accept_allowed"]) + self.assertEqual(evaluation["required_failures"], ["fixture-tests"]) + with self.assertRaisesRegex(ContractError, "blocked by required evidence"): + apply_lead_disposition( + evaluation, "accept", revisions_used=0, max_revisions=2, + ) + + def test_revise_is_bounded_and_reject_is_terminal(self): + evaluation = evaluate_branch_review( + self.task, self.workspace, self.review, [self.check], + ) + revised = apply_lead_disposition( + evaluation, "revise", revisions_used=0, max_revisions=1, + ) + self.assertEqual( + (revised["action"], revised["terminal_state"], revised["next_revision"]), + ("repeat_review", None, 1), + ) + exhausted = apply_lead_disposition( + evaluation, "revise", revisions_used=1, max_revisions=1, + ) + self.assertEqual( + (exhausted["action"], exhausted["terminal_state"]), + ("budget_exhausted", "failed"), + ) + rejected = apply_lead_disposition( + evaluation, "reject", revisions_used=0, max_revisions=1, + ) + self.assertEqual(rejected["terminal_state"], "failed") + + def test_prompt_is_deterministic_read_only_and_distinguishes_adversarial_mode(self): + first = build_review_prompt(self.task, self.workspace) + second = build_review_prompt(copy.deepcopy(self.task), copy.deepcopy(self.workspace)) + self.assertEqual(first, second) + self.assertIn("read-only reviewer", first) + self.assertIn('"candidate_sha256":"' + "c" * 64 + '"', first) + adversarial = copy.deepcopy(self.task) + adversarial["review"] = {"mode": "adversarial", "focus": "trust boundaries"} + prompt = build_review_prompt(adversarial, self.workspace) + self.assertIn('"review_mode":"adversarial"', prompt) + self.assertIn('"focus":"trust boundaries"', prompt) + self.assertNotEqual(prompt, first) + + def test_combined_evidence_recomputes_gates_profile_and_accounting(self): + evidence = make_branch_review_evidence( + self.snapshot, self.review, [self.check], + ) + self.assertEqual( + validate_branch_review_evidence(evidence, self.snapshot), evidence, + ) + for mutate, message in ( + (lambda value: value["evaluation"].__setitem__("accept_allowed", False), + "derived gates"), + (lambda value: value["attempt"].__setitem__( + "selected_profile", {"profile_id": "substituted"}), + "frozen fallback set"), + (lambda value: value["attempt"].__setitem__("worker_invocations", 2), + "invocation accounting"), + (lambda value: value["attempt"]["usage"].__setitem__("total_tokens", 0), + "cannot invent token counts"), + ): + invalid = copy.deepcopy(evidence) + mutate(invalid) + with self.assertRaisesRegex(ContractError, message): + validate_branch_review_evidence(invalid, self.snapshot) + + def test_issue_delivery_review_uses_the_same_candidate_bound_contract(self): + task = copy.deepcopy(self.task) + task["workflow"] = "issue-delivery" + task["scope"]["write_paths"] = ["src/example.py"] + snapshot = copy.deepcopy(self.snapshot) + snapshot["task"] = task + + prompt = build_review_prompt(task, self.workspace) + evidence = make_branch_review_evidence( + snapshot, self.review, [self.check], + ) + + self.assertIn("read-only reviewer", prompt) + self.assertEqual(evidence["workflow"], "issue-delivery") + self.assertEqual( + validate_branch_review_evidence(evidence, snapshot), evidence, + ) + stale = copy.deepcopy(evidence) + stale["workflow"] = "branch-review" + with self.assertRaisesRegex(ContractError, "workflow is invalid"): + validate_branch_review_evidence(stale, snapshot) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_router.py b/test/core/test_router.py new file mode 100644 index 0000000..05435ab --- /dev/null +++ b/test/core/test_router.py @@ -0,0 +1,407 @@ +import copy +from datetime import datetime, timezone +import hashlib +import json +from pathlib import Path +import sys +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ( + CapabilityUnavailable, + ContractError, + PolicyDenied, + ProfileUnsupported, +) +from devsquad.router import load_routing, resolve_routing +from devsquad.validation import validate_policy, validate_profile_registry + + +def profile( + profile_id, + *, + harness="codex", + family="family-a", + model=None, + permission="read_only", + pool="pool-a", + billing="subscription", + quality="proven", +): + return { + "id": profile_id, + "harness": harness, + "model_family": family, + "model_id": model or f"model-{profile_id}", + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": permission, + "account_pool_id": pool, + "billing_mode": billing, + "quality_status": quality, + "evidence_refs": [f"evidence-{profile_id}"], + } + + +def pool(*, modes=None, concurrency=2, unknown="allow_bounded"): + return { + "allowed_billing_modes": modes or ["subscription"], + "max_concurrency": concurrency, + "unknown_capacity_policy": unknown, + } + + +class RouterTest(unittest.TestCase): + def setUp(self): + self.task = { + "schema_version": 1, + "project": { + "repo_path": "/tmp/router-fixture", + "base_ref": "base", + "target_ref": "target", + }, + "workflow": "branch-review", + "goal": "Review the frozen change.", + "task_class": "fixture-review", + "acceptance": [{ + "id": "review", + "description": "Return a bound review.", + "evidence_kind": "review", + }], + "checks": [], + "scope": {"read_paths": ["src"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + }, + "budget": { + "wall_seconds": 300, + "max_worker_invocations": 3, + "max_revisions": 0, + "max_fallbacks_per_step": 1, + }, + "origin": {"surface": "test"}, + } + self.registry = { + "schema_version": 1, + "profiles": [ + profile("review-a"), + profile("review-b", harness="claude", family="family-b"), + ], + "bindings": { + "review.deep": {"profile_id": "review-b", "version": 7}, + }, + } + self.policy = { + "schema_version": 1, + "id": "fixture-policy", + "version": 3, + "roles": { + "reviewer": [ + {"kind": "alias", "id": "review.deep"}, + {"kind": "profile", "id": "review-a"}, + ], + }, + "task_classes": {"fixture-review": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": {"pool-a": pool()}, + "experiment_budget": {}, + } + + def test_alias_selection_is_deterministic_frozen_and_explained(self): + first = resolve_routing(self.task, self.registry, self.policy) + second = resolve_routing(self.task, self.registry, self.policy) + self.assertEqual(first, second) + reviewer = first["roles"]["reviewer"] + self.assertEqual(reviewer["source"], "automatic") + self.assertEqual(reviewer["selected"]["profile_id"], "review-b") + self.assertEqual( + reviewer["selected"]["binding"], + {"alias": "review.deep", "version": 7, "profile_id": "review-b"}, + ) + self.assertEqual( + [candidate["profile_id"] for candidate in reviewer["fallbacks"]], + ["review-a"], + ) + self.assertEqual(first["capacity"]["pool-a"]["status"], "unknown") + self.assertEqual( + first["capacity"]["pool-a"]["unknown_capacity_policy"], + "allow_bounded", + ) + + self.registry["profiles"][1]["model_id"] = "mutated-after-selection" + self.registry["bindings"]["review.deep"]["version"] = 8 + self.assertNotEqual( + reviewer["selected"]["profile"]["model_id"], + "mutated-after-selection", + ) + self.assertEqual(reviewer["selected"]["binding"]["version"], 7) + + def test_one_role_pin_leaves_headless_lead_automatic(self): + pinned = profile("review-pin", harness="grok", family="family-c") + lead = profile("lead-a", harness="claude", family="family-b") + self.registry["profiles"].extend([pinned, lead]) + self.policy["roles"]["lead"] = [{"kind": "profile", "id": "lead-a"}] + self.task["lead"] = {"mode": "headless"} + self.task["routing"]["overrides"] = { + "reviewer": {"profile_id": "review-pin", "fallback": "none"}, + } + routed = resolve_routing(self.task, self.registry, self.policy) + self.assertEqual(routed["roles"]["reviewer"]["source"], "override") + self.assertEqual( + routed["roles"]["reviewer"]["selected"]["profile_id"], "review-pin", + ) + self.assertEqual(routed["roles"]["reviewer"]["fallbacks"], []) + self.assertEqual(routed["roles"]["lead"]["source"], "automatic") + self.assertEqual(routed["roles"]["lead"]["selected"]["profile_id"], "lead-a") + + self.task["routing"]["overrides"]["implementer"] = { + "profile_id": "review-pin", + } + with self.assertRaisesRegex(ContractError, "not roles in branch-review"): + resolve_routing(self.task, self.registry, self.policy) + + def test_missing_or_statically_invalid_pin_never_falls_back(self): + self.task["routing"]["overrides"] = { + "reviewer": {"profile_id": "missing", "fallback": "policy"}, + } + with self.assertRaisesRegex(ProfileUnsupported, "does not exist"): + resolve_routing(self.task, self.registry, self.policy) + + writer = profile("writer", permission="workspace_write") + self.registry["profiles"].append(writer) + self.task["routing"]["overrides"]["reviewer"]["profile_id"] = "writer" + with self.assertRaisesRegex(ProfileUnsupported, "permission_mismatch"): + resolve_routing(self.task, self.registry, self.policy) + + def test_unavailable_pin_obeys_explicit_fallback_mode(self): + self.registry["profiles"][1]["account_pool_id"] = "pool-b" + self.policy["account_pools"]["pool-b"] = pool() + availability = { + "pool-a": {"status": "available", "in_flight": 0}, + "pool-b": {"status": "exhausted", "in_flight": 0}, + } + self.task["routing"]["overrides"] = { + "reviewer": {"profile_id": "review-b", "fallback": "none"}, + } + with self.assertRaisesRegex(CapabilityUnavailable, "account_pool_exhausted"): + resolve_routing( + self.task, self.registry, self.policy, availability=availability, + ) + + self.task["routing"]["overrides"]["reviewer"]["fallback"] = "policy" + routed = resolve_routing( + self.task, self.registry, self.policy, availability=availability, + ) + reviewer = routed["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "review-a") + self.assertEqual(reviewer["fallback_mode"], "policy") + self.assertEqual(reviewer["excluded"][0]["profile_id"], "review-b") + self.assertEqual(reviewer["excluded"][0]["reason"], "account_pool_exhausted") + + def test_static_policy_filters_are_recorded_without_relaxation(self): + candidates = [ + profile("suspended", quality="suspended"), + profile("trial", quality="trial"), + profile("writer", permission="workspace_write"), + profile("paid", billing="paid_api"), + profile("eligible"), + ] + self.registry = {"schema_version": 1, "profiles": candidates, "bindings": {}} + self.policy["roles"]["reviewer"] = [ + {"kind": "profile", "id": candidate["id"]} for candidate in candidates + ] + routed = resolve_routing(self.task, self.registry, self.policy) + reviewer = routed["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "eligible") + self.assertEqual( + [item["reason"] for item in reviewer["excluded"]], + [ + "profile_suspended", + "quality_below_task_minimum", + "permission_mismatch", + "billing_mode_not_allowed", + ], + ) + + def test_delivery_review_is_different_model_and_prefers_different_harness(self): + implementer = profile( + "implementer", + model="shared-model", + permission="workspace_write", + ) + same_model = profile( + "same-model", + harness="claude", + model="shared-model", + ) + same_harness = profile("same-harness", model="other-codex-model") + other_harness = profile( + "other-harness", harness="claude", family="family-b", model="other-model", + ) + self.registry = { + "schema_version": 1, + "profiles": [implementer, same_model, same_harness, other_harness], + "bindings": {}, + } + self.policy["roles"] = { + "implementer": [{"kind": "profile", "id": "implementer"}], + "reviewer": [ + {"kind": "profile", "id": "same-model"}, + {"kind": "profile", "id": "same-harness"}, + {"kind": "profile", "id": "other-harness"}, + ], + } + self.task["workflow"] = "issue-delivery" + self.task["scope"]["write_paths"] = ["src"] + routed = resolve_routing(self.task, self.registry, self.policy) + self.assertEqual( + routed["roles"]["implementer"]["selected"]["profile_id"], + "implementer", + ) + reviewer = routed["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "other-harness") + reasons = {item["profile_id"]: item["reason"] for item in reviewer["excluded"]} + self.assertEqual(reasons["same-model"], "review_model_not_independent") + self.assertEqual( + [candidate["profile_id"] for candidate in reviewer["fallbacks"]], + ["same-harness"], + ) + + def test_typed_unknown_and_concurrency_capacity_are_fail_closed(self): + self.policy["account_pools"]["pool-a"] = pool(unknown="block") + with self.assertRaisesRegex(CapabilityUnavailable, "currently available"): + resolve_routing(self.task, self.registry, self.policy) + + self.policy["account_pools"]["pool-a"] = pool( + concurrency=3, unknown="allow_bounded", + ) + with self.assertRaisesRegex(CapabilityUnavailable, "currently available"): + resolve_routing( + self.task, + self.registry, + self.policy, + availability={"pool-a": {"status": "unknown", "in_flight": 1}}, + ) + routed = resolve_routing( + self.task, + self.registry, + self.policy, + availability={"pool-a": {"status": "unknown", "in_flight": 0}}, + ) + self.assertEqual(routed["roles"]["reviewer"]["selected"]["profile_id"], "review-b") + + with self.assertRaisesRegex(CapabilityUnavailable, "currently available"): + resolve_routing( + self.task, + self.registry, + self.policy, + availability={"pool-a": {"status": "available", "in_flight": 3}}, + ) + + def test_profile_scoped_windows_exclude_only_applicable_candidate(self): + now = datetime(2026, 9, 27, 16, 0, tzinfo=timezone.utc).isoformat() + + def evidence(profile, status, reason): + target = None if profile is None else { + "harness": profile["harness"], + "model_family": profile["model_family"], + "model_id": profile["model_id"], + } + return { + "schema_version": 1, + "pool_id": "pool-a", + "status": status, + "in_flight": 0, + "observed_at": now, + "evaluated_at": now, + "target": target, + "windows": [{"reason": reason}], + "reasons": [] if status == "available" else [f"{reason}:weekly"], + } + + review_a, review_b = self.registry["profiles"] + availability = { + "pool-a": { + **evidence(None, "unknown", "no_applicable_observations"), + "profiles": { + "review-a": evidence(review_a, "available", "window_available"), + "review-b": evidence(review_b, "exhausted", "window_exhausted"), + }, + }, + } + routed = resolve_routing( + self.task, self.registry, self.policy, availability=availability, + ) + reviewer = routed["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "review-a") + self.assertEqual(reviewer["excluded"][0], { + "reference": {"kind": "alias", "id": "review.deep"}, + "profile_id": "review-b", + "reason": "account_pool_exhausted", + }) + self.assertEqual( + routed["capacity"]["pool-a"]["profiles"]["review-b"]["reasons"], + ["window_exhausted:weekly"], + ) + + def test_policy_missing_task_class_or_required_role_is_denied(self): + del self.policy["task_classes"]["fixture-review"] + with self.assertRaisesRegex(PolicyDenied, "does not authorize task class"): + resolve_routing(self.task, self.registry, self.policy) + self.policy["task_classes"]["fixture-review"] = "proven" + del self.policy["roles"]["reviewer"] + with self.assertRaisesRegex(PolicyDenied, "required roles"): + resolve_routing(self.task, self.registry, self.policy) + + def test_loader_rejects_duplicate_keys_nan_and_hashes_exact_bytes(self): + profiles_bytes = (json.dumps(self.registry, indent=2) + "\n").encode() + policy_bytes = (json.dumps(self.policy, separators=(",", ":")) + "\n").encode() + routed = load_routing(self.task, profiles_bytes, policy_bytes) + self.assertEqual( + routed["profile_registry"]["sha256"], hashlib.sha256(profiles_bytes).hexdigest(), + ) + self.assertEqual( + routed["policy"]["sha256"], hashlib.sha256(policy_bytes).hexdigest(), + ) + with self.assertRaisesRegex(ContractError, "duplicate key"): + load_routing( + self.task, + b'{"schema_version":1,"schema_version":1,"profiles":[],"bindings":{}}', + policy_bytes, + ) + with self.assertRaisesRegex(ContractError, "non-finite"): + load_routing( + self.task, + b'{"schema_version":1,"profiles":[],"bindings":{},"bad":NaN}', + policy_bytes, + ) + with self.assertRaisesRegex(ContractError, "UTF-8 JSON"): + load_routing(self.task, b"\xff", policy_bytes) + + def test_registry_and_account_pool_shapes_are_strict(self): + duplicate = copy.deepcopy(self.registry) + duplicate["profiles"].append(copy.deepcopy(duplicate["profiles"][0])) + with self.assertRaisesRegex(ContractError, "ids must be unique"): + validate_profile_registry(duplicate) + missing_target = copy.deepcopy(self.registry) + missing_target["bindings"]["review.deep"]["profile_id"] = "missing" + with self.assertRaisesRegex(ContractError, "target does not exist"): + validate_profile_registry(missing_target) + + invalid_policy = copy.deepcopy(self.policy) + invalid_policy["account_pools"]["pool-a"]["extra"] = True + with self.assertRaisesRegex(ContractError, "account pool policy fields invalid"): + validate_policy(invalid_policy) + invalid_policy = copy.deepcopy(self.policy) + invalid_policy["account_pools"]["pool-a"]["max_concurrency"] = True + with self.assertRaisesRegex(ContractError, "max_concurrency"): + validate_policy(invalid_policy) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_service.py b/test/core/test_service.py new file mode 100644 index 0000000..5456e21 --- /dev/null +++ b/test/core/test_service.py @@ -0,0 +1,1273 @@ +import hashlib +import json +import multiprocessing +import os +from pathlib import Path +import signal +import subprocess +import sys +import tempfile +import time +import unittest +from unittest import mock +from datetime import datetime, timedelta, timezone + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ExecutionIdentity, LaunchSpec +from devsquad.reports import TERMINAL_REPORT_NAMES +from devsquad.service import Service +from devsquad.store import ConflictError, Store, canonical_json +from devsquad.supervisor import Supervisor, inspect_process +from devsquad_test_fixtures import branch_review_routing_documents + + +def concurrent_receipt_import(database, artifacts, run_id, barrier, results): + store = None + try: + store = Store(Path(database), Path(artifacts)) + barrier.wait(timeout=10) + results.put(Supervisor(store).import_durable(run_id)) + except Exception as exc: + results.put(f"{type(exc).__name__}: {exc}") + finally: + if store is not None: + store.close() + + +def crash_after_attempt_reservation(database, artifacts, run_id, expected_version, package_digest): + store = Store(Path(database), Path(artifacts)) + store.reserve_attempt(run_id, expected_version, "crashed-supervisor", package_digest) + os._exit(23) + + +def crash_after_runner_identity( + database, artifacts, run_id, expected_version, package_path, + package_digest, repo, marker, +): + os.environ["PYTHONPATH"] = package_path + store = Store(Path(database), Path(artifacts)) + identity = ExecutionIdentity("fixture", "1", None, None, None, None) + command = ( + sys.executable, + "-c", + f"from pathlib import Path; Path({marker!r}).write_text('executed')", + ) + spec = LaunchSpec( + 1, "fixture", "cli_exec", command, repo, None, 30, identity, + ) + supervisor = Supervisor(store) + supervisor._release_runner_gate = lambda _: os._exit(24) + supervisor.launch_durable( + run_id, expected_version, spec, "crashed-after-identity", package_digest, + ) + os._exit(99) + + +def crash_after_artifact_finalize_before_import(database, artifacts, run_id): + store = Store(Path(database), Path(artifacts)) + store.commit_durable_import = lambda *args, **kwargs: os._exit(25) + Supervisor(store).import_durable(run_id) + os._exit(99) + + +def crash_after_recovery_cancel_commits(database, artifacts, run_id): + store = Store(Path(database), Path(artifacts)) + request_recovery_cancel = store.request_recovery_cancel + + def request_then_crash(cancel_run_id, attempt_token): + request_recovery_cancel(cancel_run_id, attempt_token) + os._exit(26) + + store.request_recovery_cancel = request_then_crash + Supervisor(store).cancel_orphan(run_id) + os._exit(99) + + +class ServiceTest(unittest.TestCase): + def test_detached_environment_preserves_home_without_ambient_api_credentials(self): + fake_home = self.root / "signed-in-home" + with mock.patch.dict(os.environ, { + "HOME": str(fake_home), "USER": "signed-in-fixture", "ANTHROPIC_API_KEY": "must-not-pass", + "OPENAI_API_KEY": "must-not-pass", "BASH_ENV": "must-not-pass", + }), mock.patch("devsquad.service.subprocess.Popen") as popen: + popen.return_value.pid = 12345 + self.service._spawn_daemon("private-fixture", 1, self.root / "package", "a" * 64) + environment = popen.call_args.kwargs["env"] + self.assertEqual(environment["HOME"], str(fake_home)) + self.assertEqual(environment["USER"], "signed-in-fixture") + self.assertEqual(set(environment), {"HOME", "USER", "PATH", "PYTHONPATH"}) + + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-service-") + self.root = Path(self.temp.name); self.repo = self.root / "repo"; self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run(["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], check=True) + subprocess.run(["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True) + self.profiles_json, self.policy_json = branch_review_routing_documents() + (self.repo / "src").mkdir() + (self.repo / "tests").mkdir() + (self.repo / "src/app.py").write_text("VALUE = 'base'\n") + (self.repo / "tests/test_app.py").write_text("# fixture test\n") + (self.repo / "profiles.json").write_text(self.profiles_json) + (self.repo / "policy.json").write_text(self.policy_json) + subprocess.run(["git", "-C", str(self.repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) + self.task["project"] = {"repo_path": str(self.repo), "base_ref": "HEAD", "target_ref": "HEAD"} + self.task["routing"]["profiles_file"] = "profiles.json"; self.task["routing"]["policy_file"] = "policy.json" + self.service = Service(self.runtime) + + def tearDown(self): self.temp.cleanup() + + def wait_state(self, run_id, states, timeout=8): + deadline=time.monotonic()+timeout + while time.monotonic() None: + """Stage a blob at a raw byte path, bypassing filesystem filename rules.""" + hashed = subprocess.run( + ["git", "-C", str(repo), "hash-object", "-w", "--stdin"], + input=content, check=True, capture_output=True, + ) + sha = hashed.stdout.strip().decode() + cacheinfo = f"100644,{sha},".encode() + path_bytes + subprocess.run( + ["git", "-C", str(repo), "update-index", "--add", "--cacheinfo", cacheinfo], + check=True, capture_output=True, + ) + + def _commit_index(self, repo, message): + subprocess.run( + ["git", "-C", str(repo), "commit", "-m", message], + check=True, capture_output=True, + ) + return subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + check=True, text=True, capture_output=True, + ).stdout.strip() + + def test_branch_review_freezes_exact_commits_and_embedded_routing(self): + task, summary = build_managed_task( + workflow="branch-review", + project_dir=self.repo / "src", + base_ref="main", + target_ref="HEAD", + goal="Review the exact branch delta.", + codex_identity=self.codex, + checks=parse_checks(["python3 -m unittest"]), + ) + + self.assertEqual(task["project"], { + "repo_path": str(self.repo.resolve()), + "base_ref": self.oid, + "target_ref": self.oid, + }) + self.assertEqual(task["scope"], { + "read_paths": ["."], "write_paths": [], + }) + self.assertNotIn("profiles_file", task["routing"]) + self.assertNotIn("policy_file", task["routing"]) + self.assertEqual( + [profile["harness"] for profile in task["routing"]["profiles"]["profiles"]], + ["codex"], + ) + self.assertTrue(all( + not check["required_to_pass"] for check in task["checks"] + )) + self.assertEqual( + [check["id"] for check in task["checks"]], + ["candidate-diff-check", "detected-tests", "user-check-1"], + ) + self.assertEqual(summary["base_oid"], self.oid) + self.assertTrue(summary["planned_roles"]["reviewer"]["profile_id"].startswith("managed-codex-reviewer-")) + self.assertEqual(len(summary["task_sha256"]), 64) + + def test_issue_delivery_is_bounded_to_one_writer_and_required_checks(self): + task, summary = build_managed_task( + workflow="issue-delivery", + project_dir=self.repo, + base_ref="HEAD", + target_ref="HEAD", + goal="Correct the bounded fixture issue.", + codex_identity=self.codex, + write_paths=("src", "src/"), + checks=parse_checks(["python3 -m unittest discover -s test"]), + review_mode="adversarial", + review_focus="state transitions", + claude_model="claude-sonnet-fixture", + claude_effort="high", + ) + + self.assertEqual(task["scope"]["write_paths"], ["src"]) + self.assertEqual(task["review"], { + "mode": "adversarial", "focus": "state transitions", + }) + self.assertTrue(all( + check["required_to_pass"] for check in task["checks"] + )) + self.assertEqual( + task["checks"][0]["argv"], + ["git", "diff", "--check", self.oid, "HEAD", "--"], + ) + self.assertTrue(summary["planned_roles"]["implementer"]["profile_id"].startswith("managed-claude-implementer-")) + self.assertEqual(summary["planned_roles"]["reviewer"]["harness"], "codex") + self.assertIn("different-harness", summary["selection_reason"]) + + default_task, _ = build_managed_task( + workflow="issue-delivery", + project_dir=self.repo, + base_ref="HEAD", + target_ref="HEAD", + goal="Default scope.", + codex_identity=self.codex, + ) + self.assertEqual(default_task["scope"]["write_paths"], ["."]) + + def test_normal_roles_use_stable_trial_aliases_not_concrete_defaults(self): + task, summary = build_managed_task( + workflow="issue-delivery", project_dir=self.repo, + base_ref="HEAD", target_ref="HEAD", goal="Bounded routing proof.", + codex_identity=self.codex, + ) + self.assertEqual(task["routing"]["policy"]["roles"], { + "implementer": [{"kind": "alias", "id": "implement.balanced"}], + "reviewer": [{"kind": "alias", "id": "review.deep"}], + }) + self.assertIn("bounded trial", summary["selection_reason"]) + self.assertTrue(all(p["quality_status"] == "trial" for p in task["routing"]["profiles"]["profiles"])) + + def test_approved_alias_survives_provider_default_and_explicit_pin_is_truthful(self): + original, _ = build_managed_task( + workflow="branch-review", project_dir=self.repo, + base_ref="HEAD", target_ref="HEAD", goal="Routing proof.", + codex_identity=self.codex, + ) + incumbent = copy.deepcopy(original["routing"]["profiles"]["profiles"][0]) + incumbent.update(id="approved-reviewer", quality_status="proven") + bindings = {"reviewer": {"alias": "review.deep", "version": 7, "profile": incumbent}} + changed_default = {**self.codex, "model_id": "gpt-new-default"} + for pinned in (False, True): + with self.subTest(pinned=pinned): + task, summary = build_managed_task( + workflow="branch-review", project_dir=self.repo, + base_ref="HEAD", target_ref="HEAD", goal="Routing proof.", + codex_identity=changed_default, role_bindings=bindings, + pinned_roles=("reviewer",) if pinned else (), + ) + routing = load_routing(task, canonical_json(task["routing"]["profiles"]), canonical_json(task["routing"]["policy"])) + selected = routing["roles"]["reviewer"]["selected"] + self.assertEqual(selected["profile"]["model_id"], "gpt-new-default" if pinned else "gpt-fixture") + self.assertEqual(routing["roles"]["reviewer"]["source"], "override" if pinned else "automatic") + self.assertEqual(summary["planned_roles"]["reviewer"]["selection_mode"], "pinned" if pinned else "approved_alias") + self.assertEqual(task["routing"]["profiles"]["bindings"]["review.deep"]["version"], 7) + + def test_public_promotion_changes_normal_entry_but_not_the_old_run_or_pin(self): + from experiment_runtime_fixture import ExperimentRuntimeFixture + import test_lifecycle as lifecycle_fixtures + + fake_bin = Path(self.temp.name) / "fake-bin" + fake_bin.mkdir() + (fake_bin / "codex").symlink_to(ROOT / "test/core/fakes/codex_review_cli.py") + fake_home = Path(self.temp.name) / "fake-home" + fake_home.mkdir() + (fake_home / "auth.json").write_text("{}\n") + (fake_home / "auth.json").chmod(0o600) + environment = mock.patch.dict(os.environ, {"PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", "CODEX_HOME": str(fake_home)}) + environment.start() + self.addCleanup(environment.stop) + + baseline_task, _ = build_managed_task( + workflow="branch-review", project_dir=self.repo, + base_ref="HEAD", target_ref="HEAD", goal="Normal alias proof.", codex_identity=self.codex, + ) + incumbent = copy.deepcopy(baseline_task["routing"]["profiles"]["profiles"][0]) + incumbent.update(id="approved-a", model_id="gpt-a", quality_status="proven") + candidate = {**copy.deepcopy(incumbent), "id": "approved-b", "model_id": "gpt-b"} + service = Service(Path(self.temp.name) / "runtime") + template = lifecycle_fixtures.lifecycle_template(update_mode="reviewed") + template.update( + policy={"id": "managed-normal-entry", "version": 1}, + allowed_harnesses=["codex"], allowed_model_families=["gpt"], + allowed_account_pools=["codex-subscription"], allowed_task_classes=["managed-review"], + ) + store = Store(service.database, service.artifacts) + self.addCleanup(store.close) + store.bootstrap_profile_binding(template, incumbent, version=7) + fixture = ExperimentRuntimeFixture( + Path(self.temp.name) / "alias-pair", service=service, repo=self.repo, + experiment_id="normal-alias-pair", profiles={"control": incumbent, "candidate": candidate}, + policy=baseline_task["routing"]["policy"], task_class="managed-review", native_review=True, + ) + self.addCleanup(fixture.close) + fixture.run_all() + old_task, _ = build_managed_task( + workflow="branch-review", project_dir=self.repo, + base_ref="HEAD", target_ref="HEAD", goal="Old normal run.", codex_identity=self.codex, + role_bindings=service.normal_entry_bindings("branch-review"), + ) + with mock.patch.object(service, "_spawn_daemon", return_value=0): + old = service.start(old_task, "normal-before-promotion", _internal_review_fixture={"verdict": "clean", "summary": "Offline normal review.", "findings": []}) + self.addCleanup(lambda: service.cancel(old["run_id"])) + self.assertEqual(old["state"], "queued", old) + old_snapshot = store.run(old["run_id"])["mutable_snapshot"] + evaluated = service.policy_evaluate(fixture.spec) + helper = lifecycle_fixtures.ProfileLifecycleTest() + helper.candidate = candidate + qualification = helper.qualification(evaluated) + qualification["task_class"] = "managed-review" + qualification["budget"].update(max_worker_invocations=4, worker_invocations=4) + store.record_profile_qualification(qualification) + store.change_profile_binding(helper.promotion("normal-promote-b")) + self.assertEqual(store.run(old["run_id"])["mutable_snapshot"], old_snapshot) + self.assertEqual(json.loads(old_snapshot)["routing"]["roles"]["reviewer"]["selected"]["profile_id"], "approved-a") + + for pin in (False, True): + output = io.StringIO() + requested = {**self.codex, "model_id": "gpt-new-default" if pin else "gpt-b"} + argv = ["review", "--base", "HEAD", "--project-dir", str(self.repo), + "--runtime-dir", str(service.runtime), "--dry-run", "--json"] + if pin: + argv.extend(["--model", "gpt-new-default", "--effort", "low"]) + with contextlib.redirect_stdout(output), mock.patch.object(cli, "discover_codex_identity", return_value=requested) as discovery: + self.assertEqual(cli.main(argv), 0) + planned = json.loads(output.getvalue())["data"]["planned_roles"]["reviewer"] + self.assertEqual(planned["model_id"], "gpt-new-default" if pin else "gpt-b") + self.assertEqual(planned["selection_mode"], "pinned" if pin else "approved_alias") + discovery.assert_called_once_with(self.repo.resolve(), requested_model=planned["model_id"], requested_effort="low", runtime=service.runtime) + self.assertEqual(service.normal_entry_bindings("branch-review")["reviewer"]["version"], 8) + + def test_managed_delivery_runs_offline_through_review_checks_and_handoff(self): + task, _ = build_managed_task( + workflow="issue-delivery", + project_dir=self.repo, + base_ref="HEAD", + target_ref="HEAD", + goal="Change the bounded fixture value.", + codex_identity=self.codex, + write_paths=("src/app.py",), + claude_model="claude-sonnet-fixture", + ) + service = Service(Path(self.temp.name) / "runtime") + started = service.start( + task, + "managed-delivery-offline", + _internal_implementation_fixture={ + "writes": [{"path": "src/app.py", "content": "VALUE = 2\n"}], + "delay_seconds": 0, + }, + _internal_review_fixture={ + "verdict": "clean", + "summary": "The exact managed candidate is clean.", + "findings": [], + }, + ) + deadline = time.monotonic() + 15 + resumed_review = False + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + if status["state"] == "awaiting_host": + break + if (status["state"] == "queued" + and status["next_action"] == "resume_candidate_review" + and not resumed_review): + service.resume(started["run_id"]) + resumed_review = True + if status["state"] in {"failed", "cancelled"}: + self.fail(f"managed delivery terminalized early: {status}") + time.sleep(0.05) + else: + self.fail(f"managed delivery did not reach handoff: {status}") + self.assertEqual(status["next_action"], "claim_handoff") + self.assertEqual((self.repo / "src/app.py").read_text(), "VALUE = 1\n") + + def test_entry_rejects_ambiguous_scope_focus_identity_and_checks(self): + base = { + "workflow": "issue-delivery", + "project_dir": self.repo, + "base_ref": "HEAD", + "target_ref": "HEAD", + "goal": "Bounded issue.", + "codex_identity": self.codex, + } + for writes in (("../outside",), (str(self.repo),)): + with self.subTest(writes=writes), self.assertRaises(ContractError): + build_managed_task(**base, write_paths=writes) + with self.assertRaisesRegex(ContractError, "focus requires adversarial"): + build_managed_task(**base, review_focus="security") + with self.assertRaisesRegex(ContractError, "identity is invalid"): + build_managed_task(**{**base, "codex_identity": {"model_id": "x"}}) + with self.assertRaises(ContractError): + build_managed_task(**base, check_timeout=True) + with self.assertRaisesRegex(ContractError, "cannot declare write paths"): + build_managed_task( + **{**base, "workflow": "branch-review"}, write_paths=("src",), + ) + with self.assertRaisesRegex(ContractError, "invalid quoting"): + parse_checks(['python3 -c "unterminated']) + with self.assertRaisesRegex(ContractError, "at most 12"): + parse_checks(["true"] * 13) + + def test_check_discovery_inspects_selected_target_not_current_checkout(self): + repo = self._init_repo() + (repo / "src").mkdir() + (repo / "src/app.py").write_text("VALUE = 1\n") + without_tests = self._commit(repo, "no tests") + (repo / "test").mkdir() + (repo / "test/run.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + with_tests = self._commit(repo, "add bash tests") + + subprocess.run( + ["git", "checkout", without_tests], cwd=repo, + check=True, text=True, capture_output=True, + ) + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=without_tests, target_ref=with_tests, + goal="Target tree has tests though the checkout does not.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual(detected["argv"], ["bash", "test/run.sh"]) + + subprocess.run( + ["git", "checkout", with_tests], cwd=repo, + check=True, text=True, capture_output=True, + ) + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=without_tests, target_ref=without_tests, + goal="Target tree lacks tests though the checkout has them.", + codex_identity=self.codex, + ) + self.assertNotIn("detected-tests", [c["id"] for c in task["checks"]]) + + def test_check_discovery_python_tests_follow_selected_target_not_checkout(self): + repo = self._init_repo() + (repo / "src").mkdir() + (repo / "src/app.py").write_text("VALUE = 1\n") + without_tests = self._commit(repo, "no tests tree") + (repo / "tests").mkdir() + (repo / "tests/test_sample.py").write_text("def test_ok():\n assert True\n") + with_tests = self._commit(repo, "add python tests tree") + + subprocess.run( + ["git", "checkout", without_tests], cwd=repo, + check=True, text=True, capture_output=True, + ) + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=without_tests, target_ref=with_tests, + goal="Target tree has Python tests though the checkout does not.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "tests"], + ) + + subprocess.run( + ["git", "checkout", with_tests], cwd=repo, + check=True, text=True, capture_output=True, + ) + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=without_tests, target_ref=without_tests, + goal="Target tree lacks Python tests though the checkout has them.", + codex_identity=self.codex, + ) + self.assertNotIn("detected-tests", [c["id"] for c in task["checks"]]) + + def test_check_discovery_selected_target_without_tests(self): + repo = self._init_repo() + (repo / "src").mkdir() + (repo / "src/app.py").write_text("VALUE = 1\n") + oid = self._commit(repo, "no tests at all") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="No tests are tracked anywhere in the selected target.", + codex_identity=self.codex, + ) + self.assertEqual( + [check["id"] for check in task["checks"]], ["candidate-diff-check"], + ) + + def test_check_discovery_detects_real_python_tests(self): + repo = self._init_repo() + (repo / "tests").mkdir() + (repo / "tests/test_sample.py").write_text("def test_ok():\n assert True\n") + (repo / "tests/README").write_text("Docs alongside real tests.\n") + oid = self._commit(repo, "real python tests") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="Real tracked test*.py files under tests/ are detected.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "tests"], + ) + + def test_check_discovery_detects_unicode_tab_and_newline_named_python_tests(self): + names = ("test_café.py", "test\tplan.py", "test\nplan.py") + for name in names: + with self.subTest(name=name): + repo = self._init_repo() + (repo / "tests").mkdir() + (repo / "tests" / name).write_text("def test_ok():\n assert True\n") + oid = self._commit(repo, "special filename test") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="Unicode, tab, and newline named test*.py files are detected.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "tests"], + ) + + def test_check_discovery_detects_tests_beside_non_utf8_documentation_filename(self): + repo = self._init_repo() + (repo / "tests").mkdir() + (repo / "tests/test_sample.py").write_text("def test_ok():\n assert True\n") + subprocess.run(["git", "-C", str(repo), "add", "-A"], check=True, capture_output=True) + self._stage_raw_path_blob( + repo, b"tests/doc_\xff.md", b"Non-UTF-8 named documentation.\n", + ) + oid = self._commit_index(repo, "python tests plus a non-utf8 doc filename") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="A non-UTF-8 documentation filename alongside real tests must not raise.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "tests"], + ) + + def test_check_discovery_detects_non_utf8_named_python_test_file(self): + repo = self._init_repo() + self._stage_raw_path_blob( + repo, b"tests/test_\xff.py", b"def test_ok():\n assert True\n", + ) + oid = self._commit_index(repo, "non-utf8 named python test file") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="A non-UTF-8 named test*.py file must itself be detected.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "tests"], + ) + + def test_check_discovery_excludes_readme_only_and_symlinked_test_trees(self): + repo = self._init_repo() + (repo / "tests").mkdir() + (repo / "tests/README").write_text("Not a test suite.\n") + readme_only = self._commit(repo, "readme only tests dir") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=readme_only, target_ref=readme_only, + goal="A tests/README must not be treated as Python tests.", + codex_identity=self.codex, + ) + self.assertNotIn("detected-tests", [c["id"] for c in task["checks"]]) + + (repo / "tests/README").unlink() + (repo / "tests").rmdir() + real_dir = Path(self.temp.name) / "external-tests" + real_dir.mkdir() + (real_dir / "test_real.py").write_text("def test_ok():\n assert True\n") + (repo / "tests").symlink_to(real_dir) + symlinked_dir = self._commit(repo, "symlinked tests dir") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=symlinked_dir, target_ref=symlinked_dir, + goal="A symlinked tests directory must not be treated as Python tests.", + codex_identity=self.codex, + ) + self.assertNotIn("detected-tests", [c["id"] for c in task["checks"]]) + + def test_check_discovery_excludes_symlinked_test_file(self): + repo = self._init_repo() + (repo / "tests").mkdir() + (repo / "tests/real_test.py").write_text("def test_ok():\n assert True\n") + (repo / "tests/test_link.py").symlink_to("real_test.py") + oid = self._commit(repo, "symlinked python test file") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="A symlinked test*.py file must not be treated as a real Python test.", + codex_identity=self.codex, + ) + self.assertNotIn("detected-tests", [c["id"] for c in task["checks"]]) + + def test_check_discovery_requires_regular_blob_not_symlink(self): + repo = self._init_repo() + (repo / "test").mkdir() + (repo / "test/test_sample.py").write_text("def test_ok():\n assert True\n") + (repo / "test/real.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + (repo / "test/real.sh").chmod(0o755) + (repo / "test/run.sh").symlink_to("real.sh") + symlinked = self._commit(repo, "symlinked run.sh") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=symlinked, target_ref=symlinked, + goal="A symlinked run.sh must not select Bash.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "test"], + ) + + (repo / "test/run.sh").unlink() + (repo / "test/run.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + regular = self._commit(repo, "regular run.sh") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=regular, target_ref=regular, + goal="A regular blob run.sh selects Bash.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual(detected["argv"], ["bash", "test/run.sh"]) + + def test_check_discovery_combines_bash_and_core_runner(self): + repo = self._init_repo() + (repo / "test").mkdir() + (repo / "test/run.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + (repo / "test/core").mkdir() + (repo / "test/core/test_sample.py").write_text("def test_ok():\n assert True\n") + (repo / "plugin/core/src/devsquad").mkdir(parents=True) + (repo / "plugin/core/src/devsquad/__init__.py").write_text("") + (repo / "scripts").mkdir() + (repo / "scripts/run-core-tests.py").write_text("#!/usr/bin/env python3\n") + oid = self._commit(repo, "bash plus core runner") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="Both Bash and the DevSquad core runner are detected.", + codex_identity=self.codex, + ) + self.assertEqual( + [ + check["argv"] for check in task["checks"] + if check["id"] in {"detected-tests", "detected-core-tests"} + ], + [ + ["bash", "test/run.sh"], + [ + "env", + "PYTHONPATH=plugin/core/src:test/core", + "PYTHONWARNINGS=error::ResourceWarning", + "python3", + "scripts/run-core-tests.py", + ], + ], + ) + + def test_generated_reference_gate_follows_committed_regular_target_and_deduplicates(self): + repo = self._init_repo() + for directory in ("test/core", "plugin/core/src/devsquad", "scripts"): + (repo / directory).mkdir(parents=True) + (repo / "test/run.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + (repo / "test/core/test_sample.py").write_text("# fixture test\n") + (repo / "plugin/core/src/devsquad/__init__.py").write_text("") + (repo / "scripts/run-core-tests.py").write_text("# fixture runner\n") + generator = repo / "scripts/generate-core-reference.py" + generator.write_text("# fixture generator\n") + target = self._commit(repo, "required project gates") + generator.unlink() + generator.symlink_to("run-core-tests.py") + symlink_target = self._commit(repo, "generator becomes a symlink") + for oid, expected in ((target, 1), (symlink_target, 0)): + with self.subTest(target=oid): + task, _ = build_managed_task( + workflow="issue-delivery", project_dir=repo, base_ref=oid, + target_ref=oid, goal="Keep generated contracts current.", + codex_identity=self.codex, + checks=parse_checks(["python3 scripts/generate-core-reference.py --check"] if expected else []), + ) + references = [check for check in task["checks"] if check["argv"] == [ + "python3", "scripts/generate-core-reference.py", "--check", + ]] + self.assertEqual(len(references), expected) + self.assertTrue(all(check["required_to_pass"] for check in task["checks"])) + self.assertIn("detected-core-tests", [check["id"] for check in task["checks"]]) + if expected: + self.assertEqual(references[0]["id"], "generated-core-reference") + + def test_check_discovery_deduplicates_supplied_argv_matching_detected(self): + repo = self._init_repo() + (repo / "test").mkdir() + (repo / "test/run.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + oid = self._commit(repo, "bash only") + task, _ = build_managed_task( + workflow="issue-delivery", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="An exact supplied duplicate of a detected check runs once.", + codex_identity=self.codex, + checks=parse_checks([ + "bash test/run.sh", + "python3 -m unittest discover -s test", + ]), + ) + self.assertEqual( + [(check["id"], check["argv"]) for check in task["checks"]], + [ + ("candidate-diff-check", ["git", "diff", "--check", oid, "HEAD", "--"]), + ("detected-tests", ["bash", "test/run.sh"]), + ( + "user-check-1", + ["python3", "-m", "unittest", "discover", "-s", "test"], + ), + ], + ) + self.assertTrue(all(check["required_to_pass"] for check in task["checks"])) + + def test_check_discovery_deduplicates_supplied_core_runner_argv(self): + repo = self._init_repo() + (repo / "test/core").mkdir(parents=True) + (repo / "test/core/test_sample.py").write_text("def test_ok():\n assert True\n") + (repo / "plugin/core/src/devsquad").mkdir(parents=True) + (repo / "plugin/core/src/devsquad/__init__.py").write_text("") + (repo / "scripts").mkdir() + (repo / "scripts/run-core-tests.py").write_text("#!/usr/bin/env python3\n") + oid = self._commit(repo, "core runner with python tests tree") + task, _ = build_managed_task( + workflow="issue-delivery", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="A supplied duplicate of the full core runner check runs once.", + codex_identity=self.codex, + checks=parse_checks([ + "env PYTHONPATH=plugin/core/src:test/core " + "PYTHONWARNINGS=error::ResourceWarning python3 scripts/run-core-tests.py", + ]), + ) + self.assertEqual( + [(check["id"], check["argv"]) for check in task["checks"]], + [ + ("candidate-diff-check", ["git", "diff", "--check", oid, "HEAD", "--"]), + ( + "detected-tests", + ["python3", "-m", "unittest", "discover", "-s", "test"], + ), + ( + "detected-core-tests", + [ + "env", + "PYTHONPATH=plugin/core/src:test/core", + "PYTHONWARNINGS=error::ResourceWarning", + "python3", + "scripts/run-core-tests.py", + ], + ), + ], + ) + + def test_codex_discovery_selects_requested_exact_model_and_effort(self): + manifest = mock.Mock(verified_versions=("codex-cli fixture",)) + manifest.resolve_binary.return_value = "/fixture/codex" + process = mock.Mock( + stdin=io.StringIO(), stdout=io.StringIO(), + ) + process.poll.return_value = None + models = [ + { + "id": "gpt-default", "family": "gpt", "is_default": True, + "supported_efforts": ["low", "medium"], + }, + { + "id": "gpt-requested", "family": "gpt-6", "is_default": False, + "supported_efforts": ["high", "xhigh"], + }, + ] + with ( + mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), + mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), + mock.patch("devsquad.task_entry.subprocess.Popen", return_value=process) as spawn, + mock.patch("devsquad.task_entry.capture_probe_identity", return_value="fixture-native-start") as capture, + mock.patch("devsquad.task_entry.close_probe") as close, + mock.patch.dict(os.environ, {"HOME": self.temp.name, "USER": "offline-test", "PATH": "/fixture/bin", "OPENAI_API_KEY": "synthetic-denied", "ANTHROPIC_API_KEY": "synthetic-denied", "CODEX_HOME": "/fixture/alternate-home"}), + mock.patch("devsquad.task_entry.JsonLinePeer", return_value=mock.Mock()), + mock.patch("devsquad.task_entry.receive_response", return_value={"result": {}}), + mock.patch("devsquad.task_entry.discover_models", return_value=[]), + mock.patch("devsquad.task_entry.normalize_models", return_value=models), + ): + identity = discover_codex_identity( + self.repo, + requested_model="gpt-requested", + requested_effort="xhigh", + ) + self.assertEqual(identity, { + "harness": "codex", + "harness_version": "codex-cli fixture", + "model_id": "gpt-requested", + "model_family": "gpt-6", + "effort": "xhigh", + }) + capture.assert_called_once_with(process) + close.assert_called_once_with(process, start_identity="fixture-native-start") + self.assertEqual(spawn.call_args.kwargs["env"], {"HOME": self.temp.name, "USER": "offline-test", "PATH": "/fixture/bin"}) + self.assertTrue(spawn.call_args.kwargs["start_new_session"]) + + with ( + mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), + mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), + mock.patch("devsquad.task_entry.subprocess.Popen", return_value=process), + mock.patch("devsquad.task_entry.capture_probe_identity", return_value="fixture-native-start"), + mock.patch("devsquad.task_entry.close_probe"), + mock.patch("devsquad.task_entry.JsonLinePeer", return_value=mock.Mock()), + mock.patch("devsquad.task_entry.receive_response", return_value={"result": {}}), + mock.patch("devsquad.task_entry.discover_models", return_value=[]), + mock.patch("devsquad.task_entry.normalize_models", return_value=models), + ): + with self.assertRaisesRegex(ContractError, "effort 'ultra' is unavailable"): + discover_codex_identity( + self.repo, + requested_model="gpt-requested", + requested_effort="ultra", + ) + + def test_native_discovery_does_not_read_protocol_without_owned_identity(self): + manifest = mock.Mock(verified_versions=("codex-cli fixture",)) + manifest.resolve_binary.return_value = "/fixture/codex" + process = mock.Mock(stdin=io.StringIO(), stdout=io.StringIO()) + with ( + mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), + mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), + mock.patch("devsquad.task_entry.subprocess.Popen", return_value=process), + mock.patch("devsquad.task_entry.capture_probe_identity", return_value=None), + mock.patch("devsquad.task_entry.close_probe") as close, + mock.patch("devsquad.task_entry.JsonLinePeer") as peer, + ): + with self.assertRaisesRegex(ContractError, "ownership is unavailable"): + discover_codex_identity(self.repo) + peer.assert_not_called() + close.assert_called_once_with(process, start_identity=None) + + def test_native_discovery_cleans_owned_child_after_provider_parent_exits(self): + ready = Path(self.temp.name) / "native-child.json" + child_code = ( + "import json,os,signal,sys,time\n" + "signal.signal(signal.SIGTERM, signal.SIG_IGN)\n" + "with open(sys.argv[1], 'w') as handle: json.dump({'pid':os.getpid()},handle)\n" + "while True: time.sleep(1)\n" + ) + provider_code = ( + "import json,os,subprocess,sys,time\n" + f"subprocess.Popen([sys.executable, '-B', '-c', {child_code!r}, {str(ready)!r}], stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n" + f"while not os.path.exists({str(ready)!r}): time.sleep(.01)\n" + "for line in sys.stdin:\n" + " item=json.loads(line)\n" + " if item.get('method') == 'initialize':\n" + " print(json.dumps({'id':item['id'],'result':{}}),flush=True)\n" + " elif item.get('method') == 'model/list':\n" + " print(json.dumps({'id':item['id'],'result':{'data':[{'id':'gpt-fixture','supportedReasoningEfforts':[{'reasoningEffort':'low'}]}],'nextCursor':None}}),flush=True)\n" + " os._exit(0)\n" + ) + manifest = mock.Mock(verified_versions=("codex-cli fixture",)) + manifest.resolve_binary.return_value = "/fixture/codex" + real_popen = subprocess.Popen + spawned = {} + def native_popen(argv, *positional, **keywords): + if argv[:2] != ["/fixture/codex", "app-server"]: + return real_popen(argv, *positional, **keywords) + process = real_popen([sys.executable, "-B", "-c", provider_code], *positional, **keywords) + spawned["process"] = process + return process + from devsquad.task_entry import discover_models as real_discover_models + def discover_then_confirm_parent_exit(*arguments, **keywords): + models = real_discover_models(*arguments, **keywords) + # Observe exit without wait/poll: the unreaped child reserves its + # PID while shared cleanup owns the TERM-ignoring descendant. + deadline = time.monotonic() + 2 + observed = "" + while time.monotonic() < deadline: + observed = subprocess.run(["/bin/ps", "-p", str(spawned["process"].pid), "-o", "stat="], + text=True, capture_output=True, timeout=1).stdout.strip() + if observed.startswith("Z"): + break + time.sleep(0.01) + self.assertTrue(observed.startswith("Z"), "provider parent did not exit before cleanup") + self.assertIsNone(spawned["process"].returncode) + return models + try: + with ( + mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), + mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), + mock.patch("devsquad.task_entry.subprocess.Popen", side_effect=native_popen), + mock.patch("devsquad.task_entry.discover_models", side_effect=discover_then_confirm_parent_exit), + ): + identity = discover_codex_identity(self.repo) + self.assertEqual(identity["model_id"], "gpt-fixture") + child = json.loads(ready.read_text())["pid"] + observed = subprocess.run(["/bin/ps", "-p", str(child), "-o", "stat="], + text=True, capture_output=True, timeout=1).stdout.strip() + self.assertTrue(not observed or observed.startswith("Z"), f"owned child {child} survives: {observed}") + finally: + process = spawned.get("process") + if process is not None: + try: + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + process.wait(timeout=2) + for stream in (process.stdin, process.stdout, process.stderr): + if stream is not None: + stream.close() + + def test_normal_native_discovery_caches_across_projects_and_ingests_shared_quota(self): + from datetime import datetime, timedelta, timezone + manifest = mock.Mock(verified_versions=("codex-cli fixture",)) + manifest.resolve_binary.return_value = "/fixture/codex" + process = mock.Mock(stdin=io.StringIO(), stdout=io.StringIO()) + process.poll.return_value = None + runtime = Path(self.temp.name) / "runtime" + second_repo = self._init_repo() + (second_repo / "README.md").write_text("Second project\n") + self._commit(second_repo, "second project") + real_popen = subprocess.Popen + def native_popen(argv, *positional, **keywords): + return process if argv[:2] == ["/fixture/codex", "app-server"] else real_popen(argv, *positional, **keywords) + reset = int((datetime.now(timezone.utc) + timedelta(days=2)).timestamp()) + account = {"type": "chatgpt", "email": "fixture@example.invalid", "planType": "plus"} + configuration = {"config": {"provider": "native"}} + weekly_used = 100 + failed_limits = False + def reply(peer, request_id, **unused): + if request_id == 1000 and failed_limits: + raise TimeoutError("private provider quota diagnostic") + return {"result": {1: {}, 2: {"account": account}, 3: configuration, + 1000: {"rateLimits": {"primary": {"usedPercent": 5, "windowDurationMins": 300, "resetsAt": reset}, + "secondary": {"usedPercent": weekly_used, "windowDurationMins": 10080, "resetsAt": reset}}}}[request_id]} + with ( + mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), + mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), + mock.patch("devsquad.task_entry.subprocess.Popen", side_effect=native_popen), + mock.patch("devsquad.task_entry.capture_probe_identity", return_value="fixture-native-start"), + mock.patch("devsquad.task_entry.close_probe"), + mock.patch("devsquad.task_entry.JsonLinePeer", return_value=mock.Mock()), + mock.patch("devsquad.task_entry.receive_response", side_effect=reply), + mock.patch("devsquad.task_entry.discover_models", return_value=[{"id": "gpt-fixture", "supportedReasoningEfforts": ["low"]}]) as discovery, + ): + first = discover_codex_identity(self.repo, runtime=runtime) + second = discover_codex_identity(second_repo, runtime=runtime) + discovery.assert_called_once() + self.assertEqual(first, second) + with mock.patch.object(Service, "_spawn_daemon") as spawn: + for project in (self.repo, second_repo): + output = io.StringIO() + with contextlib.redirect_stdout(output): + self.assertEqual(cli.main(["review", "--base", "HEAD", "--project-dir", str(project), "--runtime-dir", str(runtime), "--json"]), 0) + started = json.loads(output.getvalue())["data"] + self.assertEqual(started["state"], "failed") + self.assertEqual(started["service"]["error"]["error"], "CAPABILITY_UNAVAILABLE") + spawn.assert_not_called() + service = Service(runtime) + store = service._store() + try: + target = {key: first[key] for key in ("harness", "model_family", "model_id")} + self.assertEqual(store.capacity_snapshot(first["account_pool_id"], target=target)["status"], "exhausted") + finally: + store.close() + failed_limits = True + discover_codex_identity(self.repo, runtime=runtime) + store = service._store() + try: + self.assertEqual(store.capacity_snapshot(first["account_pool_id"], target=target)["status"], "exhausted") + finally: + store.close() + configuration["config"]["default_model"] = "changed" + configured = discover_codex_identity(second_repo, runtime=runtime) + self.assertEqual(configured["account_pool_id"], first["account_pool_id"]) + self.assertNotEqual(configured["native_scope"], first["native_scope"]) + store = service._store() + try: + self.assertEqual(store.capacity_snapshot(configured["account_pool_id"], target=target)["status"], "exhausted") + finally: + store.close() + account["email"] = "another@example.invalid" + other = discover_codex_identity(self.repo, runtime=runtime) + self.assertNotEqual(first["account_pool_id"], other["account_pool_id"]) + self.assertEqual(discovery.call_count, 3) + store = service._store() + try: + self.assertEqual(store.capacity_snapshot(other["account_pool_id"], target=target)["status"], "unknown") + finally: + store.close() + account["email"] = "fixture@example.invalid" + weekly_used, failed_limits = 5, False + available = discover_codex_identity(second_repo, runtime=runtime) + store = service._store() + try: + self.assertEqual(store.capacity_snapshot(available["account_pool_id"], target=target)["status"], "available") + one = store.claim_start(self.repo, "native-pool-fence-a", {"task": {}}, "fixture-a") + two = store.claim_start(second_repo, "native-pool-fence-b", {"task": {}}, "fixture-b") + store.reserve_pool_capacity(one.run_id, first["account_pool_id"], "qualification", target=target) + with self.assertRaises(ConflictError): + store.reserve_pool_capacity(two.run_id, configured["account_pool_id"], "qualification", target=target) + store.cancel_preparing(one.run_id) + store.cancel_preparing(two.run_id) + finally: + store.close() + self.assertNotIn("fixture@example.invalid", "".join(path.read_text() for path in (runtime / "catalogs").glob("*.json"))) + + def test_normal_alias_rejects_account_or_catalog_change_without_mutating_binding(self): + identity = {**self.codex, "account_pool_id": "native-pool", "native_scope": "scope-a", "catalog_fingerprint": "a" * 64} + task, _ = build_managed_task(workflow="branch-review", project_dir=self.repo, base_ref="HEAD", target_ref="HEAD", goal="Bounded review", codex_identity=identity) + incumbent = copy.deepcopy(task["routing"]["profiles"]["profiles"][0]) + incumbent["quality_status"] = "proven" + binding = {"alias": "review.deep", "version": 7, "profile": incumbent} + for change in ({"account_pool_id": "other-pool"}, {"native_scope": "scope-b"}, {"catalog_fingerprint": "b" * 64}): + with self.assertRaisesRegex(ContractError, "requalification"): + build_managed_task(workflow="branch-review", project_dir=self.repo, base_ref="HEAD", target_ref="HEAD", goal="Bounded review", codex_identity={**identity, **change}, role_bindings={"reviewer": binding}) + self.assertEqual(binding["version"], 7) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_terminal_ux.py b/test/core/test_terminal_ux.py new file mode 100644 index 0000000..3fb4479 --- /dev/null +++ b/test/core/test_terminal_ux.py @@ -0,0 +1,550 @@ +"""Normal terminal operations use real offline workers and the saved gates.""" + +import contextlib +import copy +import io +import json +import os +import shlex +from datetime import datetime, timedelta, timezone +from pathlib import Path +import subprocess +import sys +import tempfile +import time +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad import cli +from devsquad import detached +from devsquad.contracts import ContractError +from devsquad.service import Service +from devsquad.store import ConflictError, Store, canonical_json, request_hash +from devsquad.task_entry import build_managed_task + + +class TerminalUxTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-terminal-") + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.repo = self.new_repo("project") + self.service = Service(self.root / "runtime") + self.identity = { + "harness": "codex", "harness_version": "codex-cli fixture", + "model_id": "gpt-fixture", "model_family": "gpt", "effort": "low", + } + + def new_repo(self, name): + repo = self.root / name + repo.mkdir() + subprocess.run(["git", "init", "-qb", "main", str(repo)], check=True) + subprocess.run(["git", "-C", str(repo), "config", "user.name", "Test"], check=True) + subprocess.run(["git", "-C", str(repo), "config", "user.email", "test@example.invalid"], check=True) + (repo / "app.py").write_text("VALUE = 1\n") + subprocess.run(["git", "-C", str(repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(repo), "commit", "-qm", "fixture"], check=True) + return repo + + def invoke(self, argv): + output, errors = io.StringIO(), io.StringIO() + with contextlib.redirect_stdout(output), contextlib.redirect_stderr(errors): + code = cli.main([*argv, "--runtime-dir", str(self.service.runtime)]) + return code, output.getvalue(), errors.getvalue() + + def normal_run(self, key="normal-review", workflow="review", check=None): + original_start = self.service.start + def offline_start(task, *args): + fixtures = {"_internal_review_fixture": { + "verdict": "clean", "summary": "Exact fixture candidate is clean.", "findings": [], + }} + if task["workflow"] == "issue-delivery": + fixtures["_internal_implementation_fixture"] = { + "iterations": [{"writes": [{"path": "app.py", "content": f"VALUE = {value}\n"}], "delay_seconds": 0} for value in (2, 3, 4)], + } + return original_start(task, *args, **fixtures) + argv = [workflow] + if workflow == "fix": + argv += ["Correct the bounded value.", "--write-path", "app.py"] + argv += ["--base", "HEAD", "--project-dir", str(self.repo), "--idempotency-key", key, "--wait", "--json"] + if check: + argv += ["--check", check] + with mock.patch.object(cli, "discover_codex_identity", return_value=self.identity), mock.patch.object(cli, "_service", return_value=self.service), mock.patch.object(self.service, "start", side_effect=offline_start): + code, output, errors = self.invoke(argv) + if code != 2: + run = self.service.resolve_run_id(None, self.repo) + result = self.service.result(run) + output += "\n" + next(Path(item["path"]).read_text() for item in result["artifacts"] if item["name"] == "receipt.json") + self.assertEqual((code, errors), (2, ""), output) + data = json.loads(output)["data"] + self.assertEqual(data["state"], "awaiting_host") + return data["run_id"] + + def test_zero_one_multiple_and_unrelated_project_resolution(self): + with self.assertRaisesRegex(ContractError, "No saved runs.*squad review"): + self.service.resolve_run_id(None, self.repo) + code, output, _ = self.invoke(["status", "--project-dir", str(self.repo)]) + self.assertEqual(code, 64) + self.assertIn("No saved runs", output) + unrelated = self.new_repo("unrelated") + store = Store(self.service.database, self.service.artifacts) + try: + other = store.claim_start(unrelated, "other", {}, "fixture").run_id + finally: + store.close() + with self.assertRaisesRegex(ContractError, "No saved runs"): + self.service.resolve_run_id(None, self.repo) + run = self.normal_run() + self.assertEqual(self.service.resolve_run_id(None, self.repo), run) + linked = self.root / "linked" + subprocess.run(["git", "-C", str(self.repo), "worktree", "add", "-q", "--detach", str(linked), "HEAD"], check=True) + self.assertEqual(self.service.resolve_run_id(None, linked), run) + self.assertEqual(self.service.resolve_run_id(run, unrelated), run) + second = self.normal_run("second") + with self.assertRaises(ConflictError) as raised: + self.service.resolve_run_id(None, self.repo) + self.assertIn(run, str(raised.exception)) + self.assertIn(second, str(raised.exception)) + self.assertNotIn(other, str(raised.exception)) + code, output, _ = self.invoke(["result", "--project-dir", str(self.repo)]) + self.assertEqual(code, 75) + self.assertIn(run, output) + self.assertIn(second, output) + self.assertNotIn(other, output) + + def test_normal_review_and_fix_finish_without_task_or_decision_json(self): + for workflow in ("review", "fix"): + with self.subTest(workflow=workflow): + run = self.normal_run(workflow, workflow) + code, output, errors = self.invoke(["status", run]) + self.assertEqual((code, errors), (0, ""), output) + next_command = next(line.removeprefix("Next: ") for line in output.splitlines() if line.startswith("Next: ")) + self.assertEqual(shlex.split(next_command), ["squad", "finish", run, "--accept", "--reason=your assessment of the saved evidence"]) + self.assertIn("Guidance: use --reject or --revise", output) + code, output, errors = self.invoke(["finish", run, "--accept", "--reason", "Reviewed exact evidence."]) + self.assertEqual((code, errors), (0, ""), output) + self.assertIn("succeeded", output) + self.assertIn(f"squad result {run}", output) + result = self.service.result(run) + self.assertTrue(result["ready"]) + self.assertEqual(result["state"], "succeeded") + with self.assertRaises(ConflictError): + self.service.finish(run, "accept", "No terminal replay.") + self.assertEqual((self.repo / "app.py").read_text(), "VALUE = 1\n") + + def test_guided_finish_does_not_take_over_or_accept_failed_required_check(self): + run = self.normal_run("failed-check", "fix", "python3 -c 'raise SystemExit(7)'") + with self.assertRaises(ContractError): + self.service.finish(run, "accept", "Cannot override objective failure.") + self.assertEqual(self.service.status(run)["state"], "awaiting_host") + self.service.handoff_claim(run, self.service.status(run)["version"], "other-owner") + for advance in (0, 700): + with mock.patch("devsquad.store._authoritative_now", return_value=datetime.now(timezone.utc) + timedelta(seconds=advance)): + with self.assertRaisesRegex(ConflictError, "claim"): + self.service.finish(run, "reject", "Another claim owns the packet.") + + def interrupt_finish(self, run, *, service=None, reason="Reviewed exact evidence."): + service = service or self.service + captured = {} + def interrupt(run_id, claim, decision): + captured.update(claim=claim, decision=decision) + raise KeyboardInterrupt + with mock.patch.object(service, "handoff_complete", side_effect=interrupt): + with self.assertRaises(KeyboardInterrupt): + service.finish(run, "accept", reason) + self.assertEqual(service.status(run)["state"], "awaiting_host") + return captured + + def saved_artifacts(self, run): + store = self.service._store() + try: + return {item["id"]: (item, Path(item["path"]).read_bytes()) for item in store.artifacts_for_run(run)} + finally: + store.close() + + def assert_artifacts_unchanged(self, artifacts): + for artifact, content in artifacts.values(): + self.assertEqual(Path(artifact["path"]).read_bytes(), content) + + def test_interrupted_finish_recovers_its_exact_live_intent(self): + run = self.normal_run("interrupted-finish") + artifacts = self.saved_artifacts(run) + interrupted = self.interrupt_finish(run) + version = self.service.status(run)["version"] + store = self.service._store() + try: + payload = json.loads(store.connection.execute("SELECT payload FROM events WHERE run_id=? AND type='handoff.acquired' ORDER BY id DESC LIMIT 1", (run,)).fetchone()[0]) + self.assertEqual(payload["terminal_finish"]["decision"], interrupted["decision"]) + self.assertEqual(payload["terminal_finish"]["packet_sha256"], self.service.status(run)["handoff"]["packet_sha256"]) + finally: + store.close() + code, output, errors = self.invoke(["status", run]) + self.assertEqual((code, errors), (0, ""), output) + self.assertIn(f"squad finish {run} --accept --reason='Reviewed exact evidence.'", output) + self.assertIn("retry the exact saved intent", output) + next_command = next(line.removeprefix("Next: ") for line in output.splitlines() if line.startswith("Next: ")) + self.assertEqual(shlex.split(next_command), ["squad", "finish", run, "--accept", "--reason=Reviewed exact evidence."]) + code, output, _ = self.invoke(["status", run, "--json"]) + self.assertNotIn("handoff_view", json.loads(output)["data"]) + retried = self.interrupt_finish(run, service=Service(self.service.runtime)) + self.assertEqual(retried["claim"], interrupted["claim"]) + self.assertEqual(retried["decision"], interrupted["decision"]) + self.assertEqual(self.service.status(run)["version"], version) + self.assertEqual(Service(self.service.runtime).finish(run, "accept", "Reviewed exact evidence.")["state"], "succeeded") + self.assert_artifacts_unchanged(artifacts) + with self.assertRaises(ConflictError): + self.service.finish(run, "accept", "Reviewed exact evidence.") + + def test_expired_interrupted_finish_gets_only_its_own_fresh_fence(self): + run = self.normal_run("expired-finish") + artifacts = self.saved_artifacts(run) + interrupted = self.interrupt_finish(run) + later = datetime.fromisoformat(interrupted["claim"]["expires_at"]) + timedelta(seconds=1) + with mock.patch("devsquad.store._authoritative_now", return_value=later): + recovered = self.interrupt_finish(run, service=Service(self.service.runtime)) + self.assertEqual(recovered["claim"]["fencing_token"], interrupted["claim"]["fencing_token"] + 1) + self.assertEqual(recovered["decision"], interrupted["decision"]) + with self.assertRaises(ConflictError): + self.service.handoff_claim(run, self.service.status(run)["version"], "terminal-operator", interrupted["claim"]) + self.assertEqual(self.service.finish(run, "accept", "Reviewed exact evidence.")["state"], "succeeded") + self.assert_artifacts_unchanged(artifacts) + + def test_finish_retries_authoritative_expiry_rejection_with_exact_fresh_claim(self): + run = self.normal_run("finish-expired-at-completion") + artifacts = self.saved_artifacts(run) + original_complete = self.service.handoff_complete + captured = {} + def expire_before_submission(run_id, claim, decision): + captured.update(claim=claim, decision=decision) + later = datetime.fromisoformat(claim["expires_at"]) + timedelta(seconds=1) + captured["later"] = later + with mock.patch("devsquad.store._authoritative_now", return_value=later): + return original_complete(run_id, claim, decision) + with mock.patch.object(self.service, "handoff_complete", side_effect=expire_before_submission): + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.service.finish(run, "accept", "Reviewed exact evidence.") + store = self.service._store() + try: + rejected = dict(store.connection.execute("SELECT * FROM handoff_submissions WHERE handoff_id=?", (captured["claim"]["handoff_id"],)).fetchone()) + events_before = [dict(row) for row in store.connection.execute("SELECT * FROM events WHERE run_id=? ORDER BY id", (run,))] + finally: + store.close() + self.assertEqual((rejected["outcome"], rejected["rejection_code"]), ("rejected", "expired_claim")) + self.assertEqual(self.service.status(run)["state"], "awaiting_host") + with mock.patch("devsquad.store._authoritative_now", return_value=captured["later"]): + recovered = self.interrupt_finish(run, service=Service(self.service.runtime)) + self.assertEqual(recovered["decision"], captured["decision"]) + self.assertEqual(recovered["claim"]["fencing_token"], captured["claim"]["fencing_token"] + 1) + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.service.handoff_complete(run, captured["claim"], captured["decision"]) + self.assertEqual(Service(self.service.runtime).finish(run, "accept", "Reviewed exact evidence.")["state"], "succeeded") + store = self.service._store() + try: + events_after = [dict(row) for row in store.connection.execute("SELECT * FROM events WHERE run_id=? ORDER BY id", (run,))] + audit = json.loads(store.connection.execute("SELECT payload FROM events WHERE run_id=? AND type='handoff.completion_recovered'", (run,)).fetchone()[0]) + recorded = dict(store.connection.execute("SELECT * FROM handoff_submissions WHERE handoff_id=?", (captured["claim"]["handoff_id"],)).fetchone()) + finally: + store.close() + self.assertEqual(events_after[:len(events_before)], events_before) + self.assertEqual(audit["rejected_submission"], rejected) + self.assertEqual(audit["rejected_submission_sha256"], request_hash(rejected)) + self.assertEqual(audit["fencing_token"], recovered["claim"]["fencing_token"]) + self.assertEqual((recorded["outcome"], recorded["rejection_code"]), ("recorded", None)) + self.assertEqual(recorded["decision_json"], rejected["decision_json"]) + self.assertEqual(recorded["evidence_refs_json"], rejected["evidence_refs_json"]) + result = self.service.result(run) + exported_path = next(Path(item["path"]) for item in result["artifacts"] if item["name"] == "events.jsonl") + exported = [json.loads(line) for line in exported_path.read_text().splitlines()] + self.assertEqual(sum(item["type"] == "handoff.completion_rejected" for item in exported), 1) + recovery = [item for item in exported if item["type"] == "handoff.completion_recovered"] + self.assertEqual(len(recovery), 1) + self.assertEqual(recovery[0]["payload"], audit) + self.assertEqual(sum(item["type"] == "handoff.submitted" for item in exported), 1) + self.assert_artifacts_unchanged(artifacts) + with self.assertRaises(ConflictError): + self.service.finish(run, "accept", "Reviewed exact evidence.") + + def test_pending_finish_negative_leading_reason_round_trips_as_actual_cli(self): + run = self.normal_run("finish-negative-leading-reason") + interrupted = self.interrupt_finish(run, reason="--deferred 'quoted' assessment") + code, output, errors = self.invoke(["status", run]) + self.assertEqual((code, errors), (0, ""), output) + next_command = next(line.removeprefix("Next: ") for line in output.splitlines() if line.startswith("Next: ")) + command = shlex.split(next_command) + self.assertEqual(command, ["squad", "finish", run, "--accept", f"--reason={interrupted['decision']['reason']}"]) + code, output, errors = self.invoke(command[1:]) + self.assertEqual((code, errors), (0, ""), output) + self.assertEqual(self.service.status(run)["state"], "succeeded") + + def test_pending_terminal_finish_refuses_different_intent_and_corrupt_evidence(self): + run = self.normal_run("pending-finish") + interrupted = self.interrupt_finish(run) + before = self.service.status(run) + for disposition, reason in (("accept", "A changed reason."), ("reject", "Reviewed exact evidence.")): + for advance in (0, 700): + with mock.patch("devsquad.store._authoritative_now", return_value=datetime.now(timezone.utc) + timedelta(seconds=advance)): + with self.assertRaises(ConflictError): + self.service.finish(run, disposition, reason) + self.assertEqual(self.service.status(run)["version"], before["version"]) + artifacts = self.saved_artifacts(run) + artifact, content = artifacts[interrupted["decision"]["evidence_refs"][0]["artifact_id"]] + Path(artifact["path"]).write_bytes(b"corrupt pending evidence\n") + with self.assertRaisesRegex(ConflictError, "corrupt"): + self.service.finish(run, "accept", "Reviewed exact evidence.") + self.assertEqual(self.service.status(run)["version"], before["version"]) + Path(artifact["path"]).write_bytes(content) + self.assertEqual(self.service.finish(run, "accept", interrupted["decision"]["reason"])["state"], "succeeded") + + def test_app_claim_named_terminal_operator_is_never_guided_authority(self): + run = self.normal_run("named-app-claim") + self.service.handoff_claim(run, self.service.status(run)["version"], "terminal-operator") + before = self.service.status(run) + for advance in (0, 700): + with mock.patch("devsquad.store._authoritative_now", return_value=datetime.now(timezone.utc) + timedelta(seconds=advance)): + with self.assertRaisesRegex(ConflictError, "claim"): + self.service.finish(run, "accept", "Reviewed exact evidence.") + self.assertEqual(self.service.status(run)["version"], before["version"]) + + def test_later_app_claim_cannot_reuse_an_older_terminal_marker(self): + run = self.normal_run("superseded-terminal-claim") + interrupted = self.interrupt_finish(run) + later = datetime.fromisoformat(interrupted["claim"]["expires_at"]) + timedelta(seconds=1) + with mock.patch("devsquad.store._authoritative_now", return_value=later): + app_claim = self.service.handoff_claim(run, self.service.status(run)["version"], "terminal-operator") + self.assertGreater(app_claim["claim"]["fencing_token"], interrupted["claim"]["fencing_token"]) + with self.assertRaisesRegex(ConflictError, "claim"): + self.service.finish(run, "accept", "Reviewed exact evidence.") + with mock.patch("devsquad.store._authoritative_now", return_value=later + timedelta(seconds=700)): + with self.assertRaisesRegex(ConflictError, "claim"): + self.service.finish(run, "accept", "Reviewed exact evidence.") + + def test_terminal_intent_and_claim_roll_back_together_before_commit(self): + run = self.normal_run("atomic-finish-intent") + before = self.service.status(run) + def fail_marker(value): + if isinstance(value, dict) and "terminal_finish" in value: + raise KeyboardInterrupt + return canonical_json(value) + real_store = self.service._store + def interrupt_after_event(): + store = real_store() + connection = store.connection + class InterruptedConnection: + def __getattr__(self, name): + return getattr(connection, name) + def execute(self, sql, parameters=()): + result = connection.execute(sql, parameters) + if ("INSERT INTO events" in sql and len(parameters) >= 4 + and parameters[2] == "handoff.acquired" + and "terminal_finish" in json.loads(parameters[3])): + raise KeyboardInterrupt + return result + store.connection = InterruptedConnection() + return store + for point in ("before_event", "after_event"): + with self.subTest(point=point), contextlib.ExitStack() as patches: + if point == "before_event": + patches.enter_context(mock.patch("devsquad.store.canonical_json", side_effect=fail_marker)) + else: + patches.enter_context(mock.patch.object(self.service, "_store", side_effect=interrupt_after_event)) + with self.assertRaises(KeyboardInterrupt): + self.service.finish(run, "accept", "Reviewed exact evidence.") + after = self.service.status(run) + self.assertEqual(after["version"], before["version"]) + self.assertIsNone(after["handoff"]["claimed_by"]) + self.assertIsNone(self.service.handoff_view(run)["pending_finish"]) + self.assertEqual(self.service.finish(run, "accept", "Reviewed exact evidence.")["state"], "succeeded") + + def test_guided_finish_claim_is_fenced_by_a_concurrent_version_or_new_handoff(self): + for replacement in (False, True): + with self.subTest(replacement=replacement): + run = self.normal_run(f"finish-race-{replacement}", "fix" if replacement else "review") + self.addCleanup(self.service.cancel, run) + before = self.service.status(run) + real_claim = self.service.handoff_claim + competing = Service(self.service.runtime) + def race(run_id, version, owner, *args, **kwargs): + claim = competing.handoff_claim(run_id, version, "app-owner") + if replacement: + decision = copy.deepcopy(kwargs["_terminal_decision"]) + decision.update(disposition="revise", reason="Review the bounded revision.") + decision["submission_hash"] = request_hash({key: value for key, value in decision.items() if key != "submission_hash"}) + competing.handoff_complete(run_id, claim["claim"], decision) + real_status = competing.status + deadline = time.monotonic() + 30 + def bounded_status(selected_run): + status = real_status(selected_run) + self.assertLess(time.monotonic(), deadline, status) + return status + with mock.patch.object(competing, "status", side_effect=bounded_status): + response, code = cli._wait_for_run(competing, {"run_id": run_id}, resume_candidate_review=True) + self.assertEqual(code, 2, response) + self.assertNotEqual(competing.status(run_id)["handoff"]["handoff_id"], before["handoff"]["handoff_id"]) + return real_claim(run_id, version, owner, *args, **kwargs) + with mock.patch.object(self.service, "handoff_claim", side_effect=race): + with self.assertRaises(ConflictError): + self.service.finish(run, "accept", "Reviewed exact evidence.") + self.assertEqual(self.service.status(run)["state"], "awaiting_host") + + def test_saved_finish_submission_still_resumes_the_existing_continuation(self): + run = self.normal_run("saved-finish-submission") + with mock.patch.object(self.service, "_continue_branch_review_submission", side_effect=KeyboardInterrupt): + with self.assertRaises(KeyboardInterrupt): + self.service.finish(run, "accept", "Reviewed exact evidence.") + self.assertEqual(self.service.status(run)["next_action"], "handoff_submission_saved") + self.assertEqual(Service(self.service.runtime).resume(run)["state"], "succeeded") + + def test_cancelled_pending_terminal_finish_never_recovers_or_replays(self): + run = self.normal_run("cancelled-finish") + self.interrupt_finish(run) + self.assertEqual(self.service.cancel(run)["state"], "cancelled") + for advance in (0, 700): + with mock.patch("devsquad.store._authoritative_now", return_value=datetime.now(timezone.utc) + timedelta(seconds=advance)): + with self.assertRaises(ConflictError): + self.service.finish(run, "accept", "Reviewed exact evidence.") + self.assertEqual(self.service.result(run)["state"], "cancelled") + + def test_reject_and_revision_exhaustion_use_the_existing_terminal_gates(self): + rejected = self.normal_run("reject") + decision = self.service.finish(rejected, "reject", "Review remains incomplete.") + self.assertEqual((decision["state"], decision["disposition"]), ("failed", "reject")) + revised = self.normal_run("no-revision-budget") + decision = self.service.finish(revised, "revise", "Recheck the bounded review.") + self.assertEqual((decision["state"], decision["disposition"]), ("failed", "revise")) + self.assertTrue(self.service.result(revised)["ready"]) + + def test_finish_refuses_corrupt_current_evidence_before_claiming(self): + run = self.normal_run("corrupt-evidence") + before = self.service.status(run) + store = Store(self.service.database, self.service.artifacts) + try: + handoff = store.handoff_snapshot(run) + artifact = store.artifact_named(run, handoff.packet["artifacts"][0]["name"]) + finally: + store.close() + Path(artifact["path"]).write_text("corrupt fixture evidence\n") + with self.assertRaisesRegex(ConflictError, "corrupt"): + self.service.finish(run, "accept", "Must verify saved evidence.") + after = self.service.status(run) + self.assertEqual(after["version"], before["version"]) + self.assertIsNone(after["handoff"]["claimed_by"]) + + def test_status_result_omitted_ids_and_readable_default(self): + run = self.normal_run() + for command in ("status", "result"): + code, output, errors = self.invoke([command, "--project-dir", str(self.repo)]) + self.assertEqual((code, errors), (0, ""), output) + self.assertIn(run, output) + self.assertFalse(output.startswith("{")) + code, output, _ = self.invoke(["status", "--project-dir", str(self.repo), "--json"]) + self.assertEqual(code, 0) + self.assertEqual(json.loads(output)["data"]["run_id"], run) + + def test_native_offline_normal_review_uses_real_discovery_and_public_finish(self): + fake_bin = self.root / "native-bin" + fake_bin.mkdir() + (fake_bin / "codex").symlink_to(ROOT / "test/core/fakes/codex_review_cli.py") + fake_home = self.root / "native-home" + fake_home.mkdir() + (fake_home / "auth.json").write_text("{}\n") + (fake_home / "auth.json").chmod(0o600) + with mock.patch.dict(os.environ, { + "PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", + "CODEX_HOME": str(fake_home), + }): + code, output, errors = self.invoke([ + "review", "--base", "HEAD", "--project-dir", str(self.repo), + "--idempotency-key", "native-normal", "--wait", + ]) + self.assertEqual((code, errors), (2, ""), output) + self.assertIn("Review: clean", output) + self.assertIn("Check candidate-diff-check: passed", output) + self.assertIn("handoff.md", output) + run = self.service.resolve_run_id(None, self.repo) + code, output, errors = self.invoke([ + "finish", "--project-dir", str(self.repo), "--accept", + "--reason", "Reviewed exact offline native fixture evidence.", "--json", + ]) + self.assertEqual((code, errors), (0, ""), output) + self.assertEqual(json.loads(output)["data"]["run_id"], run) + self.assertTrue(self.service.result(run)["ready"]) + + def test_headless_wait_continues_awaiting_host_and_queued_handoff(self): + service = mock.Mock() + service.status.side_effect = [ + {"run_id": "headless", "state": "awaiting_host", "version": 8, "next_action": "continue_headless_lead"}, + {"run_id": "headless", "state": "queued", "version": 9, "next_action": None}, + {"run_id": "headless", "state": "succeeded", "version": 12}, + ] + with mock.patch.object(cli.time, "sleep"): + response, code = cli._wait_for_run(service, {"run_id": "headless"}) + self.assertEqual((code, response["data"]["state"]), (0, "succeeded")) + service.resume.assert_called_once_with("headless") + + def test_headless_wait_keeps_observing_after_detached_continuation_race(self): + service = mock.Mock() + service.status.side_effect = [ + {"run_id": "headless", "state": "awaiting_host", "version": 8, "next_action": "continue_headless_lead"}, + {"run_id": "headless", "state": "running", "version": 10}, + {"run_id": "headless", "state": "succeeded", "version": 12}, + ] + service.resume.side_effect = ConflictError("detached owner already continued") + with mock.patch.object(cli.time, "sleep"): + response, code = cli._wait_for_run(service, {"run_id": "headless"}) + self.assertEqual((code, response["data"]["state"]), (0, "succeeded")) + + def test_queued_headless_handoff_is_not_a_candidate_review(self): + task, _ = build_managed_task( + workflow="issue-delivery", project_dir=self.repo, base_ref="HEAD", + target_ref="HEAD", goal="Correct the bounded fixture value.", + codex_identity=self.identity, write_paths=("app.py",), + ) + lead = copy.deepcopy(task["routing"]["profiles"]["profiles"][-1]) + lead.update({"id": "fixture-lead", "model_id": "gpt-fixture-lead"}) + task["routing"]["profiles"]["profiles"].append(lead) + task["routing"]["policy"]["roles"]["lead"] = [{"kind": "profile", "id": lead["id"]}] + task["lead"] = {"mode": "headless"} + with mock.patch.object(self.service, "_spawn_daemon"): + started = self.service.start( + task, "headless-status", + _internal_implementation_fixture={"writes": [{"path": "app.py", "content": "VALUE = 2\n"}], "delay_seconds": 0}, + _internal_review_fixture={"verdict": "clean", "summary": "Exact candidate is clean.", "findings": []}, + _internal_lead_fixture={"disposition": "accept", "reason": "Exact evidence is sufficient."}, + ) + run_id = started["run_id"] + def run_stage(): + store = Store(self.service.database, self.service.artifacts) + try: + run = store.run(run_id) + finally: + store.close() + with mock.patch.dict(os.environ, {"PYTHONPATH": run["package_path"]}): + self.assertEqual(detached.main([ + "--database", str(self.service.database), "--artifacts", str(self.service.artifacts), + "--run-id", run_id, "--expected-version", str(run["version"]), + "--package-digest", run["package_digest"], + ]), 0) + run_stage() + with mock.patch.object(self.service, "_spawn_daemon"): + self.service.resume(run_id) + with mock.patch.object(Service, "resume"): + run_stage() + waiting = self.service.status(run_id) + self.assertEqual(waiting["next_action"], "continue_headless_lead") + with self.assertRaisesRegex(ConflictError, "headless"): + self.service.finish(run_id, "accept", "Host must not replace the selected lead.") + with mock.patch.object(self.service, "_spawn_daemon"): + self.service.resume(run_id) + queued = self.service.status(run_id) + self.assertEqual(queued["state"], "queued") + self.assertEqual(queued["handoff"]["status"], "open") + self.assertIsNone(queued["next_action"]) + run_stage() + self.assertEqual(self.service.status(run_id)["state"], "succeeded") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_validation.py b/test/core/test_validation.py new file mode 100644 index 0000000..9cf7900 --- /dev/null +++ b/test/core/test_validation.py @@ -0,0 +1,132 @@ +from __future__ import annotations + +import copy +import json +import math +import sys +import unittest +from pathlib import Path + +CORE = Path(__file__).resolve().parents[2] / "plugin" / "core" +sys.path.insert(0, str(CORE / "src")) + +from devsquad.contracts import ContractError, ExecutionIdentity, LaunchSpec, NormalizedResult, validate_launch_payload +from devsquad.validation import validate_policy, validate_profile, validate_task + + +class AdversarialValidationTest(unittest.TestCase): + def setUp(self) -> None: + root = Path(__file__).resolve().parents[2] + self.task = json.loads((root / "docs/plans/engineering-team/examples/issue-delivery.json").read_text()) + self.identity = ExecutionIdentity("codex", None, "openai", "gpt", "gpt-test", "low", ("read",), "read_only", "pool", "verified") + + def assert_contract_error(self, fn, *args): + with self.assertRaises(ContractError): + fn(*args) + + def test_launch_rejects_bool_nan_string_argv_and_environment_abuse(self): + base = LaunchSpec(1, "codex", "cli_exec", ("codex",), "/tmp", None, 3, self.identity).to_dict() + for key, bad in (("timeout_seconds", True), ("timeout_seconds", math.nan), ("argv", "codex exec")): + value = copy.deepcopy(base); value[key] = bad + self.assert_contract_error(validate_launch_payload, value) + for env in ({"PATH": "/tmp"}, {"DEVSQUAD_WORKER": True}): + value = copy.deepcopy(base); value["environment"] = env + self.assert_contract_error(validate_launch_payload, value) + + def test_identity_and_result_validate_all_nested_fields(self): + for change in ({"harness_version": 1}, {"model": ""}, {"tools": ["read", "read"]}, {"account_pool": {}}): + raw = {**self.identity.__dict__, **change} + raw["tools"] = tuple(raw["tools"]) + with self.assertRaises(ContractError): + ExecutionIdentity(**raw) + self.assert_contract_error(NormalizedResult, 1, "succeeded", None, None, "unknown", "not_evaluated", self.identity, None, {"turn": 1}) + + def test_task_rejects_adversarial_nested_types(self): + mutations = [ + lambda t: t["acceptance"][0].__setitem__("id", {"nested": "id"}), + lambda t: t["checks"][0].__setitem__("argv", "python -m test"), + lambda t: t["checks"][0].__setitem__("timeout_seconds", True), + lambda t: t["budget"].__setitem__("wall_seconds", True), + lambda t: t["scope"]["read_paths"].append("../escape"), + lambda t: t["origin"].__setitem__("session_ref", []), + lambda t: t["acceptance"].append(copy.deepcopy(t["acceptance"][0])), + lambda t: t["checks"].append(copy.deepcopy(t["checks"][0])), + lambda t: t.__setitem__( + "acceptance", [copy.deepcopy(t["acceptance"][0])] * 101, + ), + lambda t: t.__setitem__( + "checks", [copy.deepcopy(t["checks"][0])] * 17, + ), + lambda t: t["scope"].__setitem__( + "read_paths", [f"path-{i}" for i in range(257)], + ), + ] + for mutate in mutations: + value = copy.deepcopy(self.task); mutate(value) + self.assert_contract_error(validate_task, value) + + def test_check_outputs_are_explicit_bounded_relative_paths(self): + valid = copy.deepcopy(self.task) + valid["checks"][0]["output_paths"] = ["build", "tests/result.json"] + validate_task(valid) + for outputs in ("build", ["."], ["../escape"], ["/tmp/out"], + [".git"], ["build/../src"], ["./build"], ["build", "build"], + ["build/"] , [None], [f"out-{i}" for i in range(33)]): + invalid = copy.deepcopy(self.task) + invalid["checks"][0]["output_paths"] = outputs + self.assert_contract_error(validate_task, invalid) + + def test_task_accepts_exact_embedded_routing_and_rejects_mixed_sources(self): + profiles = { + "schema_version": 1, + "profiles": [{ + "id": "reviewer", "harness": "codex", + "model_family": "gpt", "model_id": "gpt-test", + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "codex-subscription", + "billing_mode": "subscription", "quality_status": "trial", + "evidence_refs": ["managed-entry"], + }], + "bindings": {}, + } + policy = { + "schema_version": 1, "id": "managed", "version": 1, + "roles": {"reviewer": [{"kind": "profile", "id": "reviewer"}]}, + "task_classes": {"fixture-bugfix-small": "trial"}, + "require_different_model_for_review": True, + "account_pools": { + "codex-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + }, + }, + "experiment_budget": {}, + } + embedded = copy.deepcopy(self.task) + embedded["routing"] = {"profiles": profiles, "policy": policy} + validate_task(embedded) + + mixed = copy.deepcopy(embedded) + mixed["routing"]["profiles_file"] = "profiles.json" + mixed["routing"]["policy_file"] = "policy.json" + self.assert_contract_error(validate_task, mixed) + + partial = copy.deepcopy(embedded) + del partial["routing"]["policy"] + self.assert_contract_error(validate_task, partial) + + def test_profile_and_policy_reject_nested_type_confusion(self): + profile = {"id":"p","harness":"codex","model_family":"gpt","model_id":"m","effort":{"value":"low","transport":"native"},"required_tools":["read"],"permission_policy":"read_only","account_pool_id":"pool","billing_mode":"subscription","quality_status":"proven","evidence_refs":[]} + bad = copy.deepcopy(profile); bad["required_tools"] = [""] + self.assert_contract_error(validate_profile, bad) + policy = {"schema_version":1,"id":"p","version":1,"roles":{"reviewer":[{"kind":"profile","id":"p"}]},"task_classes":{},"require_different_model_for_review":True,"account_pools":{},"experiment_budget":{}} + validate_policy(policy) + for field, value in (("version", True), ("id", {}), ("experiment_budget", {"limit": math.nan})): + bad = copy.deepcopy(policy); bad[field] = value + self.assert_contract_error(validate_policy, bad) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_workspaces.py b/test/core/test_workspaces.py new file mode 100644 index 0000000..af00666 --- /dev/null +++ b/test/core/test_workspaces.py @@ -0,0 +1,269 @@ +import hashlib +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.review_worker import run_review_and_checks +from devsquad.workspaces import ( + assert_clean_inputs, + committed_regular_file, + prepare_check_workspace, + prepare_review_workspace, + repo_relative_config, + resolve_commit, +) + + +class ReviewWorkspaceTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-workspace-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True, + ) + for directory in ("src", "tests", "devsquad"): + (self.repo / directory).mkdir() + (self.repo / "src/app.py").write_text("VALUE = 'base'\n") + (self.repo / "tests/test_app.py").write_text("# base test\n") + (self.repo / "devsquad/profiles.json").write_text('{"profiles":"base"}\n') + (self.repo / "devsquad/policy.json").write_text('{"policy":"base"}\n') + subprocess.run(["git", "-C", str(self.repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.base = resolve_commit(self.repo, "HEAD") + (self.repo / "src/app.py").write_text("VALUE = 'candidate'\n") + subprocess.run(["git", "-C", str(self.repo), "add", "src/app.py"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "candidate"], check=True, + ) + self.target = resolve_commit(self.repo, "HEAD") + + def prepare(self, run_id="run-1", target=None, scope=("src", "tests")): + return prepare_review_workspace( + self.repo, + self.runtime, + "project-1", + run_id, + self.base, + target or self.target, + scope, + required_clean_paths=("devsquad/profiles.json", "devsquad/policy.json"), + ) + + def git(self, *args): + return subprocess.run( + ["git", "-C", str(self.repo), *args], + check=True, + stdout=subprocess.PIPE, + ).stdout + + def test_check_integrity_covers_tracked_modes_index_head_and_hidden_changes(self): + mutations = { + "delete": "p.unlink()", + "mode": "p.chmod(0o755)", + "staged": "p.write_text('changed\\n'); git('add','src/app.py')", + "index-only": "git('update-index','--chmod=+x','src/app.py')", + "hidden": "git('update-index','--assume-unchanged','src/app.py'); p.write_text('hidden\\n')", + "checkout": f"git('checkout','--detach','{self.base}')", + "attach": "git('checkout','-b','check-attached-branch')", + "symlink": "p.unlink(); p.symlink_to('../tests/test_app.py')", + "ignored-source": "Path('.gitignore').write_text('src/hidden.py\\n'); Path('src/hidden.py').write_text('hidden\\n')", + "review-tree": "(Path.cwd().parent/'review-worktree/src/app.py').write_text('changed review\\n')", + } + for name, mutation in mutations.items(): + with self.subTest(name=name): + workspace = self.prepare(run_id=name) + checks_workspace = prepare_check_workspace( + self.repo, self.runtime, "project-1", name, self.target, ("src", "tests"), + ) + task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) + script = ( + "from pathlib import Path; import subprocess; p=Path('src/app.py'); " + "git=lambda *args: subprocess.run(['git',*args],check=True,capture_output=True); " + + mutation + "; print('mutation completed')" + ) + task["checks"] = [{ + "id": "mutation", "argv": [sys.executable, "-c", script], "cwd": ".", + "required_to_pass": False, "timeout_seconds": 10, + # Listing tracked source as an output never permits editing it. + "output_paths": ["src/app.py"], + }, { + "id": "later", "argv": [sys.executable, "-c", "print('must not run')"], + "cwd": ".", "required_to_pass": False, "timeout_seconds": 10, + }] + review = { + "schema_version": 1, "candidate_sha256": workspace["candidate_sha256"], + "base_oid": self.base, "target_oid": self.target, + "review_mode": "standard", "verdict": "clean", "summary": "Fixture", "findings": [], + } + selected = { + "profile_id": "fixture-reviewer", "profile_sha256": "1" * 64, + "profile": {"harness": "fixture"}, + "reference": {"kind": "profile", "id": "fixture-reviewer"}, "binding": None, + } + evidence = run_review_and_checks({ + "task": task, "workspace": workspace, "check_workspace": checks_workspace, + "routing": {"roles": {"reviewer": {"selected": selected}}}, + }, review) + self.assertEqual(evidence["checks"][0]["returncode"], 0) + self.assertEqual(evidence["checks"][0]["status"], "invalidated") + self.assertEqual(evidence["checks"][1]["status"], "not_run") + self.assertEqual(evidence["checks"][1]["stdout"]["total_bytes"], 0) + self.assertFalse(evidence["evaluation"]["accept_allowed"]) + self.assertEqual((self.repo / "src/app.py").read_text(), "VALUE = 'candidate'\n") + + def test_workspace_is_detached_frozen_idempotent_and_checkout_preserving(self): + (self.repo / "notes.txt").write_text("dirty but outside declared inputs\n") + before_head = self.git("rev-parse", "HEAD") + before_status = self.git("status", "--porcelain=v1", "-z") + before_index = hashlib.sha256((self.repo / ".git/index").read_bytes()).hexdigest() + + first = self.prepare() + after_head = self.git("rev-parse", "HEAD") + after_status = self.git("status", "--porcelain=v1", "-z") + after_index = hashlib.sha256((self.repo / ".git/index").read_bytes()).hexdigest() + self.assertEqual((after_head, after_status, after_index), ( + before_head, before_status, before_index, + )) + workspace = Path(first["path"]) + self.assertEqual((workspace / "src/app.py").read_text(), "VALUE = 'candidate'\n") + self.assertEqual( + subprocess.run( + ["git", "-C", str(workspace), "rev-parse", "--abbrev-ref", "HEAD"], + check=True, + text=True, + capture_output=True, + ).stdout.strip(), + "HEAD", + ) + self.assertEqual(first["base_oid"], self.base) + self.assertEqual(first["target_oid"], self.target) + self.assertEqual(first["changed_paths"], ["src/app.py"]) + self.assertEqual(first["scope"], ["src", "tests"]) + + (self.repo / "src/app.py").write_text("VALUE = 'later'\n") + subprocess.run(["git", "-C", str(self.repo), "add", "src/app.py"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "later"], check=True) + second = self.prepare() + self.assertEqual(second, first) + self.assertEqual((workspace / "src/app.py").read_text(), "VALUE = 'candidate'\n") + + def test_checks_get_a_separate_detached_candidate_workspace(self): + review = self.prepare() + check = prepare_check_workspace( + self.repo, + self.runtime, + "project-1", + "run-1", + self.target, + ("src", "tests"), + required_clean_paths=("devsquad/profiles.json", "devsquad/policy.json"), + ) + review_path, check_path = Path(review["path"]), Path(check["path"]) + self.assertNotEqual(review_path, check_path) + self.assertEqual((check_path / "src/app.py").read_text(), "VALUE = 'candidate'\n") + self.assertEqual( + self.git("-C", str(check_path), "rev-parse", "--abbrev-ref", "HEAD").strip(), + b"HEAD", + ) + (check_path / "src/app.py").write_text("check side effect\n") + self.assertEqual((review_path / "src/app.py").read_text(), "VALUE = 'candidate'\n") + + def test_dirty_scoped_tracked_path_is_rejected(self): + (self.repo / "src/app.py").write_text("dirty\n") + with self.assertRaisesRegex(ContractError, "src/app.py"): + self.prepare() + + def test_dirty_required_config_is_rejected_even_outside_scope(self): + (self.repo / "devsquad/policy.json").write_text("dirty\n") + subprocess.run( + ["git", "-C", str(self.repo), "add", "devsquad/policy.json"], check=True, + ) + with self.assertRaisesRegex(ContractError, "devsquad/policy.json"): + self.prepare(scope=("src",)) + + def test_untracked_scoped_path_is_rejected(self): + (self.repo / "tests/new_test.py").write_text("dirty\n") + with self.assertRaisesRegex(ContractError, "tests/new_test.py"): + self.prepare() + + def test_scoped_symlink_escape_is_rejected(self): + outside = self.root / "outside.txt" + outside.write_text("private\n") + (self.repo / "src/leak").symlink_to(outside) + subprocess.run(["git", "-C", str(self.repo), "add", "src/leak"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "symlink"], check=True) + target = resolve_commit(self.repo, "HEAD") + with self.assertRaisesRegex(ContractError, "symlink escapes"): + self.prepare(run_id="symlink-run", target=target) + + def test_existing_workspace_for_another_target_is_rejected(self): + self.prepare(run_id="reused-path") + with self.assertRaisesRegex(ContractError, "different commit"): + self.prepare(run_id="reused-path", target=self.base) + + def test_committed_config_reads_exact_regular_blob(self): + expected = (self.repo / "devsquad/policy.json").read_bytes() + self.assertEqual( + committed_regular_file(self.repo, self.target, "devsquad/policy.json"), + expected, + ) + outside = self.root / "outside-config" + outside.write_text("outside\n") + (self.repo / "devsquad/link.json").symlink_to(outside) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad/link.json"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "linked config"], check=True, + ) + with self.assertRaisesRegex(ContractError, "regular file"): + committed_regular_file( + self.repo, resolve_commit(self.repo, "HEAD"), "devsquad/link.json", + ) + + def test_config_paths_and_identifiers_cannot_escape(self): + self.assertEqual( + repo_relative_config( + self.repo, str(self.repo / "devsquad/policy.json"), "policy_file", + ), + "devsquad/policy.json", + ) + with self.assertRaisesRegex(ContractError, "without traversal"): + repo_relative_config(self.repo, "../policy.json", "policy_file") + with self.assertRaisesRegex(ContractError, "escapes project"): + repo_relative_config(self.repo, str(self.root / "outside.json"), "policy_file") + with self.assertRaisesRegex(ContractError, "safe path segment"): + prepare_review_workspace( + self.repo, + self.runtime, + "../project", + "run", + self.base, + self.target, + ("src",), + ) + + def test_clean_input_helper_accepts_unrelated_dirty_files(self): + (self.repo / "notes.txt").write_text("unrelated\n") + assert_clean_inputs( + self.repo, + ("src", "tests"), + ("devsquad/profiles.json", "devsquad/policy.json"), + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/test_hooks.sh b/test/test_hooks.sh index 385f0a5..74fbf3a 100644 --- a/test/test_hooks.sh +++ b/test/test_hooks.sh @@ -17,6 +17,8 @@ fresh_env() { export CLAUDE_PROJECT_DIR="$TEST_DIR" export HOME="$FAKE_HOME" unset DEVSQUAD_HOOK_DEPTH 2>/dev/null || true + unset DEVSQUAD_WORKER 2>/dev/null || true + unset DEVSQUAD_DELEGATION_DEPTH 2>/dev/null || true } run_hook() { @@ -44,6 +46,16 @@ assert_empty() { fi } +assert_path_missing() { + local label="$1" path="$2" + if [ ! -e "$path" ]; then + PASS=$((PASS + 1)) + else + FAIL=$((FAIL + 1)) + echo " FAIL: $label — unexpected path exists: $path" + fi +} + init_state() { bash -c "source '$PLUGIN_ROOT/lib/state.sh'; d=\$(init_state_dir); init_session_state \"\$d\"" >/dev/null } @@ -164,5 +176,23 @@ fresh_env init_state assert_contains "state dir self-ignores" "$(cat "$TEST_DIR/.devsquad/.gitignore" 2>/dev/null)" "*" +# --- Group 9: durable worker/delegation recursion guards --- +fresh_env +export DEVSQUAD_WORKER=1 +OUT=$(run_hook '{"tool_name":"WebSearch","tool_input":{"query":"nested"}}') +assert_empty "worker pre-tool hook is silent" "$OUT" +OUT=$(bash "$PLUGIN_ROOT/hooks/scripts/stop.sh" 2>/dev/null) +assert_empty "worker stop hook is silent" "$OUT" +OUT=$(bash "$PLUGIN_ROOT/hooks/scripts/session-start.sh" 2>/dev/null) +assert_contains "worker session-start returns valid empty context" "$OUT" '"additionalContext":""' +OUT=$(bash "$PLUGIN_ROOT/hooks/scripts/pre-compact.sh" 2>/dev/null) +assert_contains "worker pre-compact returns valid hook response" "$OUT" '"hookEventName":"PreCompact"' +assert_path_missing "worker hooks do not initialize plugin state" "$TEST_DIR/.devsquad" + +fresh_env +export DEVSQUAD_DELEGATION_DEPTH=2 +OUT=$(run_hook '{"tool_name":"WebSearch","tool_input":{"query":"nested-depth"}}') +assert_empty "delegated pre-tool hook is silent" "$OUT" + echo " hooks: ${PASS} passed, ${FAIL} failed" [ "$FAIL" -eq 0 ] diff --git a/test/test_m1_legacy.sh b/test/test_m1_legacy.sh new file mode 100755 index 0000000..8cc3fe9 --- /dev/null +++ b/test/test_m1_legacy.sh @@ -0,0 +1,340 @@ +#!/usr/bin/env bash +set -u + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +PASS=0 FAIL=0 +ok() { PASS=$((PASS + 1)); } +bad() { FAIL=$((FAIL + 1)); echo " FAIL: $1"; } + +T=$(mktemp -d) +trap 'rm -rf "$T"' EXIT +mkdir -p "$T/bin" "$T/home" "$T/project/.devsquad" +cat > "$T/bin/codex" <<'EOF' +#!/usr/bin/env bash +case "${FAKE_MODE:-fast}" in + fast) printf 'ok' ;; + error) printf 'deliberate offline failure\n' >&2; exit 7 ;; + tree) + sh -c 'trap "" TERM; echo $$ > "$DESC_PID_FILE"; while :; do sleep 1; done' & + wait + ;; + root_ignore) + trap '' TERM + printf '%s\n' "$$" > "$ROOT_PID_FILE" + : > "$ROOT_READY_FILE" + while :; do sleep 1; done + ;; +esac +EOF +chmod +x "$T/bin/codex" + +# Make process inspection deliberately expensive for the deadline regression. +# The old watchdog counted probes rather than elapsed time, so every delayed +# ps call extended the configured timeout. Normal test calls remain unchanged. +cat > "$T/bin/ps" <<'EOF' +#!/usr/bin/env bash +if [[ "${FAKE_PROBE_DELAY:-0}" != "0" ]]; then + /bin/sleep "$FAKE_PROBE_DELAY" +fi +exec /bin/ps "$@" +EOF +chmod +x "$T/bin/ps" + +# A fake provider and its descendant must not retain the private FIFO FD. +# Brief real work also exercises cancellation while retaining the existing +# sub-second fast-call contract. +cat > "$T/bin/fast-echo" <<'EOF' +#!/usr/bin/env bash +[[ ! -p /dev/fd/9 ]] || exit 33 +/bin/bash -c '[[ ! -p /dev/fd/9 ]]' || exit 34 +/bin/sleep 0.05 +exec /bin/echo "$@" +EOF +chmod +x "$T/bin/fast-echo" + +# Observe only the adapter's cancellation seam: record the actual timer job +# and its private paths, call the real helper, then check its wait/cleanup. +# The gated timer-entry seam waits before FIFO open until cancellation is written. +cat > "$T/fast-call.sh" <<'EOF' +#!/usr/bin/env bash +[[ "${IGNORE_TERM:-0}" != "1" ]] || trap '' TERM +source "$1/plugin/lib/codex-wrapper.sh" +eval "$(declare -f _adapter_stop_timer | sed '1s/_adapter_stop_timer/_test_stop_timer/')" +_adapter_stop_timer() { + [[ -n "${1:-}" ]] || return 0 + printf '%s\n' "$1" > "$OBSERVATION_DIR/timer.pid" + printf '%s\n' "$2" > "$OBSERVATION_DIR/control.path" + if [[ "$(LC_ALL=C ls -ld "$2" | awk '{ print substr($1, 1, 10) }')" == "prw-------" ]] && + [[ "$(LC_ALL=C ls -ld "${2%/*}" | awk '{ print substr($1, 1, 10) }')" == "drwx------" ]]; then + : > "$OBSERVATION_DIR/private" + fi + _test_stop_timer "$@" +} +printf() { + builtin printf "$@" || return $? + if [[ "${DELAY_TIMER_OPEN:-0}" == "1" && "${1:-}" == 'cancel\n' && ! -f "$OBSERVATION_DIR/open-started" ]]; then + : > "$OBSERVATION_DIR/cancel-before-open" + fi +} +trap() { + builtin trap "$@" + # Gate the exact timer-entry seam before any FIFO redirection (not inside + # read(), whose redirect would already be applied). Release only after the + # real cancellation write succeeds, while the parent retains duplex FD9. + if [[ "${DELAY_TIMER_OPEN:-0}" == "1" && "${1:-}" == "-" && "${2:-}" == "EXIT" ]]; then + local test_polls=0 + : > "$OBSERVATION_DIR/delay-started" + while [[ ! -f "$OBSERVATION_DIR/cancel-before-open" && "$test_polls" -lt 200 ]]; do + /bin/sleep 0.01 + test_polls=$((test_polls + 1)) + done + [[ -f "$OBSERVATION_DIR/cancel-before-open" ]] || exit 19 + : > "$OBSERVATION_DIR/open-started" + fi +} +kill() { printf '%s\n' "$*" >> "$OBSERVATION_DIR/signals"; builtin kill "$@"; } +if [[ "${TRIGGER_POLL_EXIT:-0}" == "1" ]]; then + _adapter_job_running() { exit 17; } +fi +exec 9>"$OBSERVATION_DIR/caller-fd9" +s=$($NOW_BIN) +if invoke_codex hello 10 2; then call_rc=0; else call_rc=$?; fi +f=$($NOW_BIN) +awk -v s="$s" -v f="$f" 'BEGIN { printf "%.3f", f-s }' > "$OBSERVATION_DIR/elapsed" +printf 'restored\n' >&9 +jobs -pr > "$OBSERVATION_DIR/remaining-jobs" +exit "$call_rc" +EOF +chmod +x "$T/fast-call.sh" + +# Force the portable path even on systems with timeout/gtimeout installed. +NOW_BIN="$T/bin/now"; export NOW_BIN +cat > "$NOW_BIN" <<'EOF' +#!/usr/bin/perl +use Time::HiRes qw(time); +printf "%.6f", time; +EOF +chmod +x "$NOW_BIN" +mkdir "$T/fast-observed" +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" OBSERVATION_DIR="$T/fast-observed" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/fast-echo" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=fast \ + /bin/bash "$T/fast-call.sh" "$ROOT" > "$T/out" 2> "$T/err" +fast_elapsed=$(cat "$T/fast-observed/elapsed") +grep -q 'exec hello' "$T/out" && ok || bad "portable watchdog output" +awk -v e="$fast_elapsed" 'BEGIN { exit !(e < 1.0) }' && ok || bad "portable fast call took ${fast_elapsed}s" +[[ -s "$T/fast-observed/timer.pid" ]] && ok || bad "portable success timer was not observed" +if [[ -s "$T/fast-observed/remaining-jobs" ]] || + { [[ -s "$T/fast-observed/timer.pid" ]] && kill -0 "$(cat "$T/fast-observed/timer.pid")" 2>/dev/null; }; then + bad "portable success left watchdog timer alive" +else + ok +fi +[[ -f "$T/fast-observed/private" ]] && ok || bad "timer FIFO or directory permissions are not private" +[[ -s "$T/fast-observed/control.path" && ! -e "$(dirname "$(cat "$T/fast-observed/control.path")")" ]] && ok || bad "timer private directory was not removed" +grep -qx 'restored' "$T/fast-observed/caller-fd9" && ok || bad "timer cancellation damaged caller FD9" +[[ ! -s "$T/fast-observed/signals" ]] && ok || bad "fast timer cancellation signalled a numeric PID" + +# An ignored TERM disposition is inherited by the timer. Cancellation must +# still reap it immediately rather than waiting out the whole two-second limit. +mkdir "$T/ignored-observed" +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" OBSERVATION_DIR="$T/ignored-observed" IGNORE_TERM=1 DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/fast-echo" CLAUDE_PROJECT_DIR="$T/project" \ + /bin/bash "$T/fast-call.sh" "$ROOT" > "$T/ignored.out" 2> "$T/ignored.err" +ignored_elapsed=$(cat "$T/ignored-observed/elapsed") +grep -q 'exec hello' "$T/ignored.out" && ok || bad "ignored-TERM fast output" +awk -v e="$ignored_elapsed" 'BEGIN { exit !(e < 1.0) }' && ok || bad "ignored-TERM fast call took ${ignored_elapsed}s" +[[ -s "$T/ignored-observed/timer.pid" ]] && ok || bad "ignored-TERM timer was not observed" +if [[ -s "$T/ignored-observed/remaining-jobs" ]] || + { [[ -s "$T/ignored-observed/timer.pid" ]] && kill -0 "$(cat "$T/ignored-observed/timer.pid")" 2>/dev/null; }; then + bad "ignored-TERM success left watchdog timer alive" +else + ok +fi + +mkdir "$T/early-observed" +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" OBSERVATION_DIR="$T/early-observed" DELAY_TIMER_OPEN=1 DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/fast-echo" CLAUDE_PROJECT_DIR="$T/project" \ + /bin/bash "$T/fast-call.sh" "$ROOT" > "$T/early.out" 2> "$T/early.err" +early_elapsed=$(cat "$T/early-observed/elapsed") +grep -q 'exec hello' "$T/early.out" && ok || bad "early timer cancellation lost CLI output" +awk -v e="$early_elapsed" 'BEGIN { exit !(e < 1.0) }' && ok || bad "cancellation before timer read took ${early_elapsed}s" +[[ -s "$T/early-observed/timer.pid" && ! -s "$T/early-observed/remaining-jobs" ]] && ok || bad "early cancellation did not reap timer" +[[ -s "$T/early-observed/control.path" && ! -e "$(dirname "$(cat "$T/early-observed/control.path")")" ]] && ok || bad "early timer cancellation left its FIFO" +[[ -f "$T/early-observed/cancel-before-open" && -f "$T/early-observed/open-started" ]] && ok || bad "cancellation-before-timer-open seam was not exercised" + +# Cancellation after a timer has naturally exited/reaped is safe, and must +# never fall back to signalling its former numeric PID. The helper restores +# the caller's FD9 even when there is no longer another FIFO reader. +TIMER_FIXTURE_DIR="$T/exited-timer"; export TIMER_FIXTURE_DIR +mkdir -m 700 "$TIMER_FIXTURE_DIR" +mkfifo -m 600 "$TIMER_FIXTURE_DIR/control" +/bin/bash -c ' + source "$1/plugin/lib/adapter.sh" + kill() { printf "%s\n" "$*" >> "$TIMER_FIXTURE_DIR/signals"; return 99; } + exec 9>"$TIMER_FIXTURE_DIR/caller-fd9" + (trap - EXIT; if IFS= read -r -t 1 message <>"$TIMER_FIXTURE_DIR/control"; then exit 8; else printf "timeout\n" > "$TIMER_FIXTURE_DIR/deadline"; fi) /dev/null 2>&1 & + timer_pid=$! + wait "$timer_pid" + _adapter_stop_timer "$timer_pid" "$TIMER_FIXTURE_DIR/control" + printf "restored\n" >&9 + jobs -pr > "$TIMER_FIXTURE_DIR/remaining-jobs" +' _ "$ROOT" > "$T/exited.out" 2> "$T/exited.err" +grep -qx 'timeout' "$TIMER_FIXTURE_DIR/deadline" && ok || bad "timer did not reach its natural deadline" +[[ ! -s "$TIMER_FIXTURE_DIR/signals" && ! -s "$TIMER_FIXTURE_DIR/remaining-jobs" ]] && ok || bad "late timer cancellation signalled a PID or retained a job" +grep -qx 'restored' "$TIMER_FIXTURE_DIR/caller-fd9" && ok || bad "late timer cancellation damaged caller FD9" + +# Actual CLI exit status, not timer completion, determines ordinary failures. +mkdir "$T/error-observed" +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" OBSERVATION_DIR="$T/error-observed" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/codex" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=error \ + /bin/bash "$T/fast-call.sh" "$ROOT" > "$T/error.out" 2> "$T/error.err" || error_rc=$? +[[ "${error_rc:-0}" -eq 1 ]] && grep -q '^CLI_ERROR:.*exit 7.*deliberate offline failure' "$T/error.err" && ok || bad "portable watchdog lost CLI failure status/classification" +[[ -s "$T/error-observed/control.path" && ! -e "$(dirname "$(cat "$T/error-observed/control.path")")" && ! -s "$T/error-observed/remaining-jobs" ]] && ok || bad "ordinary CLI failure left a timer or FIFO" + +# The invocation's EXIT cleanup must not survive its local scope and resolve +# caller globals. Preserve a prior caller EXIT hook on success and failure, +# on both the portable branch and the timeout-binary compatibility branch. +mkdir -p "$T/gnu/bin" +cat > "$T/gnu/bin/timeout" <<'EOF' +#!/usr/bin/env bash +shift +exec "$@" +EOF +chmod +x "$T/gnu/bin/timeout" +cat > "$T/caller-scope.sh" <<'EOF' +#!/usr/bin/env bash +source "$1/plugin/lib/codex-wrapper.sh" +timer_dir="$CALLER_STATE" +timer_pid="" +timer_control_file="" +trap 'printf "restored\n" > "$CALLER_STATE/prior-exit"' EXIT +if invoke_codex hello 10 2; then call_rc=0; else call_rc=$?; fi +[[ -f "$timer_dir/control" && -f "$timer_dir/deadline" && -f "$timer_dir/processes" ]] || exit 18 +exit "$call_rc" +EOF +for caller_case in portable-success gnu-success portable-failure; do + caller_state="$T/caller-$caller_case" + mkdir "$caller_state" + printf 'caller\n' > "$caller_state/control" + printf 'caller\n' > "$caller_state/deadline" + printf 'caller\n' > "$caller_state/processes" + caller_path="$T/bin:/usr/bin:/bin" caller_force=1 caller_mode=fast caller_expected=0 + if [[ "$caller_case" == "gnu-success" ]]; then caller_path="$T/gnu/bin:$caller_path"; caller_force=0; fi + if [[ "$caller_case" == "portable-failure" ]]; then caller_mode=error; caller_expected=1; fi + if HOME="$T/home" PATH="$caller_path" CALLER_STATE="$caller_state" DEVSQUAD_FORCE_PORTABLE_TIMEOUT="$caller_force" DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/codex" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE="$caller_mode" \ + /bin/bash "$T/caller-scope.sh" "$ROOT" > "$T/$caller_case.out" 2> "$T/$caller_case.err"; then caller_rc=0; else caller_rc=$?; fi + [[ "$caller_rc" -eq "$caller_expected" && -f "$caller_state/control" && -f "$caller_state/deadline" && -f "$caller_state/processes" ]] && ok || bad "$caller_case EXIT cleanup deleted caller-owned paths" + [[ -f "$caller_state/prior-exit" ]] && grep -qx restored "$caller_state/prior-exit" && ok || bad "$caller_case failed to restore caller EXIT trap" +done + +# An exit while invocation locals are still live must cancel/reap its timer +# and remove its private FIFO, rather than simply dropping the cleanup trap. +mkdir "$T/aborted-observed" +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" OBSERVATION_DIR="$T/aborted-observed" TRIGGER_POLL_EXIT=1 DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/fast-echo" CLAUDE_PROJECT_DIR="$T/project" \ + /bin/bash "$T/fast-call.sh" "$ROOT" > "$T/aborted.out" 2> "$T/aborted.err" || aborted_rc=$? +[[ "${aborted_rc:-0}" -eq 17 && -s "$T/aborted-observed/timer.pid" ]] && ok || bad "active-scope exit cleanup seam was not exercised" +if [[ -s "$T/aborted-observed/timer.pid" ]] && kill -0 "$(cat "$T/aborted-observed/timer.pid")" 2>/dev/null; then + bad "active-scope exit left watchdog timer alive" +else + ok +fi +[[ -s "$T/aborted-observed/control.path" && ! -e "$(dirname "$(cat "$T/aborted-observed/control.path")")" ]] && ok || bad "active-scope exit left private timer paths" + +# Model lookup remains optional when jq is absent from PATH. +mkdir -p "$T/nojq" +ln -s /bin/bash "$T/nojq/bash" +ln -s /usr/bin/dirname "$T/nojq/dirname" +PATH="$T/nojq" CLAUDE_PROJECT_DIR="$T/project" /bin/bash -c \ + 'source "$1/plugin/lib/codex-wrapper.sh"; [[ -z "$(_resolve_codex_model)" ]]' _ "$ROOT" && ok || bad "jq-absent model fallback" + +DESC_PID_FILE="$T/desc.pid"; export DESC_PID_FILE +start=$(date +%s) +PATH="$T/bin:/usr/bin:/bin" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=tree \ + bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; invoke_codex hello 10 1' _ "$ROOT" > "$T/tree.out" 2> "$T/tree.err" || rc=$? +elapsed=$(( $(date +%s) - start )) +[[ "${rc:-0}" -eq 1 ]] && grep -q '^TIMEOUT:' "$T/tree.err" && ok || bad "portable timeout classification" +[[ "$elapsed" -lt 4 ]] && ok || bad "portable timeout elapsed ${elapsed}s" +if [[ -s "$DESC_PID_FILE" ]] && kill -0 "$(cat "$DESC_PID_FILE")" 2>/dev/null; then + bad "portable timeout left descendant alive" +else + ok +fi + +# A root process that ignores TERM must still be KILLed by the watchdog itself; +# escalation after wait would deadlock forever. The readiness marker proves the +# fake entered its signal-resistant loop before the deadline. +ROOT_PID_FILE="$T/root.pid" ROOT_READY_FILE="$T/root.ready" +export ROOT_PID_FILE ROOT_READY_FILE +start=$(perl -MTime::HiRes=time -e 'printf "%.6f", time') +HOME="$T/home" PATH="/usr/bin:/bin" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/codex" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=root_ignore \ + /bin/bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; invoke_codex hello 10 3' _ "$ROOT" > "$T/root.out" 2> "$T/root.err" || root_rc=$? +finish=$(perl -MTime::HiRes=time -e 'printf "%.6f", time') +root_elapsed=$(awk -v s="$start" -v f="$finish" 'BEGIN { printf "%.3f", f-s }') +[[ -f "$ROOT_READY_FILE" ]] && ok || bad "TERM-ignoring root never reached readiness" +[[ "${root_rc:-0}" -eq 1 ]] && grep -q '^TIMEOUT:' "$T/root.err" && ok || bad "TERM-ignoring root timeout classification" +awk -v e="$root_elapsed" 'BEGIN { exit !(e >= 2.8 && e < 5.0) }' && ok || bad "TERM-ignoring root took ${root_elapsed}s" +if [[ -s "$ROOT_PID_FILE" ]] && kill -0 "$(cat "$ROOT_PID_FILE")" 2>/dev/null; then + bad "portable timeout left TERM-ignoring root alive" +else + ok +fi + +# A costly inspection must not extend a one-second real deadline. Keep the +# existing <4s timeout assertion, and verify the resistant root is still gone. +start=$(perl -MTime::HiRes=time -e 'printf "%.6f", time') +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/codex" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=root_ignore FAKE_PROBE_DELAY=0.2 ROOT_PID_FILE="$T/slow-root.pid" ROOT_READY_FILE="$T/slow-root.ready" \ + /bin/bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; invoke_codex hello 10 1' _ "$ROOT" > "$T/slow.out" 2> "$T/slow.err" || slow_rc=$? +finish=$(perl -MTime::HiRes=time -e 'printf "%.6f", time') +slow_elapsed=$(awk -v s="$start" -v f="$finish" 'BEGIN { printf "%.3f", f-s }') +[[ -f "$T/slow-root.ready" ]] && ok || bad "slow-probe root never reached readiness" +[[ "${slow_rc:-0}" -eq 1 ]] && grep -q '^TIMEOUT:' "$T/slow.err" && ok || bad "slow-probe timeout classification" +awk -v e="$slow_elapsed" 'BEGIN { exit !(e >= 0.8 && e < 4.0) }' && ok || bad "slow process probes extended one-second deadline to ${slow_elapsed}s" +if [[ -s "$T/slow-root.pid" ]] && kill -0 "$(cat "$T/slow-root.pid")" 2>/dev/null; then + bad "slow-probe timeout left TERM-ignoring root alive" +else + ok +fi + +# Newline manifests preserve spaces and enumerate TSX. +mkdir -p "$T/project/ui/My Folder" +printf 'export const Card = 1;\n' > "$T/project/ui/My Folder/Card.tsx" +git -C "$T/project" init -q +git -C "$T/project" add "ui/My Folder/Card.tsx" +cat > "$T/bin/agy" <<'EOF' +#!/usr/bin/env bash +cat +printf '{"ok":true}\n' +EOF +chmod +x "$T/bin/agy" +PATH="$T/bin:/usr/bin:/bin" CLAUDE_PROJECT_DIR="$T/project" \ + bash -c 'cd "$1"; source "$2/plugin/lib/gemini-wrapper.sh"; invoke_gemini_with_files "@ui/My Folder" inspect 10 2' _ "$T/project" "$ROOT" > "$T/context" 2>/dev/null +grep -q 'Card.tsx' "$T/context" && ok || bad "TSX path with spaces omitted" + +# Ignored, oversized, escaping and symlink inputs are reported rather than +# silently read or truncated. +printf 'ignored.txt\n' > "$T/project/.gitignore" +printf 'secret\n' > "$T/project/ignored.txt" +printf '123456789\n' > "$T/project/large.ts" +ln -s /etc/passwd "$T/project/escape.ts" +git -C "$T/project" add .gitignore large.ts +git -C "$T/project" add -f ignored.txt escape.ts +PATH="$T/bin:/usr/bin:/bin" CLAUDE_PROJECT_DIR="$T/project" DEVSQUAD_CONTEXT_MAX_FILE_BYTES=4 \ + bash -c 'source "$1/plugin/lib/gemini-wrapper.sh"; invoke_gemini_with_files "$2" inspect 10 2' _ "$ROOT" \ + $'@ignored.txt\n@large.ts\n@escape.ts\n@../outside' > "$T/omitted.out" 2> "$T/omitted.err" +grep -q 'ignored path excluded: ignored.txt' "$T/omitted.err" && ok || bad "ignored file omission not reported" +grep -q 'file exceeds 4 byte limit: large.ts' "$T/omitted.err" && ok || bad "byte limit omission not reported" +grep -q 'symlink input is not followed: escape.ts' "$T/omitted.err" && ok || bad "symlink omission not reported" +grep -q 'path escapes project scope: ../outside' "$T/omitted.err" && ok || bad "path escape not reported" + +# Replacing a tracked directory with a symlink must not let a tracked path read +# bytes outside the project through a symlink ancestor. +mkdir -p "$T/project/safe" "$T/outside" +printf 'inside\n' > "$T/project/safe/code.ts" +git -C "$T/project" add safe/code.ts +mv "$T/project/safe" "$T/project/safe.real" +printf 'OUTSIDE_MARKER\n' > "$T/outside/code.ts" +ln -s "$T/outside" "$T/project/safe" +PATH="$T/bin:/usr/bin:/bin" CLAUDE_PROJECT_DIR="$T/project" \ + bash -c 'source "$1/plugin/lib/gemini-wrapper.sh"; _adapter_invoke() { cat "$ADAPTER_STDIN_FILE"; }; invoke_gemini_with_files "@safe/code.ts" inspect 10 2' _ "$ROOT" > "$T/ancestor.out" 2> "$T/ancestor.err" +grep -q 'OUTSIDE_MARKER' "$T/ancestor.out" && bad "symlink ancestor leaked outside content" || ok +grep -q 'symlink ancestor escapes project scope: safe/code.ts' "$T/ancestor.err" && ok || bad "symlink ancestor omission not reported" + +echo " watchdog timing: fast=${fast_elapsed}s ignored-TERM=${ignored_elapsed}s early-cancel=${early_elapsed}s tree=${elapsed}s resistant=${root_elapsed}s slow-probe=${slow_elapsed}s" +echo " m1_legacy: ${PASS} passed, ${FAIL} failed" +[[ "$FAIL" -eq 0 ]] diff --git a/test/test_m7_packaging.sh b/test/test_m7_packaging.sh new file mode 100755 index 0000000..dd84f7b --- /dev/null +++ b/test/test_m7_packaging.sh @@ -0,0 +1,67 @@ +#!/usr/bin/env bash +# M7 documentation, generator and installer compatibility checks. +# Bash 3.2 compatible; no network or provider CLIs. +set -u + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)" +PASS=0 +FAIL=0 + +pass() { PASS=$((PASS + 1)); } +fail() { FAIL=$((FAIL + 1)); echo " FAIL: $1"; } + +if /bin/bash -n "$REPO_ROOT/install.sh" "$REPO_ROOT/scripts/install-core.sh"; then + pass +else + fail "install scripts must parse under the system Bash" +fi + +if /bin/bash "$REPO_ROOT/install.sh" --help 2>/dev/null | grep -q -- '--core-only'; then + pass +else + fail "composite installer help must document standalone mode" +fi + +if /bin/bash "$REPO_ROOT/scripts/install-core.sh" --help 2>/dev/null | grep -q -- '--mcp-wheelhouse'; then + pass +else + fail "core installer help must document offline MCP installation" +fi + +if python3 "$REPO_ROOT/scripts/generate-core-reference.py" --check >/dev/null 2>&1; then + pass +else + fail "generated command/schema reference must be current" +fi + +if grep -q 'scripts/install-core.sh --status --json' "$REPO_ROOT/docs/RUNTIME-GUIDE.md"; then + pass +else + fail "runtime guide must document drift inspection" +fi + +if grep -q 'native Claude-to-Codex transcript import' "$REPO_ROOT/docs/RUNTIME-GUIDE.md"; then + pass +else + fail "runtime guide must state the native transcript-import boundary" +fi + +if python3 - "$REPO_ROOT/plugin/core/schemas" <<'PY' >/dev/null 2>&1 +import json +from pathlib import Path +import sys +paths = sorted(Path(sys.argv[1]).glob("*.schema.json")) +assert paths +for path in paths: + value = json.loads(path.read_text()) + assert value["$id"].startswith("https://devsquad.local/schemas/") +PY +then + pass +else + fail "all packaged schemas must parse and retain stable identifiers" +fi + +echo " m7_packaging: ${PASS} passed, ${FAIL} failed" +[ "$FAIL" -eq 0 ] diff --git a/test/test_models.sh b/test/test_models.sh index d226e5a..5f43b1b 100644 --- a/test/test_models.sh +++ b/test/test_models.sh @@ -72,10 +72,31 @@ CATEOF CAT="$PLUGIN_ROOT/lib/model-catalog.sh" assert_eq "tier fast picks newest flash" "$(bash "$CAT" resolve gemini fast)" "Gemini 4.0 Flash (Medium)" -assert_eq "tier frontier picks highest ver" "$(bash "$CAT" resolve gemini frontier)" "Claude Opus 4.6 (Thinking)" +assert_eq "tier frontier stays in family" "$(bash "$CAT" resolve gemini frontier)" "Gemini 3.1 Pro (High)" assert_eq "grok fast" "$(bash "$CAT" resolve grok fast)" "grok-composer-2.5-fast" assert_eq "grok frontier non-fast fallback" "$(bash "$CAT" resolve grok frontier)" "grok-build" +# Structured compatibility is authoritative: unrelated families and entries +# without a declared tier cannot participate in selection. +cat > "$CATDIR/models.json" <<'CATEOF' +{"fetched_at":"2026-07-06T00:00:00Z", + "gemini":{"status":"ok","models":[ + {"id":"gemini-pro","family":"gemini","compatibility":{"tiers":["frontier"]}}, + {"id":"claude-opus-99","family":"claude","compatibility":{"tiers":["frontier"]}}, + {"id":"gemini-unknown","family":"gemini","compatibility":{"tiers":[]}} + ]},"grok":{"status":"ok","models":[]},"codex":{"status":"unlistable","models":[]}} +CATEOF +assert_eq "structured compatibility stays in family" "$(bash "$CAT" resolve gemini frontier)" "gemini-pro" +assert_eq "structured unsupported tier is empty" "$(bash "$CAT" resolve gemini fast)" "" + +# Restore the legacy-string fixture for the adapter compatibility checks. +cat > "$CATDIR/models.json" <<'CATEOF' +{"fetched_at":"2026-07-06T00:00:00Z", + "gemini":{"status":"ok","models":["Gemini 3.5 Flash (Medium)","Gemini 4.0 Flash (Medium)","Gemini 3.1 Pro (High)","Claude Opus 4.6 (Thinking)"]}, + "grok":{"status":"ok","models":["grok-composer-2.5-fast","grok-build"]}, + "codex":{"status":"unlistable","models":[]}} +CATEOF + # Adapter integration: tier pin in agent_models resolves through the catalog T4=$(mktemp -d); mkdir -p "$T4/.devsquad" printf '%s' '{"agent_models":{"gemini-reader":"tier:fast"}}' > "$T4/.devsquad/config.json" @@ -102,5 +123,23 @@ else PASS=$((PASS + 1)) fi +# A failed discovery refresh records the error while retaining the prior +# successful model set; it is not interpreted as every model being removed. +REFRESH_BIN=$(mktemp -d) +cat > "$REFRESH_BIN/agy" <<'EOF' +#!/usr/bin/env bash +exit 1 +EOF +cat > "$REFRESH_BIN/grok" <<'EOF' +#!/usr/bin/env bash +exit 1 +EOF +chmod +x "$REFRESH_BIN/agy" "$REFRESH_BIN/grok" +before_models=$(jq -c '.gemini.models' "$CATDIR/models.json") +PATH="$REFRESH_BIN:$PATH" bash -c 'source "$1"; refresh_model_catalog' _ "$CAT" >/dev/null 2>&1 +after_models=$(jq -c '.gemini.models' "$CATDIR/models.json") +assert_eq "failed refresh retains last-good models" "$after_models" "$before_models" +assert_eq "failed refresh records error" "$(jq -r '.gemini.last_refresh_error' "$CATDIR/models.json")" "error" + echo " models: ${PASS} passed, ${FAIL} failed" [ "$FAIL" -eq 0 ] diff --git a/test/test_wrapper_contract.sh b/test/test_wrapper_contract.sh index 5c4d56f..71bc37f 100644 --- a/test/test_wrapper_contract.sh +++ b/test/test_wrapper_contract.sh @@ -18,7 +18,7 @@ ok() { PASS=$((PASS + 1)); } bad() { FAIL=$((FAIL + 1)); echo " FAIL: $1"; } FAKE=$(mktemp -d) -for bin in agy codex grok; do +for bin in agy claude codex grok; do cat > "$FAKE/$bin" <<'FAKESH' #!/bin/bash case "${FAKE_MODE:-success}" in @@ -27,6 +27,7 @@ case "${FAKE_MODE:-success}" in auth) echo "401 unauthorized request" >&2; exit 1 ;; migrate) echo "please migrate to the new suite: IneligibleTierError while authenticating" >&2; exit 1 ;; banner) echo "Signing in with Grok..." ;; + empty) : ;; esac FAKESH chmod +x "$FAKE/$bin" @@ -45,7 +46,7 @@ run_case() { ERR_TXT=$(cat "$errf" 2>/dev/null) } -for spec in "gemini-wrapper.sh:invoke_gemini:gemini" "codex-wrapper.sh:invoke_codex:codex" "grok-wrapper.sh:invoke_grok:grok"; do +for spec in "gemini-wrapper.sh:invoke_gemini:gemini" "claude-wrapper.sh:invoke_claude:claude" "codex-wrapper.sh:invoke_codex:codex" "grok-wrapper.sh:invoke_grok:grok"; do wrapper="${spec%%:*}"; rest="${spec#*:}"; fn="${rest%%:*}"; agent="${rest#*:}" # 1. success: stdout + exit 0 + usage record + contract log @@ -76,6 +77,12 @@ for spec in "gemini-wrapper.sh:invoke_gemini:gemini" "codex-wrapper.sh:invoke_co [ -f "$TDIR/.devsquad/usage/$agent.json" ] && ok || bad "$agent failure usage record" done +# Exit zero without a usable response is a contract failure, not success. +run_case codex-wrapper.sh invoke_codex empty +[ "$EC" -ne 0 ] && ok || bad "codex empty exit-zero treated as success" +printf '%s' "$ERR_TXT" | grep -q '^CLI_ERROR:' && ok || bad "codex empty exit-zero prefix" +[ -f "$TDIR/.devsquad/usage/codex.json" ] && ok || bad "codex empty exit-zero usage record" + # grok-specific: unauthenticated CLI exits 0 with a sign-in banner — the # wrapper must classify that as AUTH_ERROR, not success run_case grok-wrapper.sh invoke_grok banner