From ae0c46d5abfc63f155d028bc86b02d5c7551fad2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Enrique=20Gonz=C3=A1lez=20Paredes?= Date: Sat, 25 Jul 2026 22:44:14 +0200 Subject: [PATCH 01/10] refactor: adopt the development/ tree layout from template v0.6.0 --- {docs => development}/adr/0001-record-architecture-decisions.md | 0 {docs => development}/adr/0002-task-runner-make-over-just.md | 0 {docs => development}/adr/0003-gpu-suite-waiver-for-0.1.0.md | 0 {docs => development}/adr/README.md | 0 {docs => development}/architecture.md | 0 {work => development}/devmm-design.md | 0 {work => development}/devmm-implementation-plan.md | 0 {docs => development}/harness-usage.md | 0 {docs => development}/style.md | 0 {docs => development}/testing.md | 0 {docs => development}/tool-bootstrap.md | 0 {work => development/work}/2026-07-p00-scaffold/plan.md | 0 {work => development/work}/2026-07-p00-scaffold/spec.md | 0 {work => development/work}/2026-07-p00-scaffold/tasks.md | 0 {work => development/work}/2026-07-p01-dtypes-device/plan.md | 0 {work => development/work}/2026-07-p01-dtypes-device/spec.md | 0 {work => development/work}/2026-07-p01-dtypes-device/tasks.md | 0 {work => development/work}/2026-07-p02-layout/plan.md | 0 {work => development/work}/2026-07-p02-layout/spec.md | 0 {work => development/work}/2026-07-p02-layout/tasks.md | 0 .../work}/2026-07-p03-stream-memory-resource/plan.md | 0 .../work}/2026-07-p03-stream-memory-resource/spec.md | 0 .../work}/2026-07-p03-stream-memory-resource/tasks.md | 0 {work => development/work}/2026-07-p04-cpu-mrs/plan.md | 0 {work => development/work}/2026-07-p04-cpu-mrs/spec.md | 0 {work => development/work}/2026-07-p04-cpu-mrs/tasks.md | 0 {work => development/work}/2026-07-p05-buffer-registry/plan.md | 0 {work => development/work}/2026-07-p05-buffer-registry/spec.md | 0 {work => development/work}/2026-07-p05-buffer-registry/tasks.md | 0 {work => development/work}/2026-07-p06-dlpack-abi/plan.md | 0 {work => development/work}/2026-07-p06-dlpack-abi/spec.md | 0 {work => development/work}/2026-07-p06-dlpack-abi/tasks.md | 0 .../work}/2026-07-p07-export-tensor-empty/plan.md | 0 .../work}/2026-07-p07-export-tensor-empty/spec.md | 0 .../work}/2026-07-p07-export-tensor-empty/tasks.md | 0 {work => development/work}/2026-07-p08-runtimes-cpu/plan.md | 0 {work => development/work}/2026-07-p08-runtimes-cpu/spec.md | 0 {work => development/work}/2026-07-p08-runtimes-cpu/tasks.md | 0 {work => development/work}/2026-07-p09-cuda/plan.md | 0 {work => development/work}/2026-07-p09-cuda/spec.md | 0 {work => development/work}/2026-07-p09-cuda/tasks.md | 0 {work => development/work}/2026-07-p10-rocm/plan.md | 0 {work => development/work}/2026-07-p10-rocm/spec.md | 0 {work => development/work}/2026-07-p10-rocm/tasks.md | 0 {work => development/work}/2026-07-p11-integrations/plan.md | 0 {work => development/work}/2026-07-p11-integrations/spec.md | 0 {work => development/work}/2026-07-p11-integrations/tasks.md | 0 .../work}/2026-07-p12-conformance-docs-release/plan.md | 0 .../work}/2026-07-p12-conformance-docs-release/spec.md | 0 .../work}/2026-07-p12-conformance-docs-release/tasks.md | 0 50 files changed, 0 insertions(+), 0 deletions(-) rename {docs => development}/adr/0001-record-architecture-decisions.md (100%) rename {docs => development}/adr/0002-task-runner-make-over-just.md (100%) rename {docs => development}/adr/0003-gpu-suite-waiver-for-0.1.0.md (100%) rename {docs => development}/adr/README.md (100%) rename {docs => development}/architecture.md (100%) rename {work => development}/devmm-design.md (100%) rename {work => development}/devmm-implementation-plan.md (100%) rename {docs => development}/harness-usage.md (100%) rename {docs => development}/style.md (100%) rename {docs => development}/testing.md (100%) rename {docs => development}/tool-bootstrap.md (100%) rename {work => development/work}/2026-07-p00-scaffold/plan.md (100%) rename {work => development/work}/2026-07-p00-scaffold/spec.md (100%) rename {work => development/work}/2026-07-p00-scaffold/tasks.md (100%) rename {work => development/work}/2026-07-p01-dtypes-device/plan.md (100%) rename {work => development/work}/2026-07-p01-dtypes-device/spec.md (100%) rename {work => development/work}/2026-07-p01-dtypes-device/tasks.md (100%) rename {work => development/work}/2026-07-p02-layout/plan.md (100%) rename {work => development/work}/2026-07-p02-layout/spec.md (100%) rename {work => development/work}/2026-07-p02-layout/tasks.md (100%) rename {work => development/work}/2026-07-p03-stream-memory-resource/plan.md (100%) rename {work => development/work}/2026-07-p03-stream-memory-resource/spec.md (100%) rename {work => development/work}/2026-07-p03-stream-memory-resource/tasks.md (100%) rename {work => development/work}/2026-07-p04-cpu-mrs/plan.md (100%) rename {work => development/work}/2026-07-p04-cpu-mrs/spec.md (100%) rename {work => development/work}/2026-07-p04-cpu-mrs/tasks.md (100%) rename {work => development/work}/2026-07-p05-buffer-registry/plan.md (100%) rename {work => development/work}/2026-07-p05-buffer-registry/spec.md (100%) rename {work => development/work}/2026-07-p05-buffer-registry/tasks.md (100%) rename {work => development/work}/2026-07-p06-dlpack-abi/plan.md (100%) rename {work => development/work}/2026-07-p06-dlpack-abi/spec.md (100%) rename {work => development/work}/2026-07-p06-dlpack-abi/tasks.md (100%) rename {work => development/work}/2026-07-p07-export-tensor-empty/plan.md (100%) rename {work => development/work}/2026-07-p07-export-tensor-empty/spec.md (100%) rename {work => development/work}/2026-07-p07-export-tensor-empty/tasks.md (100%) rename {work => development/work}/2026-07-p08-runtimes-cpu/plan.md (100%) rename {work => development/work}/2026-07-p08-runtimes-cpu/spec.md (100%) rename {work => development/work}/2026-07-p08-runtimes-cpu/tasks.md (100%) rename {work => development/work}/2026-07-p09-cuda/plan.md (100%) rename {work => development/work}/2026-07-p09-cuda/spec.md (100%) rename {work => development/work}/2026-07-p09-cuda/tasks.md (100%) rename {work => development/work}/2026-07-p10-rocm/plan.md (100%) rename {work => development/work}/2026-07-p10-rocm/spec.md (100%) rename {work => development/work}/2026-07-p10-rocm/tasks.md (100%) rename {work => development/work}/2026-07-p11-integrations/plan.md (100%) rename {work => development/work}/2026-07-p11-integrations/spec.md (100%) rename {work => development/work}/2026-07-p11-integrations/tasks.md (100%) rename {work => development/work}/2026-07-p12-conformance-docs-release/plan.md (100%) rename {work => development/work}/2026-07-p12-conformance-docs-release/spec.md (100%) rename {work => development/work}/2026-07-p12-conformance-docs-release/tasks.md (100%) diff --git a/docs/adr/0001-record-architecture-decisions.md b/development/adr/0001-record-architecture-decisions.md similarity index 100% rename from docs/adr/0001-record-architecture-decisions.md rename to development/adr/0001-record-architecture-decisions.md diff --git a/docs/adr/0002-task-runner-make-over-just.md b/development/adr/0002-task-runner-make-over-just.md similarity index 100% rename from docs/adr/0002-task-runner-make-over-just.md rename to development/adr/0002-task-runner-make-over-just.md diff --git a/docs/adr/0003-gpu-suite-waiver-for-0.1.0.md b/development/adr/0003-gpu-suite-waiver-for-0.1.0.md similarity index 100% rename from docs/adr/0003-gpu-suite-waiver-for-0.1.0.md rename to development/adr/0003-gpu-suite-waiver-for-0.1.0.md diff --git a/docs/adr/README.md b/development/adr/README.md similarity index 100% rename from docs/adr/README.md rename to development/adr/README.md diff --git a/docs/architecture.md b/development/architecture.md similarity index 100% rename from docs/architecture.md rename to development/architecture.md diff --git a/work/devmm-design.md b/development/devmm-design.md similarity index 100% rename from work/devmm-design.md rename to development/devmm-design.md diff --git a/work/devmm-implementation-plan.md b/development/devmm-implementation-plan.md similarity index 100% rename from work/devmm-implementation-plan.md rename to development/devmm-implementation-plan.md diff --git a/docs/harness-usage.md b/development/harness-usage.md similarity index 100% rename from docs/harness-usage.md rename to development/harness-usage.md diff --git a/docs/style.md b/development/style.md similarity index 100% rename from docs/style.md rename to development/style.md diff --git a/docs/testing.md b/development/testing.md similarity index 100% rename from docs/testing.md rename to development/testing.md diff --git a/docs/tool-bootstrap.md b/development/tool-bootstrap.md similarity index 100% rename from docs/tool-bootstrap.md rename to development/tool-bootstrap.md diff --git a/work/2026-07-p00-scaffold/plan.md b/development/work/2026-07-p00-scaffold/plan.md similarity index 100% rename from work/2026-07-p00-scaffold/plan.md rename to development/work/2026-07-p00-scaffold/plan.md diff --git a/work/2026-07-p00-scaffold/spec.md b/development/work/2026-07-p00-scaffold/spec.md similarity index 100% rename from work/2026-07-p00-scaffold/spec.md rename to development/work/2026-07-p00-scaffold/spec.md diff --git a/work/2026-07-p00-scaffold/tasks.md b/development/work/2026-07-p00-scaffold/tasks.md similarity index 100% rename from work/2026-07-p00-scaffold/tasks.md rename to development/work/2026-07-p00-scaffold/tasks.md diff --git a/work/2026-07-p01-dtypes-device/plan.md b/development/work/2026-07-p01-dtypes-device/plan.md similarity index 100% rename from work/2026-07-p01-dtypes-device/plan.md rename to development/work/2026-07-p01-dtypes-device/plan.md diff --git a/work/2026-07-p01-dtypes-device/spec.md b/development/work/2026-07-p01-dtypes-device/spec.md similarity index 100% rename from work/2026-07-p01-dtypes-device/spec.md rename to development/work/2026-07-p01-dtypes-device/spec.md diff --git a/work/2026-07-p01-dtypes-device/tasks.md b/development/work/2026-07-p01-dtypes-device/tasks.md similarity index 100% rename from work/2026-07-p01-dtypes-device/tasks.md rename to development/work/2026-07-p01-dtypes-device/tasks.md diff --git a/work/2026-07-p02-layout/plan.md b/development/work/2026-07-p02-layout/plan.md similarity index 100% rename from work/2026-07-p02-layout/plan.md rename to development/work/2026-07-p02-layout/plan.md diff --git a/work/2026-07-p02-layout/spec.md b/development/work/2026-07-p02-layout/spec.md similarity index 100% rename from work/2026-07-p02-layout/spec.md rename to development/work/2026-07-p02-layout/spec.md diff --git a/work/2026-07-p02-layout/tasks.md b/development/work/2026-07-p02-layout/tasks.md similarity index 100% rename from work/2026-07-p02-layout/tasks.md rename to development/work/2026-07-p02-layout/tasks.md diff --git a/work/2026-07-p03-stream-memory-resource/plan.md b/development/work/2026-07-p03-stream-memory-resource/plan.md similarity index 100% rename from work/2026-07-p03-stream-memory-resource/plan.md rename to development/work/2026-07-p03-stream-memory-resource/plan.md diff --git a/work/2026-07-p03-stream-memory-resource/spec.md b/development/work/2026-07-p03-stream-memory-resource/spec.md similarity index 100% rename from work/2026-07-p03-stream-memory-resource/spec.md rename to development/work/2026-07-p03-stream-memory-resource/spec.md diff --git a/work/2026-07-p03-stream-memory-resource/tasks.md b/development/work/2026-07-p03-stream-memory-resource/tasks.md similarity index 100% rename from work/2026-07-p03-stream-memory-resource/tasks.md rename to development/work/2026-07-p03-stream-memory-resource/tasks.md diff --git a/work/2026-07-p04-cpu-mrs/plan.md b/development/work/2026-07-p04-cpu-mrs/plan.md similarity index 100% rename from work/2026-07-p04-cpu-mrs/plan.md rename to development/work/2026-07-p04-cpu-mrs/plan.md diff --git a/work/2026-07-p04-cpu-mrs/spec.md b/development/work/2026-07-p04-cpu-mrs/spec.md similarity index 100% rename from work/2026-07-p04-cpu-mrs/spec.md rename to development/work/2026-07-p04-cpu-mrs/spec.md diff --git a/work/2026-07-p04-cpu-mrs/tasks.md b/development/work/2026-07-p04-cpu-mrs/tasks.md similarity index 100% rename from work/2026-07-p04-cpu-mrs/tasks.md rename to development/work/2026-07-p04-cpu-mrs/tasks.md diff --git a/work/2026-07-p05-buffer-registry/plan.md b/development/work/2026-07-p05-buffer-registry/plan.md similarity index 100% rename from work/2026-07-p05-buffer-registry/plan.md rename to development/work/2026-07-p05-buffer-registry/plan.md diff --git a/work/2026-07-p05-buffer-registry/spec.md b/development/work/2026-07-p05-buffer-registry/spec.md similarity index 100% rename from work/2026-07-p05-buffer-registry/spec.md rename to development/work/2026-07-p05-buffer-registry/spec.md diff --git a/work/2026-07-p05-buffer-registry/tasks.md b/development/work/2026-07-p05-buffer-registry/tasks.md similarity index 100% rename from work/2026-07-p05-buffer-registry/tasks.md rename to development/work/2026-07-p05-buffer-registry/tasks.md diff --git a/work/2026-07-p06-dlpack-abi/plan.md b/development/work/2026-07-p06-dlpack-abi/plan.md similarity index 100% rename from work/2026-07-p06-dlpack-abi/plan.md rename to development/work/2026-07-p06-dlpack-abi/plan.md diff --git a/work/2026-07-p06-dlpack-abi/spec.md b/development/work/2026-07-p06-dlpack-abi/spec.md similarity index 100% rename from work/2026-07-p06-dlpack-abi/spec.md rename to development/work/2026-07-p06-dlpack-abi/spec.md diff --git a/work/2026-07-p06-dlpack-abi/tasks.md b/development/work/2026-07-p06-dlpack-abi/tasks.md similarity index 100% rename from work/2026-07-p06-dlpack-abi/tasks.md rename to development/work/2026-07-p06-dlpack-abi/tasks.md diff --git a/work/2026-07-p07-export-tensor-empty/plan.md b/development/work/2026-07-p07-export-tensor-empty/plan.md similarity index 100% rename from work/2026-07-p07-export-tensor-empty/plan.md rename to development/work/2026-07-p07-export-tensor-empty/plan.md diff --git a/work/2026-07-p07-export-tensor-empty/spec.md b/development/work/2026-07-p07-export-tensor-empty/spec.md similarity index 100% rename from work/2026-07-p07-export-tensor-empty/spec.md rename to development/work/2026-07-p07-export-tensor-empty/spec.md diff --git a/work/2026-07-p07-export-tensor-empty/tasks.md b/development/work/2026-07-p07-export-tensor-empty/tasks.md similarity index 100% rename from work/2026-07-p07-export-tensor-empty/tasks.md rename to development/work/2026-07-p07-export-tensor-empty/tasks.md diff --git a/work/2026-07-p08-runtimes-cpu/plan.md b/development/work/2026-07-p08-runtimes-cpu/plan.md similarity index 100% rename from work/2026-07-p08-runtimes-cpu/plan.md rename to development/work/2026-07-p08-runtimes-cpu/plan.md diff --git a/work/2026-07-p08-runtimes-cpu/spec.md b/development/work/2026-07-p08-runtimes-cpu/spec.md similarity index 100% rename from work/2026-07-p08-runtimes-cpu/spec.md rename to development/work/2026-07-p08-runtimes-cpu/spec.md diff --git a/work/2026-07-p08-runtimes-cpu/tasks.md b/development/work/2026-07-p08-runtimes-cpu/tasks.md similarity index 100% rename from work/2026-07-p08-runtimes-cpu/tasks.md rename to development/work/2026-07-p08-runtimes-cpu/tasks.md diff --git a/work/2026-07-p09-cuda/plan.md b/development/work/2026-07-p09-cuda/plan.md similarity index 100% rename from work/2026-07-p09-cuda/plan.md rename to development/work/2026-07-p09-cuda/plan.md diff --git a/work/2026-07-p09-cuda/spec.md b/development/work/2026-07-p09-cuda/spec.md similarity index 100% rename from work/2026-07-p09-cuda/spec.md rename to development/work/2026-07-p09-cuda/spec.md diff --git a/work/2026-07-p09-cuda/tasks.md b/development/work/2026-07-p09-cuda/tasks.md similarity index 100% rename from work/2026-07-p09-cuda/tasks.md rename to development/work/2026-07-p09-cuda/tasks.md diff --git a/work/2026-07-p10-rocm/plan.md b/development/work/2026-07-p10-rocm/plan.md similarity index 100% rename from work/2026-07-p10-rocm/plan.md rename to development/work/2026-07-p10-rocm/plan.md diff --git a/work/2026-07-p10-rocm/spec.md b/development/work/2026-07-p10-rocm/spec.md similarity index 100% rename from work/2026-07-p10-rocm/spec.md rename to development/work/2026-07-p10-rocm/spec.md diff --git a/work/2026-07-p10-rocm/tasks.md b/development/work/2026-07-p10-rocm/tasks.md similarity index 100% rename from work/2026-07-p10-rocm/tasks.md rename to development/work/2026-07-p10-rocm/tasks.md diff --git a/work/2026-07-p11-integrations/plan.md b/development/work/2026-07-p11-integrations/plan.md similarity index 100% rename from work/2026-07-p11-integrations/plan.md rename to development/work/2026-07-p11-integrations/plan.md diff --git a/work/2026-07-p11-integrations/spec.md b/development/work/2026-07-p11-integrations/spec.md similarity index 100% rename from work/2026-07-p11-integrations/spec.md rename to development/work/2026-07-p11-integrations/spec.md diff --git a/work/2026-07-p11-integrations/tasks.md b/development/work/2026-07-p11-integrations/tasks.md similarity index 100% rename from work/2026-07-p11-integrations/tasks.md rename to development/work/2026-07-p11-integrations/tasks.md diff --git a/work/2026-07-p12-conformance-docs-release/plan.md b/development/work/2026-07-p12-conformance-docs-release/plan.md similarity index 100% rename from work/2026-07-p12-conformance-docs-release/plan.md rename to development/work/2026-07-p12-conformance-docs-release/plan.md diff --git a/work/2026-07-p12-conformance-docs-release/spec.md b/development/work/2026-07-p12-conformance-docs-release/spec.md similarity index 100% rename from work/2026-07-p12-conformance-docs-release/spec.md rename to development/work/2026-07-p12-conformance-docs-release/spec.md diff --git a/work/2026-07-p12-conformance-docs-release/tasks.md b/development/work/2026-07-p12-conformance-docs-release/tasks.md similarity index 100% rename from work/2026-07-p12-conformance-docs-release/tasks.md rename to development/work/2026-07-p12-conformance-docs-release/tasks.md From 91d9e4b94db2dcf4946d3545f6e8b99015c538b5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Enrique=20Gonz=C3=A1lez=20Paredes?= Date: Sat, 25 Jul 2026 22:49:23 +0200 Subject: [PATCH 02/10] chore: update the agent harness to copier template v0.6.0 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adopts the template's development/ tree: the agent-facing docs, ADRs and per-feature work units move out of docs/ and work/, leaving docs/ for user documentation (docs/api.md) as the new convention reserves it. The move is done as a pre-step so _skip_if_exists protects the project-authored docs; a bare `copier update` deletes them and substitutes the template scaffolds, because _skip_if_exists does not cover the delete side of a rename. Harness changes absorbed from v0.6.0: - report.md as a fifth work-unit artifact, owned by the Developer and audited by the Reviewer for honesty. - DECISION-PENDING: escalation marker plus the decision register in development/adr/README.md, seeded with the two decisions already granted (the ADR 0003 GPU-suite waiver and the coverage thresholds). - development/glossary.md, the document-liveness table, and the architecture > spec > plan > tasks authority order. - Role playbook skills (product-owner, architect, developer, reviewer) over a shared design-principles core. - PreToolUse hook fails closed when jq cannot parse the tool input. - .github/PULL_REQUEST_TEMPLATE.md with the definition-of-done checklist. AGENTS.md keeps devmm's stricter rule that a required runtime dependency needs an ADR, rather than the template's softer ADR-bar wording: the empty required-dependency set is a design invariant tests/test_packaging.py enforces. Gate: make verify green — 900 passed, 75 skipped (GPU suites, off hardware). Co-Authored-By: Claude Opus 5 (1M context) --- .agents/commands/build.md | 8 +- .agents/commands/plan.md | 18 ++- .agents/commands/spec.md | 31 ++++- .agents/commands/verify.md | 23 ++-- .agents/hooks/ensure-toolchain.sh | 4 +- .agents/skills/architect-playbook/SKILL.md | 80 ++++++++++++ .agents/skills/design-principles/SKILL.md | 119 ++++++++++++++++++ .agents/skills/developer-playbook/SKILL.md | 72 +++++++++++ .../skills/product-owner-playbook/SKILL.md | 81 ++++++++++++ .agents/skills/reviewer-playbook/SKILL.md | 71 +++++++++++ .agents/skills/verify/SKILL.md | 6 +- .agents/subagents/architect.md | 108 +++++++++++++--- .agents/subagents/developer.md | 39 +++++- .agents/subagents/explorer.md | 5 +- .agents/subagents/product-owner.md | 64 ++++++++-- .agents/subagents/reviewer.md | 90 +++++++++++-- .claude/rules/comments.md | 2 +- .claude/settings.json | 4 +- .copier-answers.yml | 3 +- .github/PULL_REQUEST_TEMPLATE.md | 23 ++++ .github/workflows/ci.yml | 2 +- .gitignore | 3 +- .opencode/opencode.jsonc | 6 +- AGENTS.md | 67 +++++++--- CHANGELOG.md | 15 ++- CLAUDE.md | 3 +- Makefile | 16 +-- README.md | 6 +- development/README.md | 44 +++++++ .../adr/0001-record-architecture-decisions.md | 2 +- .../adr/0002-task-runner-make-over-just.md | 4 +- .../adr/0003-gpu-suite-waiver-for-0.1.0.md | 4 +- development/adr/README.md | 51 +++++++- development/architecture.md | 2 +- development/glossary.md | 28 +++++ development/harness-usage.md | 86 ++++++++++--- development/style.md | 2 +- development/testing.md | 14 +++ development/work/2026-07-p00-scaffold/plan.md | 2 +- development/work/2026-07-p00-scaffold/spec.md | 4 +- .../work/2026-07-p01-dtypes-device/plan.md | 2 +- .../work/2026-07-p01-dtypes-device/spec.md | 4 +- development/work/2026-07-p02-layout/plan.md | 2 +- development/work/2026-07-p02-layout/spec.md | 4 +- .../plan.md | 2 +- .../spec.md | 4 +- development/work/2026-07-p04-cpu-mrs/plan.md | 2 +- development/work/2026-07-p04-cpu-mrs/spec.md | 4 +- .../work/2026-07-p05-buffer-registry/plan.md | 2 +- .../work/2026-07-p05-buffer-registry/spec.md | 4 +- .../work/2026-07-p06-dlpack-abi/plan.md | 2 +- .../work/2026-07-p06-dlpack-abi/spec.md | 4 +- .../2026-07-p07-export-tensor-empty/plan.md | 2 +- .../2026-07-p07-export-tensor-empty/spec.md | 4 +- .../work/2026-07-p08-runtimes-cpu/plan.md | 2 +- .../work/2026-07-p08-runtimes-cpu/spec.md | 4 +- development/work/2026-07-p09-cuda/plan.md | 2 +- development/work/2026-07-p09-cuda/spec.md | 4 +- development/work/2026-07-p10-rocm/plan.md | 2 +- development/work/2026-07-p10-rocm/spec.md | 4 +- .../work/2026-07-p11-integrations/plan.md | 2 +- .../work/2026-07-p11-integrations/spec.md | 4 +- .../plan.md | 2 +- .../spec.md | 4 +- .../tasks.md | 2 +- pyproject.toml | 2 +- src/devmm/__init__.py | 2 +- src/devmm/_runtimes/_hostcopy.py | 2 +- tests/test_import_hygiene.py | 2 +- tests/test_public_api.py | 2 +- tests/traceability.md | 14 +-- 71 files changed, 1114 insertions(+), 196 deletions(-) create mode 100644 .agents/skills/architect-playbook/SKILL.md create mode 100644 .agents/skills/design-principles/SKILL.md create mode 100644 .agents/skills/developer-playbook/SKILL.md create mode 100644 .agents/skills/product-owner-playbook/SKILL.md create mode 100644 .agents/skills/reviewer-playbook/SKILL.md create mode 100644 .github/PULL_REQUEST_TEMPLATE.md create mode 100644 development/README.md create mode 100644 development/glossary.md diff --git a/.agents/commands/build.md b/.agents/commands/build.md index 6df43bf..21ac4bc 100644 --- a/.agents/commands/build.md +++ b/.agents/commands/build.md @@ -1,14 +1,14 @@ --- description: Implement the current feature one phase at a time, ticking tasks.md and running the verification gate at each phase boundary -argument-hint: (optional; defaults to the most recent work/* directory) +argument-hint: (optional; defaults to the most recent development/work/* directory) --- You are carrying out the implementation phase of a feature. 1. Identify the target spec directory. - - If `$ARGUMENTS` is provided, use `work/$ARGUMENTS/`. + - If `$ARGUMENTS` is provided, use `development/work/$ARGUMENTS/`. - Otherwise, use the most recently modified directory under - `work/`. + `development/work/`. 2. Read `spec.md`, `plan.md`, and `tasks.md` in full. If `plan.md` is missing or empty, stop and tell the user to run `/plan` first. 3. If the plan touches an unfamiliar area of the codebase, run an @@ -25,6 +25,8 @@ You are carrying out the implementation phase of a feature. - Work one phase at a time, writing tests first where the plan calls for behaviour change. - Tick `tasks.md` checkboxes in the same commit as the code change. + - Keep `report.md` current: deviations, abandoned approaches, and + `DECISION-PENDING:` escalations are recorded when they happen. - Run `make verify` at every phase boundary. - Stop at the end of each phase and hand off to `/verify` (Reviewer) before starting the next. diff --git a/.agents/commands/plan.md b/.agents/commands/plan.md index abe115b..b5b2c97 100644 --- a/.agents/commands/plan.md +++ b/.agents/commands/plan.md @@ -1,20 +1,32 @@ --- description: Expand spec.md into a numbered, testable plan.md for the current feature -argument-hint: (optional; defaults to the most recent work/* directory) +argument-hint: (optional; defaults to the most recent development/work/* directory) --- You are expanding a feature spec into an implementation plan. 1. Identify the target spec directory. - - If `$ARGUMENTS` is provided, use `work/$ARGUMENTS/`. + - If `$ARGUMENTS` is provided, use `development/work/$ARGUMENTS/`. - Otherwise, use the most recently modified directory under - `work/`. + `development/work/`. 2. Confirm `spec.md` exists and has been reviewed. If it's missing or empty, stop and tell the user to run `/spec` first. 3. Delegate the planning to the **architect** subagent (`.agents/subagents/architect.md`). It owns the phased-plan format, the architecture-decisions block, the "each phase has tests" contract, and the "stop and ask before coding" boundary. +4. If the architect hands back a **spike request** — a + `SPIKE-REQUEST:` line in `scratch.md` naming one question (it cannot + run code itself) — run the smallest throwaway experiment that answers + it, in a temp dir or scratch space, never left in the source tree. + Append the result to `scratch.md` on a `SPIKE-FINDING:` line + (question → method → answer → evidence), then re-invoke the + architect; it starts with fresh context and reads `scratch.md` to + pick the answer up. Spike code is disposable; only the findings + survive, in the plan's **Spike findings** section. + Cap this at **three rounds per plan**. A fourth request means the + uncertainty is not a design experiment — stop and put the question + to the user. The architect subagent will write `plan.md` and mirror it into `tasks.md`, then stop for user review. Once the user confirms the diff --git a/.agents/commands/spec.md b/.agents/commands/spec.md index 64391a0..ac2ee53 100644 --- a/.agents/commands/spec.md +++ b/.agents/commands/spec.md @@ -1,5 +1,5 @@ --- -description: Create a new feature spec directory under work/-/ +description: Create a new feature spec directory under development/work/-/ argument-hint: --- @@ -8,17 +8,36 @@ You are creating a new feature spec. 1. If `$ARGUMENTS` is empty, ask the user for a slug before creating anything. 2. Compute today's date as `YYYY-MM`. The directory is - `work/-$ARGUMENTS/`. + `development/work/-$ARGUMENTS/`. 3. Create the directory if it doesn't exist. **Do not** pre-create - `plan.md`, `tasks.md`, or `scratch.md` — each role creates the - artifacts it owns (Architect: `plan.md`/`tasks.md`; Developer: - `scratch.md`), and pre-creating them would force those roles to - `edit` files they should be able to `write` from scratch. + `plan.md`, `tasks.md`, `report.md`, or `scratch.md` — each role + creates the deliverables it owns (Architect: `plan.md`/`tasks.md`; + Developer: `report.md`), and pre-creating them would force those + roles to `edit` files they should be able to `write` from scratch. + `scratch.md` is not a deliverable but the feature's shared channel: + whoever needs it first creates it, and everyone after that appends. 4. Delegate the actual spec authoring to the **product-owner** subagent (`.agents/subagents/product-owner.md`). It owns the spec format, the testable-criteria rule, the non-goals requirement, and the "stop and ask before planning" boundary. +5. If the product-owner hands back a **clarifying question** instead of + `spec.md`, put it to the user, then re-invoke the product-owner + subagent with the question and the user's answer included in the + prompt — it starts each invocation with fresh context and cannot see + the exchange otherwise. Repeat until it produces the spec. Cap this + at **five rounds**: if the questions have not converged by then, stop + and ask the user to settle the scope directly. The product-owner subagent will write `spec.md` and then stop for user review. Once the user confirms the spec, the next step is `/plan` (Architect role). Do not start implementing yet. + +If the reviewed spec's **Glossary** section pins down new domain terms, +promote them to `development/glossary.md` as part of the review wrap-up +(the product-owner subagent cannot write outside the feature +directory). The glossary is a **register**: promotion from a reviewed +spec is its one sanctioned mid-feature channel (see Document liveness +in `development/harness-usage.md`), and the Reviewer checks each +promoted entry against the spec's Glossary section — so promote the +reviewed terms verbatim, and don't fold in renames or meaning changes +of existing entries (those are trunk-gated, a dedicated PR). diff --git a/.agents/commands/verify.md b/.agents/commands/verify.md index 03ccf9d..ac5deed 100644 --- a/.agents/commands/verify.md +++ b/.agents/commands/verify.md @@ -7,18 +7,25 @@ You are running the review phase for the current feature. Delegate the work to the **reviewer** subagent (`.agents/subagents/reviewer.md`). It will: -- Read `spec.md`, `plan.md`, `tasks.md`, and the current diff. -- Run `make verify`. -- Check spec conformance, plan conformance, and implementation - quality. +- Read `spec.md`, `plan.md` (including its **Review checklist** + section, if present), `tasks.md`, `report.md` (if the Developer has + started it — expected by the final phase), and the current diff. +- Run `make verify` itself — it never takes the Developer's + word for the gate. +- Check spec conformance, plan conformance, implementation quality, + and report honesty (undeclared deviations are defects). - Produce a `GO` or `NEEDS-WORK` verdict with a citation-rich defect - list (`path/to/file.ext:LINE`) when work remains. + list (`path/to/file.ext:LINE`), ranked MAJOR / MINOR / INFO. Any + MAJOR means NEEDS-WORK. Hand the reviewer's verdict back to the user verbatim. If `NEEDS-WORK`, the next step is `/build` (Developer role) to address -the defects. If `GO`, summarise what changed since the last verify -(use `git diff --stat` and `git log -1`) and stop. +the defects. If `GO` on the feature's final phase, confirm `report.md` +is complete (what was built, deviations, negative results, follow-ups, +gate result) before the feature is declared mergeable — it freezes at +merge. Then summarise what changed since the last verify (use +`git diff --stat` and `git log -1`) and stop. Never silently skip, disable, or `@ignore` a failing test. If a test -must be skipped, draft an ADR under `docs/adr/` and ask for +must be skipped, draft an ADR under `development/adr/` and ask for confirmation. diff --git a/.agents/hooks/ensure-toolchain.sh b/.agents/hooks/ensure-toolchain.sh index 6b90c4a..743bf86 100755 --- a/.agents/hooks/ensure-toolchain.sh +++ b/.agents/hooks/ensure-toolchain.sh @@ -9,7 +9,7 @@ # - Any other agent / CI / devcontainer may call it too. # # The install command and URL come from the toolchain_install_* macros in -# _macros.jinja (the single source; docs/tool-bootstrap.md renders the same +# _macros.jinja (the single source; development/tool-bootstrap.md renders the same # macro for its manual fallback). The installer appends $HOME/.local/bin to the # shell profile, so the binary is on PATH for *new* shells (e.g. an agent's # subsequent tool calls), not the current one. A caller that needs it within its @@ -33,5 +33,5 @@ if curl -LsSf https://astral.sh/uv/install.sh | sh >&2 && [ -x "$HOME/.local/bin fi echo "ensure-toolchain: ERROR — could not install uv automatically (offline or restricted network?)." >&2 -echo "ensure-toolchain: install it manually, then retry. See docs/tool-bootstrap.md." >&2 +echo "ensure-toolchain: install it manually, then retry. See development/tool-bootstrap.md." >&2 exit 1 diff --git a/.agents/skills/architect-playbook/SKILL.md b/.agents/skills/architect-playbook/SKILL.md new file mode 100644 index 0000000..18cfb13 --- /dev/null +++ b/.agents/skills/architect-playbook/SKILL.md @@ -0,0 +1,80 @@ +--- +name: architect-playbook +description: | + How the Architect turns an approved spec into a plan. Use when writing + or revising plan.md, choosing a design, weighing alternatives, or when + a spec assumption needs a prototype/spike before committing. Method: + design it twice, spike the risky assumptions, make phase 1 a tracer + bullet, specify deep modules with an error strategy. Companion to the + architect subagent and the /plan command. +--- + +# Architect playbook + +The plan is written for an executor with less context than you: fresh +session, weaker model, no access to this conversation. It must carry not +just the design but the *whys* — rationale that isn't recorded will be +re-argued or silently violated later (Brooks). + +Shared ground rules: `.agents/skills/design-principles/SKILL.md`. + +## Method + +1. **Study exemplars first.** Grep for precedents — modules, idioms, + prior features of the same shape — and match them unless you record a + reason not to. Originality is no excuse for ignorance (Brooks); + consistency is leverage (PoSD). +2. **Design it twice.** Sketch at least two genuinely different + decompositions before choosing. Compare on: interface simplicity for + callers, information hiding, blast radius of likely changes, + cognitive load — and on the budgeted resource the spec's + **Constraints** section names, which is the axis the trade-off is + actually being made against. Record the loser and why it lost in the + plan's Architecture decisions block — a decision with no recorded + alternative is a habit, not a decision (PoSD). +3. **Spike before you commit.** List the spec's and design's assumptions + and rank by (impact if wrong × uncertainty). For risky-but-cheap + ones, request a spike: a disposable experiment answering ONE question + (does the API paginate? is the parser fast enough?). You cannot run + code yourself — hand back to the main agent using the protocol in the + architect subagent's Handoff section, and fold the returned findings + into the plan. Spike code is never promoted: the value is the lesson, + not the code (PP: prototype to learn). +4. **Phase 1 is a tracer bullet.** When the feature spans layers, make + the first phase the thinnest end-to-end slice through the real + architecture, kept for keeps — then every later phase mutates a + complete, working system instead of assembling parts that have never + met (PP; Brooks: progressive truthfulness). +5. **Specify deep modules.** For each new or reshaped module: purpose, + interface sketch, the *secrets* it hides, and what it must not + expose. Minimal implementation, slightly general interface. Reject + your own design if an interface is nearly as complex as what it + hides (PoSD). +6. **Design the error strategy, don't inherit it.** Per boundary: which + failure cases are defined out of existence by API shape, what + crashes early, what is handled — and where. "Wrap it in try/catch" + is not a strategy (PoSD; PP). +7. **Name in the ubiquitous language.** Take names from + `development/glossary.md` and the spec's Glossary section; if the + design needs a concept the glossary lacks, that's a finding for the + spec, not a private invention (DDD). +8. **Tests are part of the design.** Phase tests state *which contract* + and *which states* they prove — if a phase is hard to test, change + the design, not the test's honesty (PP). +9. **Flag the irreversible.** Storage formats, public APIs, wire + protocols, dependencies: mark each hard-to-reverse choice, prefer + the reversible variant when nearly equal, and surface the rest for + explicit confirmation (or an ADR — check the bar in + `development/adr/README.md`). + +## Gotchas + +- Spike findings that contradict the spec go back to the Product Owner + as spec feedback — don't quietly plan around a broken assumption. +- Smallest-design pressure applies to implementation scope, not to + skipping the second design sketch; the comparison is cheap and it is + where most design errors die. +- Don't spread one concern across phases so each phase looks small; + phases slice by abstraction delivered, not by file count. +- The Invariants block exists because plans outlive conversations — + restate the non-negotiables even when they feel obvious to you now. diff --git a/.agents/skills/design-principles/SKILL.md b/.agents/skills/design-principles/SKILL.md new file mode 100644 index 0000000..2b11055 --- /dev/null +++ b/.agents/skills/design-principles/SKILL.md @@ -0,0 +1,119 @@ +--- +name: design-principles +description: | + The project's software-design ground rules and red-flag checklist. Use + whenever designing, planning, implementing, refactoring, or reviewing + code — any time you choose module boundaries, name things, handle + errors, or judge whether a change is "good enough". The role playbooks + (product-owner, architect, developer, reviewer) build on this file; + it is the shared core they all cite. +--- + +# Design principles + +Distilled from four sources: Ousterhout, *A Philosophy of Software Design* +(PoSD); Hunt & Thomas, *The Pragmatic Programmer* (PP); Evans, +*Domain-Driven Design* (DDD); Brooks, *The Design of Design* (DoD). +Tags in parentheses give provenance, not further reading assignments. + +## Prime directives + +1. **Complexity is the enemy, and it is incremental.** No "too small to + matter"; fix broken windows when you touch them (PoSD; PP). +2. **Working code isn't enough.** Invest continually in design, don't + just take the fastest diff that passes (PoSD). +3. **The domain is the heart of the software.** Model it in the domain's + own language and keep model and code bound together (DDD). +4. **Design is iterative discovery.** Goals, requirements, and + constraints are found while designing, not given up front (DoD). +5. **DRY.** Every piece of knowledge — code, schema, doc, config — has + one authoritative representation (PP). +6. **Design for reading, not writing.** The future reader outranks the + present writer (PoSD). +7. **You can't write perfect software — be paranoid.** Contracts, + assertions, crash early: a dead program does less damage than a + crippled one (PP). +8. **There are no final decisions.** Keep choices reversible; + abstractions outlive details (PP). + +## Modules and interfaces + +- Make modules **deep**: simple interface over powerful implementation. + The interface is the cost; a simple interface matters more than a + simple implementation (PoSD). +- **Hide information.** A design decision reflected in more than one + module is a leak. Decompose by knowledge, never by execution order + (PoSD). +- Implement what's needed now, but shape the **interface** for the class + of needs — somewhat general-purpose, not special-cased to today's + caller (PoSD). +- **Different layer, different abstraction.** Pass-through methods and + thin wrappers signal a wrong decomposition (PoSD). +- **Pull complexity downward.** Better the module author suffers than + every caller; don't export your problem as a config parameter (PoSD). +- **Eliminate effects between unrelated things.** Shy code, Law of + Demeter; one requirement change should land in one module (PP). +- **Define errors out of existence.** Redesign the API so the + exceptional case is normal semantics where possible; handle the rest + in as few places as possible; exceptions only for the exceptional + (PoSD; PP). +- **Policy in metadata, mechanism in code**; views separate from models + (PP). + +## Domain and naming + +- Name every code element with the exact terms the domain experts use + (see `development/glossary.md`); a vocabulary change is a rename, in + the same change (DDD). +- Business rules are **named concepts**, not buried conditionals; when + experts use a word the code doesn't have, reify it (DDD). +- Domain objects carry their behaviour — don't drain logic into service + code and leave anemic data bags (DDD). +- Precise, consistent names; if a precise name is hard to find, the + design — not the vocabulary — is unclear (PoSD). + +## Construction + +- **Comments say what the code can't**: why, invariants, units, + ownership. Interface comments never describe implementation. A comment + that repeats the code is noise (PoSD). +- **Design by contract**: preconditions, postconditions, invariants as + assertions that stay on; crash early rather than limp; the allocator + of a resource frees it (PP). +- **Never program by coincidence.** Rely only on documented behaviour; + prove assumptions with real data; understand generated code before + committing it (PP). +- **Test the contract, cover the states** — not the lines. Every bug + found gets a regression test before it gets a fix (PP). +- **Refactor on its own commits**, never mixed with behaviour change + (PP; PoSD). + +## Red-flag checklist + +Stop and reconsider when you see one of these: + +- Shallow module — interface barely simpler than its implementation. +- Information leak — one decision visible in several modules. +- Pass-through method — same signature relayed one level down. +- Temporal decomposition — structure mirrors execution order. +- Nontrivial repetition — knowledge represented twice. +- Special-purpose code tangled with general-purpose code. +- Vague or hard-to-pick name; entity that's hard to describe briefly. +- Comment that repeats the code; implementation detail in an interface + comment. +- Train wreck — `a.getB().getC().doD()`. +- Coincidental correctness — it works but nobody can say why. +- Anemic domain object; business rule living only in a conditional; + code vocabulary drifting from the experts' speech. +- Manual procedure a script should own. +- Broken window left standing. + +## Gotchas + +- "Smallest design that satisfies the spec" (the Architect's rule) and + "somewhat general-purpose" do not conflict: minimal *implementation*, + slightly general *interface*. +- Zero tolerance for red flags means *name and rank them*, not + "everything is a blocker" — severity still matters. +- These rules bind the *new* code you write; bringing an entire legacy + file up to standard is its own task, not a drive-by. diff --git a/.agents/skills/developer-playbook/SKILL.md b/.agents/skills/developer-playbook/SKILL.md new file mode 100644 index 0000000..810bf45 --- /dev/null +++ b/.agents/skills/developer-playbook/SKILL.md @@ -0,0 +1,72 @@ +--- +name: developer-playbook +description: | + How the Developer implements a plan phase. Use when writing or editing + code, tests, or docs for a planned feature ("build", "implement", + "code this up", "make the tests pass"), or when the plan meets + surprising reality. Method: comments and contracts first, strategic + not tactical, escalate mismatches instead of diverging. Companion to + the developer subagent and the /build command. +--- + +# Developer playbook + +Working code isn't enough (PoSD): the deliverable is code a stranger can +read, change, and trust — plus the tests and docs that prove it. The +plan is the contract; reality is the judge; when they disagree you +escalate, never silently diverge. + +Shared ground rules: `.agents/skills/design-principles/SKILL.md`. + +## Method + +1. **Write in the codebase's voice.** Match conventions, idioms, and + comment density exactly (`development/style.md` and neighbouring + code). Names come from `development/glossary.md` and the spec — + naming drift is a defect, not a preference (DDD; PoSD). +2. **Interface comment before body.** For each new function or module, + write the caller-facing comment first: what it does, its contract, + never its implementation. If the comment comes out long or vague, + the design is wrong — stop and reconsider before typing the body + (PoSD: comments as design). +3. **Contracts on, crash early.** Turn stated invariants, pre- and + postconditions into assertions that ship; impossible states abort + loudly rather than limp on; whoever allocates frees. Implement the + plan's error strategy exactly — no ad-hoc catch-and-log (PP). +4. **Test first against the contract.** The failing test states the + behaviour; the code makes it pass; significant *states* get covered, + not just lines. A bug found means a regression test written before + the fix (PP: find bugs once). +5. **DRY across artifacts.** Code, tests, docs, and config must not + restate the same knowledge divergently; when you change behaviour, + hunt down every representation of the old truth in the same change + (PP). +6. **Never program by coincidence.** Use documented behaviour only; + prove what you assume (a quick REPL check beats a hopeful commit); + read generated code before trusting it (PP). +7. **Leave it better — separately.** Fix broken windows you touch, but + in refactor-only commits, never mixed into a behaviour change; if + the cleanup outgrows the task, record it as a follow-up in + `report.md` instead of a drive-by rewrite (PP; PoSD). +8. **Escalate plan/reality mismatches.** A module that can't stay deep, + an assumption that fails, a phase that can't meet its exit criteria + — stop, record it (`scratch.md` note for the Architect, or a + `DECISION-PENDING:` line in `report.md` when it exceeds your + authority), and hand back. Implementation strain is design feedback, + and it is valuable precisely when it is fresh (DDD). +9. **Self-review before handoff.** Walk the red-flag checklist in + `.agents/skills/design-principles/SKILL.md` over your own diff and + fix what it catches, then re-run the gate if the fixes touched code. + The Reviewer is for what you *can't* see, not for what you didn't + look at. + +## Gotchas + +- "Read just enough context" means enough to make the change *safely* — + the callers, the tests, the invariants — not just enough to make it + compile. +- Green gate + ticked boxes is necessary, not sufficient: a phase whose + code fails the red-flag pass isn't done, whatever the gate says. +- Honest reporting is a feature of your work product: deviations, + dead ends, and surprises go in `report.md` when they happen, not + reconstructed at the end. diff --git a/.agents/skills/product-owner-playbook/SKILL.md b/.agents/skills/product-owner-playbook/SKILL.md new file mode 100644 index 0000000..46c068f --- /dev/null +++ b/.agents/skills/product-owner-playbook/SKILL.md @@ -0,0 +1,81 @@ +--- +name: product-owner-playbook +description: | + How the Product Owner discovers what to build. Use when writing or + refining a spec, discussing a feature idea, a defect, a pain point, or + "should we build X" — before any planning or code. Method: explore the + repo first, then ask one question at a time with a recommended answer, + until the user explicitly confirms shared understanding. Companion to + the product-owner subagent and the /spec command. +--- + +# Product Owner playbook + +The hardest part of design is deciding *what* to design; a chief service +of the designer is helping the client discover what they actually want +(Brooks). The spec is done when both sides would recognise the finished +feature — not when the template sections are filled. + +Shared ground rules: `.agents/skills/design-principles/SKILL.md`. + +## Method + +1. **Look it up before asking.** Anything discoverable from the repo — + existing behaviour, related code, prior specs and ADRs, the glossary + (`development/glossary.md`) — you find with Read/Grep/Glob and state + as findings. Ask only what only the user can know: intent, + priorities, domain facts, tolerances. +2. **Dig for the need, not the feature.** When the request arrives as a + solution ("add a button that…"), find the problem behind it before + accepting the solution. Requirements are needs; policy and UI are + details that change (PP). +3. **One question per invocation, with a recommendation.** Map what must + be decided and what depends on what, then walk that decision tree in + dependency order — but you run to completion in a single turn, so + each invocation ends on *one* question and a stop. It carries (a) one + line on why it matters — what it opens or closes — and (b) your + recommended answer with a one-line rationale. Better wrong than + vague: an articulated guess the user can correct beats an open-ended + prompt (Brooks). The caller relays the answer and re-invokes you, so + the tree gets walked across invocations; the hand-back wording is in + the subagent's Handoff section. +4. **Test understanding with concrete scenarios.** Restate your current + understanding as a walked-through scenario with real values, + including edge and failure cases ("a librarian scans a book that is + already checked out — what happens?"). Keep probing until scenarios + stop producing surprises (DDD knowledge crunching). +5. **Advocate for the product.** Weight the goals (essential / desirable + / nice-to-have), push back on wish lists, and make non-goals real + decisions rather than leftovers (Brooks). +6. **Make constraints and the user model explicit.** Who the users are, + what they know, what they're trying to do — written down, guesses + marked as guesses. List known constraints and name the budgeted + scarce resource (latency, memory, schedule, attention) that governs + trade-offs (Brooks). +7. **Grow the shared vocabulary.** Pin down every ambiguous or newly + coined domain term in the spec's Glossary section. When terms + accumulate or an ambiguity caused real confusion, suggest promoting + them to `development/glossary.md` — the project's ubiquitous + language (DDD). +8. **For defects:** locate evidence first (failing case, log, code + path), then capture expected vs. actual as a scenario pair. Fix the + problem, not the blame (PP). +9. **Stop at understanding.** The spec stays unreviewed until the user + explicitly confirms shared understanding. Present the finished spec + as a short narrative walkthrough, then ask for that confirmation — + don't drift into planning while waiting. + +## Gotchas + +- Do not batch questions to seem efficient. Three questions at once get + one answered well and two answered badly. +- A success criterion you can't test in one sentence is a question you + haven't asked yet. +- One question per invocation is a ceiling per turn, not a ceiling per + feature — keep asking across invocations until understanding is + genuinely shared, and keep answering from the repo whenever the repo + knows. Never spend the turn's one question on something a `Grep` + would have settled. +- Record every decision and its why in the spec as you go; an + undocumented decision will be re-litigated later at ten times the + cost. diff --git a/.agents/skills/reviewer-playbook/SKILL.md b/.agents/skills/reviewer-playbook/SKILL.md new file mode 100644 index 0000000..c49f39f --- /dev/null +++ b/.agents/skills/reviewer-playbook/SKILL.md @@ -0,0 +1,71 @@ +--- +name: reviewer-playbook +description: | + How the Reviewer judges a phase. Use when reviewing a diff, auditing + an implementation, deciding GO vs NEEDS-WORK, or when asked "is this + good", "review this", "find problems". Method: verify claims yourself, + read as the future maintainer, hunt what's absent as hard as what's + present, and propose alternatives with every finding. Companion to the + reviewer subagent and the /verify command. +--- + +# Reviewer playbook + +Complexity is incremental (PoSD): a hundred locally fine changes can +still rot a system, so the review judges the design trajectory, not just +the diff. Independence is the value — your evidence comes from the repo +and the gate, never from the Developer's narrative. + +Shared ground rules: `.agents/skills/design-principles/SKILL.md`. + +## Method + +1. **Read as the future maintainer first.** Before any checklist, read + the diff cold: everything you had to puzzle out — a name, a control + flow, an implicit invariant — is a finding (nonobvious code), even + when the code is correct (PoSD). +2. **Assume competence, then check.** When something looks wrong, ask + "what led them to do this?" — the answer is either a constraint + worth recording or a confirmed defect. Both are findings (Brooks). +3. **Run the red-flag checklist** from + `.agents/skills/design-principles/SKILL.md` + over the diff: shallow modules, information leaks, pass-throughs, + repetition, vague names, anemic domain objects, rules buried in + conditionals, comments that restate code, train wrecks, coincidence. +4. **Probe the change amplification.** Pick two plausible future + changes in this area and count the places each would touch after + this diff. A rising count is a design defect with all tests green + (PoSD). +5. **Hunt what's absent.** Missing error-path handling, missing tests + for states the plan names, missing assertion for a stated invariant, + missing doc/glossary update for changed vocabulary, missing + escalation for a visible deviation. Absence is where defects hide + from diff-reading. +6. **Interrogate the tests, not just the code.** Do they test the + contract or mirror the implementation? Would they catch three + plausible bugs you can name (saboteur thinking)? Is state space + covered or only the happy line? A vacuous test is worse than a + missing one (PP). +7. **Check the language.** New names against `development/glossary.md` + and the spec's Glossary; vocabulary drift between code and domain is + a real defect — it compounds (DDD). +8. **Every finding proposes a way out.** Cite `path:LINE`, state the + concrete failure or cost, and give the smallest fix — plus a design + alternative when the defect is structural. Options, not lame + objections (PP). Discard what you can't substantiate; "might be an + issue" is not a finding. +9. **Close with the trajectory verdict.** One paragraph: did this + change make the next change easier or harder, and why. This is the + sentence maintainers will thank you for. + +## Gotchas + +- Findings still rank MAJOR / MINOR / INFO and the gate contract stays + as `.agents/subagents/reviewer.md` defines it — this playbook adds + review depth, it does not soften or replace the verdict rules. +- Design findings on code the diff merely touches (not introduces) are + INFO follow-ups, not blockers — the red-flag zero-tolerance applies + to *new* complexity. +- Don't pad the verdict with praise; the absence of findings is the + praise. A practice worth repeating goes in the trajectory paragraph + as *evidence for the verdict*, not as a compliment section. diff --git a/.agents/skills/verify/SKILL.md b/.agents/skills/verify/SKILL.md index d767f7f..6bda08b 100644 --- a/.agents/skills/verify/SKILL.md +++ b/.agents/skills/verify/SKILL.md @@ -30,9 +30,9 @@ description: | ## Gotchas -- The `Stop` hook in `.claude/settings.json` already runs `make verify`. - This skill is for the *interactive* case where the user wants verification - before the agent's natural stop. +- The `Stop` hook in `.claude/settings.json` already runs `make verify` + (Claude Code only). This skill is for the *interactive* case where the user + wants verification before the agent's natural stop. - `verify.sh` exits with code 2 on failure (Stop-hook convention). Don't treat exit 2 as a different signal from exit 1 — both mean "fix it". - On slow machines `make verify` may exceed 60s. If that becomes diff --git a/.agents/subagents/architect.md b/.agents/subagents/architect.md index 8b98f19..bbb2004 100644 --- a/.agents/subagents/architect.md +++ b/.agents/subagents/architect.md @@ -5,11 +5,11 @@ description: | turn it into a phased, testable plan.md with explicit technical decisions and delivery steps. Invoked by the /plan slash command. Stops before any code is written. -tools: Read, Grep, Glob, Write +tools: Read, Grep, Glob, Write, Edit permission: read: allow write: allow - edit: deny + edit: allow bash: deny mode: subagent model: inherit @@ -19,30 +19,65 @@ You are the **Architect**. Your job is to translate an approved `spec.md` into an implementation plan that the Developer can execute phase-by-phase, with tests as the contract for each phase. +Your method lives in `.agents/skills/architect-playbook/SKILL.md` +(which links the shared `.agents/skills/design-principles/SKILL.md`). +Read it before starting; it is part of your instructions. + ## Goal Produce `plan.md` and mirror it into a checkbox `tasks.md` in the same -`work/-/` directory. +`development/work/-/` directory. ## Constraints - Read `spec.md` in full first. If a success criterion is unclear or untestable, stop and ask before planning. +- Read `scratch.md` too if it exists. It is the shared working memory + for this feature: prior spike findings, `explorer` summaries, and + Developer hand-back notes all land there, and after a spike round it + holds the answer to the question *you* asked. Skipping it means + re-asking a question that has already been answered. +- Treat the spec's **Constraints** section as binding, not background. + The budgeted resource named there (latency, memory, schedule, + attention) is the axis your alternatives are compared on, and each + phase must stay inside it. If the smallest design that satisfies the + spec cannot meet the budget, that is a finding for the Product Owner — + say so in **Risks & open questions** and stop; never quietly relax the + number. - Surface every non-trivial technical decision (dependency, persistence, - protocol, framework, auth) and either resolve it inline or flag it - as needing an ADR under `docs/adr/`. + protocol, framework, auth) and either resolve it inline or flag it as needing + an ADR under `development/adr/` when it meets the criteria there + (`development/adr/README.md`). - Each phase must be small enough to verify independently (≤1 day of work) and must list the test(s) that prove it works. +- Write the plan for an agent with **less context than you have now**: + it may be executed in a fresh session, by a weaker model, or after + compaction. Restate the non-negotiable invariants inline (see the + Invariants block below) instead of assuming ambient knowledge, and + make every step executable without reading this conversation. - Prefer the smallest design that satisfies the spec. No speculative - abstractions. No features the spec does not require. + abstractions. No features the spec does not require. (Smallest + *implementation* — module interfaces may still be shaped for the + class of needs, not special-cased to today's caller.) - Reuse existing code and patterns where possible — use Grep/Glob to find them before proposing new modules. -- Never edit code. Write only `plan.md` and `tasks.md` under - `work/-/`. If the design needs an ADR, surface it - in `plan.md`'s **Architecture decisions** block (with a one-line - rationale and an "ADR needed: " marker); the human or the - Developer authors the ADR file under `docs/adr/` as a separate - step. Do not create files outside the spec directory. +- Your *method* — design it twice, spike the risky assumptions, make + Phase 1 a tracer bullet, specify deep modules, design the error + strategy — lives in the playbook and is deliberately not restated + here. Its outputs are contractual: every decision records the + alternative it beat, and every spike records question → method → + answer → evidence. +- Never edit code. Inside `development/work/-/` you own + `plan.md` and `tasks.md`, and you may add a spike request to the + shared `scratch.md` (see Handoff). Never write outside that directory. + If the design needs an ADR, surface it in `plan.md`'s **Architecture + decisions** block (with a one-line rationale and an "ADR needed: + " marker); the human or the Developer authors the ADR file + under `development/adr/` as a separate step. +- `scratch.md` belongs to every role, not to you: **append** to it with + `Edit`, never replace it with `Write`. Overwriting it destroys spike + findings and Developer hand-back notes, and it is gitignored, so what + you clobber is gone. ## Output format @@ -53,9 +88,14 @@ Produce `plan.md` and mirror it into a checkbox `tasks.md` in the same ## Architecture decisions - : — . + Considered: — . ADR: " if a new one is required before code lands, or "n/a">. +## Spike findings +- → . Method: . Evidence: . (Omit the section if no spikes were needed.) + ## Phase 1 — **Scope.** **Steps.** @@ -69,13 +109,49 @@ Produce `plan.md` and mirror it into a checkbox `tasks.md` in the same ## Risks & open questions - : + +## Invariants +- + spec.md > plan.md > tasks.md; decisions beyond your authority are + escalated in report.md using the exact marker token defined in + development/adr/README.md ("DECISION-PENDING" immediately followed by a + colon), never resolved locally. Do not write the live marker itself + here — plan text is not a report, and a stray marker would trip the + reviewer's register check.> + +## Review checklist +- ``` `tasks.md` mirrors the steps as `- [ ]` checkboxes, grouped by phase. ## Handoff -When `plan.md` and `tasks.md` are written, **stop**. Reply with a -1-line summary per phase and the list of architecture decisions. Ask -the user to review. Once confirmed, the next step is `/build` -(Developer role). Do not invoke the Developer yourself. +You have two ways to stop, and only two. + +**Spike needed.** You cannot run code, so when the plan would otherwise +commit to an assumption you can't check, do not guess. Append to +`scratch.md` a line of the form + +``` +SPIKE-REQUEST: +``` + +then **stop** and reply with that question plus this instruction, in +your own words: *run the smallest throwaway experiment that answers it, +append the result to `scratch.md` as `SPIKE-FINDING: → +. Method: . Evidence: `, then re-invoke +the architect subagent.* State it every time. Do not assume the caller +loaded `/plan` — the role is also reached by description match, and then +the slash command's instructions were never read. If a finding +contradicts `spec.md`, hand back to the Product Owner rather than +planning around it. + +**Plan written.** When `plan.md` and `tasks.md` are written, **stop**. +Reply with a 1-line summary per phase and the list of architecture +decisions. Ask the user to review. Once confirmed, the next step is +`/build` (Developer role). Do not invoke the Developer yourself. diff --git a/.agents/subagents/developer.md b/.agents/subagents/developer.md index b79f335..d6e9dbc 100644 --- a/.agents/subagents/developer.md +++ b/.agents/subagents/developer.md @@ -19,6 +19,10 @@ You are the **Developer**. Your job is to deliver the code and assets that satisfy `plan.md`, ticking off `tasks.md` as you go and proving each phase with the tests the Architect specified. +Your method lives in `.agents/skills/developer-playbook/SKILL.md` +(which links the shared `.agents/skills/design-principles/SKILL.md`). +Read it before starting; it is part of your instructions. + ## Goal Land the smallest set of changes that makes all phases of `plan.md` @@ -31,21 +35,39 @@ phase boundary. code. If `scratch.md` exists, read it too — the main agent may have left an `explorer` summary or a prior phase's hand-back note there. If the plan diverges from the spec or a step is ambiguous, - stop and ask. + stop and ask. On any conflict, the authority order is + `development/architecture.md` > `spec.md` > `plan.md` > `tasks.md`; never + silently resolve a contradiction — record it in `report.md` and + stop if it blocks a success criterion. +- Keep `report.md` current as you work: record each deviation from the + plan when it happens, and each approach you tried and abandoned (with + the evidence that killed it), not reconstructed at the end. Complete + it before the final `/verify` — the Reviewer audits it for honesty, + and an undeclared deviation is a review defect. +- When a decision exceeds your authority — loosening a test tolerance + or assertion, adding a dependency, changing behaviour the spec froze — + write a `DECISION-PENDING:` line in `report.md` and stop. Do not + resolve it locally. - Work **one phase at a time**. Do not begin phase N+1 until phase N's tests pass and its `tasks.md` boxes are ticked. - Write the test **first** when the plan calls for behaviour change — tests are the spec (see `AGENTS.md`). - Comments describe the code, not the PR: explain *why*, keep them accurate, and never commit review/release-process prose or - commented-out code (see `docs/style.md`, "Comments"). + commented-out code (see `development/style.md`, "Comments"). - Run the verification gate at every phase boundary. Do not declare a phase done until the gate is green. - Update `tasks.md` checkboxes as you complete each step, in the same commit as the code change. - Never silently skip, disable, or `@ignore` a failing test. If a test - must be skipped, draft an ADR under `docs/adr/` and ask before + must be skipped, draft an ADR under `development/adr/` and ask before proceeding. +- When reading gate output: passed counts may only grow. A drop you + cannot attribute to your own intentional, declared test removal means + stop, don't commit, and report the failure verbatim. Never add + `-x`/`-k`/`--ignore`-style narrowing, or edit markers or assertions, + to make the gate pass. Any new skip is a finding to explain in + `report.md`, not to ignore. - Never edit anything under `*/generated/`. - Never run destructive Git (`push --force`, `reset --hard origin/*`, history rewrites on shared branches). @@ -69,8 +91,15 @@ For each unchecked task in `tasks.md`, in order: moving on. 5. Tick the box in `tasks.md` and continue. -At the end of each phase, stop and hand off to the Reviewer -(`/verify`). +At the end of each phase, before handing off: walk the red-flag +checklist (`.agents/skills/design-principles/SKILL.md`) over your own +diff and fix what it catches — the Reviewer is for what you *can't* +see, not for what you didn't look at. **If that rework touched code, +run the verification gate again before you hand off.** The gate status +you report describes the tree you are actually leaving behind, not the +tree as it stood before the cleanup; the Reviewer re-runs the gate +itself, and a stale green is a false report claim, not a near miss. +Then stop and hand off to the Reviewer (`/verify`). ## Handoff diff --git a/.agents/subagents/explorer.md b/.agents/subagents/explorer.md index 25532c7..a4d77f7 100644 --- a/.agents/subagents/explorer.md +++ b/.agents/subagents/explorer.md @@ -37,8 +37,9 @@ repository and return a focused, citation-rich answer. - Allowed bash: `rg`, `grep`, `ls`, `cat`, `head`, `tail`, `wc`, `git log`, `git blame`, `git show`, `git diff` (read-only). `find` is not allowed — its `-delete`/`-exec` forms are not read-only; use `rg`/`ls` instead. -- Limit scope. If the question is broad, ask one clarifying question before - exploring. +- Limit scope. If the question is too broad to answer well, return one + clarifying question as your result instead of exploring — you run to + completion in a single turn, so the caller answers and re-invokes you. ## Output format diff --git a/.agents/subagents/product-owner.md b/.agents/subagents/product-owner.md index 24b0c78..9900632 100644 --- a/.agents/subagents/product-owner.md +++ b/.agents/subagents/product-owner.md @@ -3,7 +3,7 @@ name: product-owner description: | Use proactively at the start of any new feature, bug, or change request to turn a raw idea into a crisp, testable feature spec under - work/-/spec.md. Invoked by the /spec slash command. + development/work/-/spec.md. Invoked by the /spec slash command. Stops before any planning or implementation begins. tools: Read, Grep, Glob, Write permission: @@ -20,12 +20,18 @@ idea into a clear, scoped feature spec that captures user intent, success criteria, and out-of-scope items — **without** prescribing implementation. +Your method lives in `.agents/skills/product-owner-playbook/SKILL.md` +(which links the shared `.agents/skills/design-principles/SKILL.md`). +Read it before starting; it is part of your instructions. + ## Goal -Produce `work/-/spec.md` so the Architect can plan -against it. Do not create `plan.md`, `tasks.md`, or `scratch.md` — -those are owned by the Architect and Developer roles respectively -and they will write them from scratch. +Produce `development/work/-/spec.md` so the Architect can plan +against it. Do not create `plan.md`, `tasks.md`, `report.md`, or +`scratch.md` — the Architect owns `plan.md`/`tasks.md`, the Developer +owns `report.md`, and each writes its own files from scratch; +`scratch.md` is the feature's shared channel, created by whoever +needs it first. ## Constraints @@ -36,11 +42,25 @@ and they will write them from scratch. — rewrite it. - Non-goals are mandatory. List at least one thing this spec deliberately does not cover. -- Never edit files outside the feature's `work/-/` +- Never edit files outside the feature's `development/work/-/` directory. -- If the request is ambiguous (unclear user, unclear value, unclear - done-condition), stop and ask **one** clarifying question before - writing anything. +- Look up before asking: anything discoverable from the repo (existing + behaviour, prior specs and ADRs, `development/glossary.md`) you find + yourself and state as findings. Ask only what only the user can know. +- Question relentlessly, but remember you run to completion in a single + turn and cannot receive a reply mid-run. Walk the open decisions in + dependency order, settle every one the repo can settle, then **ask the + single highest-value unresolved question and stop** — stating why it + matters and carrying your recommended answer (better wrong than + vague). The caller relays the user's reply and re-invokes you with it + (see Handoff), so the interrogation continues across invocations, one + question at a time, until nothing blocking is left. Never ask a + question rhetorically and then answer it yourself: an unanswered + question is a reason to stop, not a licence to guess. +- Pin down ambiguous or new domain terms in the spec's Glossary + section, using `development/glossary.md` terms where they exist. + Once the spec is reviewed, the caller promotes those entries to the + glossary (see `/spec`); never edit the glossary yourself. ## Output format @@ -65,13 +85,35 @@ and they will write them from scratch. ## Non-goals - +## Constraints +- +- Budgeted resource: + +## Glossary +- **** — + ## Open questions - ``` ## Handoff -When `spec.md` is written, **stop**. Reply with a 3-bullet summary -(problem / goal / top success criterion) and ask the user to review. +You have two ways to stop, and only two. + +**Question outstanding.** If something blocking is still unresolved, do +not write `spec.md` on a guess. **Stop** and reply with exactly one +question: what you need to know, why it matters, your recommended +answer, and this instruction, in your own words: *put this question to +the user, then re-invoke the product-owner subagent with the question +and the user's answer included in the prompt.* State it every time. Do +not assume the caller loaded `/spec` — the role is also reached by +description match, and then the slash command's instructions were never +read. + +**Spec written.** When `spec.md` is written, **stop**. Reply with a +3-bullet summary (problem / goal / top success criterion), name the +shared understanding you recorded, and ask the user to confirm it. Once the user confirms, the next step is `/plan` (Architect role). Do not invoke the Architect yourself. diff --git a/.agents/subagents/reviewer.md b/.agents/subagents/reviewer.md index 8b7b4c9..04ed630 100644 --- a/.agents/subagents/reviewer.md +++ b/.agents/subagents/reviewer.md @@ -42,33 +42,94 @@ Produce a clear **GO** or **NEEDS-WORK** verdict, with a citation-rich defect list when NEEDS-WORK, so the Developer knows exactly what to fix before the next push. +Your method lives in `.agents/skills/reviewer-playbook/SKILL.md` +(which links the shared `.agents/skills/design-principles/SKILL.md`). +Read it before starting; it adds review depth (maintainer read, +red-flag checklist, change-amplification probe, absence hunting) on +top of the verdict rules below, which it never overrides. + ## Constraints -- Read `spec.md`, `plan.md`, the current `tasks.md`, and the diff - (`git diff` against the integration branch, plus `git log` for the - feature branch) before judging anything. +- Read `spec.md`, `plan.md`, the current `tasks.md`, `report.md` (if the + Developer has started it), and the diff (`git diff` against the + integration branch, plus `git log` for the feature branch) before + judging anything. If `plan.md` has a **Review checklist** section, it + is part of your instructions. +- **Scope check first.** Run `git diff --stat` against the integration + branch. Every touched file must be plausibly required by the plan. + Touching another feature's `development/work/` directory, a merged feature's + `report.md`, or the body of an existing ADR is an automatic MAJOR + defect unless the plan explicitly called for it. The two registers + have their own scope contracts instead: a decision-register row must + pair with a `DECISION-PENDING:` line in this diff, and a new + `development/glossary.md` entry must match a term in this feature's + reviewed spec **Glossary** section — a glossary edit with no matching + spec term, or any rename or meaning change of an existing entry, is + out of scope mid-feature (MAJOR). +- **Never trust the Developer's narrative.** Re-run the gate yourself; + re-derive every claim you can check (line numbers, test names, + measured values) instead of quoting the hand-off summary. +- **Probe that new tests can fail.** For each nontrivial new test, reason + about what wrong implementation it would catch. A vacuous test — an + always-true assertion, a value compared to itself, a tolerance so wide + it cannot fail — is a MAJOR defect, not coverage. - Allowed bash: the project's verify/test/lint commands, plus read-only inspection (`git log`, `git diff`, `git blame`, `git show`, `rg`, `grep`, `ls`, `cat`, `head`, `tail`, `wc`). `find` is not allowed — its `-delete`/`-exec` forms are not read-only. - Never edit files. Never run state-changing commands. Never auto-fix defects yourself — your job is to identify them, not to patch them. -- Check **all three** axes: +- Check **all four** axes: 1. **Spec conformance** — does each Success criterion in `spec.md` have observable evidence (a passing test, a screenshot, a log - line)? + line)? And does the delivered work respect the spec's + **Constraints** section, including the budgeted resource it names + (latency, memory, schedule, attention)? A constraint that was + silently exceeded is a defect even when every success criterion + passes — and a budget that nothing in the diff measures is a + missing-evidence finding, not a pass. 2. **Plan conformance** — were the phases delivered as planned, or - were undocumented detours taken? + were undocumented detours taken? On any conflict between the + documents, judge against the higher authority: + `development/architecture.md` > `spec.md` > `plan.md` > `tasks.md`. 3. **Implementation quality** — bugs, dead code, style violations, missing tests, untouched `tasks.md` boxes, undocumented decisions that should have been ADRs, and comment hygiene: review/release-process prose, stale or inaccurate comments, comments that merely restate the code, and duplicated rationale - blocks. See `docs/style.md` ("Comments") for the patterns to flag. + blocks. See `development/style.md` ("Comments") for the patterns to flag. + Plus design quality on *new* code: the red-flag checklist in + `.agents/skills/design-principles/SKILL.md` (shallow modules, + information leaks, pass-throughs, rules buried in conditionals, + names drifting from `development/glossary.md`), and the + change-amplification probe from the reviewer playbook. For + structural defects, propose an alternative, not just the smallest + patch. + 4. **Report honesty** — `report.md` must match the code. Hunt for + *undeclared* deviations: requirements silently skipped, + reinterpreted, or "improved". An inaccurate report claim is a + defect even when the code is correct. Every `DECISION-PENDING:` + line **added by this diff** must come with a row in the decision + register (`development/adr/README.md`) in the same change; markers in + already-merged reports are history, judged by the register alone. - Run the project's verification gate. A failing gate is an automatic - NEEDS-WORK with the failure cited verbatim. + NEEDS-WORK with the failure cited verbatim. When reading gate output: + passed counts may only grow (a drop the diff doesn't explain is a + defect); any *new* skip is a defect to explain, not to ignore; any + `-x`/`-k`/`--ignore`-style narrowing or marker/assertion edit that + exists to make the gate pass is a MAJOR defect. - Be specific. Every defect must cite `path/to/file.ext:LINE` and a - one-sentence reason. + one-sentence reason, ranked by severity: + - **MAJOR** — blocks the merge: red or vacuous gate/test, scope + violation, weakened tolerance or assertion, undeclared substantive + deviation, false report claim, a `DECISION-PENDING:` line added + without its register row. + - **MINOR** — should be fixed, but a maintainer could waive it: + fragile test construction, misleading comment/docstring, missing + declaration of a harmless deviation. + - **INFO** — observations and follow-up candidates. No action needed. + Any MAJOR (or a red gate) means NEEDS-WORK; MINOR/INFO alone may + still be GO, with the findings listed. ## Output format @@ -85,9 +146,13 @@ fix before the next push. ## Verification gate -## Defects (if NEEDS-WORK) +## Defects +MAJOR 1. `path/file.ext:42` — . -2. ... +MINOR +... +INFO +... ## Suggested next step -/ directory this PR delivers, or state + that the change was small enough to skip the four-phase loop. --> + +## Definition of done + +- [ ] Gate green: `make verify` run locally, exit 0 +- [ ] Every success criterion in `spec.md` has observable evidence + (a passing test, a screenshot, a log line) +- [ ] `report.md` written: what was built, deviations declared, negative + results recorded, follow-ups listed +- [ ] No test, tolerance, or assertion weakened to make the gate pass; + any new skip is explained in `report.md` +- [ ] Every `DECISION-PENDING:` line added by this PR has a row in the + decision register (`development/adr/README.md`) +- [ ] New structural decisions (dependency, persistence, protocol, auth) + have an ADR in `development/adr/` + +## Deviations & notes for the reviewer + + diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index b427559..dbb91c5 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -12,7 +12,7 @@ # # GPU (T2/T3) jobs are pending dedicated runners; the suites exist under the # `gpu_cuda`/`gpu_rocm` markers and are opt-in via DEVMM_GPU (see -# docs/adr/0003-gpu-suite-waiver-for-0.1.0.md). +# development/adr/0003-gpu-suite-waiver-for-0.1.0.md). name: CI on: diff --git a/.gitignore b/.gitignore index cbc7954..67f88da 100644 --- a/.gitignore +++ b/.gitignore @@ -219,6 +219,7 @@ __marimo__/ # >>> ai-agent-harness (managed by copier) >>> # Personal overrides — never commit +AGENTS.local.md CLAUDE.local.md .claude/settings.local.json .claude/.last_* @@ -229,7 +230,7 @@ CLAUDE.local.md .cursor/settings.local.json # Agent scratch space (team decision; comment out to commit) -work/*/scratch.md +development/work/*/scratch.md .agents/.cache/ # <<< ai-agent-harness (managed by copier) <<< diff --git a/.opencode/opencode.jsonc b/.opencode/opencode.jsonc index df976c9..cc27289 100644 --- a/.opencode/opencode.jsonc +++ b/.opencode/opencode.jsonc @@ -7,8 +7,8 @@ // bloating AGENTS.md itself. "instructions": [ "AGENTS.md", - "docs/architecture.md", - "docs/style.md" + "development/architecture.md", + "development/style.md" ], // Default mode. `build` is the full-access primary; switch to `plan` for @@ -49,7 +49,7 @@ // We route formatting through the same per-file entry point Claude Code uses // (`scripts/fmt-file.sh`) so editor-time formatting is byte-for-byte what the // verify gate enforces. "$FILE" is replaced with the edited file's path; - // matching is by `extensions`. See docs/harness-usage.md. + // matching is by `extensions`. See development/harness-usage.md. "formatter": { // Disable the built-in formatter for this language so we don't double-format. "ruff": { "disabled": true }, diff --git a/AGENTS.md b/AGENTS.md index 27d419c..5d00166 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -8,7 +8,7 @@ existing allocators (rmm, hipMM, CuPy pools, libc) rather than implementing allocation strategies, and exposes every allocation as a zero-copy **DLPack ≥ 1.0** producer. `ctypes` is used only for DLPack structs/capsules and raw C-ABI runtime interop. Authoritative design: -[`work/devmm-design.md`](work/devmm-design.md). +[`development/devmm-design.md`](development/devmm-design.md). ## Stack @@ -16,7 +16,7 @@ raw C-ABI runtime interop. Authoritative design: - Package / build manager: **uv** - License: BSD-3-Clause - Tool versions, install steps, and new-machine setup: - see [`docs/tool-bootstrap.md`](docs/tool-bootstrap.md). + see [`development/tool-bootstrap.md`](development/tool-bootstrap.md). ## Commands (prefer these over guessing) @@ -30,12 +30,14 @@ this file. ## Where things live (capabilities, not paths) -- Architecture overview: [`docs/architecture.md`](docs/architecture.md) -- Driving the harness (Claude Code & OpenCode): [`docs/harness-usage.md`](docs/harness-usage.md) -- Style guide: [`docs/style.md`](docs/style.md) -- Testing strategy: [`docs/testing.md`](docs/testing.md) -- ADRs (decisions of record): [`docs/adr/`](docs/adr/) -- Per-feature specs: [`work/-/`](work/) +- Architecture overview: [`development/architecture.md`](development/architecture.md) +- Driving the harness (Claude Code & OpenCode): [`development/harness-usage.md`](development/harness-usage.md) +- Style guide: [`development/style.md`](development/style.md) +- Domain vocabulary (use these terms in code and specs): [`development/glossary.md`](development/glossary.md) +- Testing strategy: [`development/testing.md`](development/testing.md) +- ADRs (decisions of record): [`development/adr/`](development/adr/) +- Per-feature work units: [`development/work/-/`](development/work/) +- Index of the whole tree: [`development/README.md`](development/README.md) - Supported agents & how to add one: [`.agents/README.md`](.agents/README.md) ## Do @@ -43,21 +45,36 @@ this file. - `uv` is bootstrapped automatically at session start by [`.agents/hooks/ensure-toolchain.sh`](.agents/hooks/ensure-toolchain.sh); if you land in a bare shell without it, run that script (details in - [`docs/tool-bootstrap.md`](docs/tool-bootstrap.md)). + [`development/tool-bootstrap.md`](development/tool-bootstrap.md)). - Run `make verify` before claiming a task is done. - For a net-new feature, follow the four-phase loop: `/spec` (Product Owner) → `/plan` (Architect) → `/build` (Developer) → `/verify` (Reviewer). Each phase stops for review before the next begins. See `.agents/commands/` and `.agents/subagents/`. -- For a new architectural choice (dependency, framework, persistence, auth), - add an ADR in `docs/adr/`. ADRs are append-only; supersede with a new file. +- On any conflict between documents, the authority order is + [`development/architecture.md`](development/architecture.md) > `spec.md` > `plan.md` > + `tasks.md`. Never silently resolve a contradiction — record it in the + feature's `report.md` and stop if it blocks a success criterion. +- When a decision exceeds your authority (loosening a test tolerance or + assertion, adding a dependency, changing behaviour the spec froze), write a + `DECISION-PENDING:` line in the feature's `report.md` and stop; every such + line gets a row in the decision register + ([`development/adr/README.md`](development/adr/README.md)) in the same PR. +- Some significant decisions warrant an ADR under `development/adr/` — check the + criteria in [`development/adr/README.md`](development/adr/README.md) before + writing one; skip trivial or easily reversed choices. ADRs are append-only; + supersede with a new file. - When investigating a large codebase, prefer the `explorer` subagent (read-only) over loading large files into the main context. ## Don't - Don't edit anything under `*/generated/` — it's overwritten by your project's codegen pipeline. -- Don't add a runtime dependency without an ADR. +- Don't add a **required** runtime dependency without an ADR. `devmm`'s required + dependency set is empty by design and `tests/test_packaging.py` enforces it; + new capabilities go in an optional extra. Other structural choices (datastore, + wire protocol) go through the ADR bar + ([`development/adr/README.md`](development/adr/README.md)). - Don't run destructive Git: `push --force`, `reset --hard origin/*`, history rewrites on shared branches. - Don't put secrets, hostnames, or per-developer paths in this file — @@ -67,32 +84,42 @@ this file. ## Conventions -- Code style: see `docs/style.md`. One worked example > a page of prose. +- Code style: see `development/style.md`. One worked example > a page of prose. - Comments describe the code, not the process: explain *why*, keep them accurate, no review/release-process prose. See - [`docs/style.md`](docs/style.md#comments). + [`development/style.md`](development/style.md#comments). - Tests are the spec. If you change behaviour, change a test first. - Commit messages: **Conventional Commits 1.0.0** — apply the format to the **PR title** (squash-merge). - See [`docs/style.md`](docs/style.md#commit-messages) for the format, + See [`development/style.md`](development/style.md#commit-messages) for the format, type list, breaking-change syntax, examples, and full merge-strategy guidance. - Changelog: if the project keeps a `CHANGELOG.md`, log user-facing changes under `[Unreleased]` as one concise bullet each, leading with the - file/behaviour. See [`docs/style.md`](docs/style.md#changelog). + file/behaviour. See [`development/style.md`](development/style.md#changelog). - Branch names: `/` for personal branches; bare slug for shared feature branches. ## Working memory -Per-feature spec directories use this layout: +`development/` is the repo's process memory — agent-facing docs, ADRs, and +work units; it is never published. `docs/`, if the project has one, is user +documentation only (like `src/` is for sources). Per-feature work-unit +directories use this layout: ``` -work/-/ +development/work/-/ ├─ spec.md # WHAT and WHY; no implementation detail ├─ plan.md # numbered phased plan; each phase has tests ├─ tasks.md # checkbox list the agent ticks off -└─ scratch.md # agent's working notes; cleared on completion (gitignored) +├─ report.md # what actually happened: deviations, negative results, follow-ups +└─ scratch.md # shared working channel: hand-backs, spike Q&A, explorer + # summaries; dies at completion (gitignored) ``` `scratch.md` is gitignored by default. Promote anything durable into `spec.md`, -`plan.md`, an ADR, or `docs/`. +`plan.md`, `report.md`, an ADR, or the `development/` docs. + +Liveness: `spec.md` freezes once reviewed, `plan.md` once build starts, +`report.md` at merge. Never retro-edit a merged report — new findings go in +the report of the feature that finds them. Full table: +[`development/harness-usage.md`](development/harness-usage.md#document-liveness). diff --git a/CHANGELOG.md b/CHANGELOG.md index 2fe91ce..b01b8d6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,8 +11,17 @@ and this project adheres to ### Added - `gpu-test-cuda` extra and `make test-gpu-cuda` to run the CUDA (T2) hardware - suite, with the setup recipe in [`docs/testing.md`](docs/testing.md#running-the-cuda-gpu-suite-on-hardware). - First hardware run recorded in [ADR 0003](docs/adr/0003-gpu-suite-waiver-for-0.1.0.md). + suite, with the setup recipe in [`development/testing.md`](development/testing.md#running-the-cuda-gpu-suite-on-hardware). + First hardware run recorded in [ADR 0003](development/adr/0003-gpu-suite-waiver-for-0.1.0.md). + +### Changed + +- Repository layout: the agent-facing docs, ADRs and per-feature work units moved + from `docs/` and `work/` into `development/`, and `docs/` is now reserved for + user documentation (`docs/api.md`). Harness updated to template v0.6.0, which + adds `report.md` per work unit, a decision register in + [`development/adr/README.md`](development/adr/README.md), a + [glossary](development/glossary.md), and role playbook skills. ### Fixed @@ -30,7 +39,7 @@ allocating and managing device memory across CPU, CUDA and ROCm, exporting every allocation as a DLPack ≥ 1.0 producer. - GPU hardware test suites (CUDA T2, ROCm T3) are **waived for this release** - per [ADR 0003](docs/adr/0003-gpu-suite-waiver-for-0.1.0.md): no GPU runner + per [ADR 0003](development/adr/0003-gpu-suite-waiver-for-0.1.0.md): no GPU runner has executed them yet, so GPU code paths are validated only through the fake-API suites on CPU-only CI. The suites ship skip-gated behind the `gpu_cuda`/`gpu_rocm` markers (`DEVMM_GPU=cuda|rocm` opts in). diff --git a/CLAUDE.md b/CLAUDE.md index 9deb90d..bc0f3f3 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -6,7 +6,8 @@ - Use **TodoWrite** liberally; it doubles as harness echo and helps you stay on track during long runs. - **Skills** are under `.claude/skills/` (symlink to `.agents/skills/`). - Invoke by capability, e.g. "use the verify skill". + Invoke by capability, e.g. "use the architect-playbook skill" or + "use the verify skill". - **Subagents** are under `.claude/agents/` (symlink to `.agents/subagents/`). Role agents pair 1:1 with the slash commands below (`product-owner`, `architect`, `developer`, `reviewer`); `explorer` diff --git a/Makefile b/Makefile index 8fde7c3..c75aac4 100644 --- a/Makefile +++ b/Makefile @@ -14,21 +14,21 @@ test-all: test ## Run the full suite (override to add integration/e2e) @echo "test-all: extend this target with integration suites as needed" # The refleak harness and the shutdown subprocess tests run with dev-mode -# allocator/warning checks and faulthandler enabled (see docs/testing.md). +# allocator/warning checks and faulthandler enabled (see development/testing.md). test-devmode: ## Run the suite under PYTHONDEVMODE=1 with faulthandler PYTHONDEVMODE=1 PYTHONFAULTHANDLER=1 uv run --extra test pytest -q -# The CUDA (T2) hardware suite (docs/adr/0003). Install the deps first — +# The CUDA (T2) hardware suite (development/adr/0003). Install the deps first — # `uv pip install '.[test,gpu-test-cuda]'` plus the PyTorch CUDA wheel — then # this runs against the live device without re-resolving (`--no-sync` keeps the # hand-installed torch/toolchain wheels). CUDA_HOME is cleared so Numba links # libnvvm/nvjitlink from the cu12 wheels, not a newer system CUDA. Full setup -# recipe and rationale: docs/testing.md. -test-gpu-cuda: ## Run the CUDA (T2) GPU suite on hardware (deps must be installed; see docs/testing.md) +# recipe and rationale: development/testing.md. +test-gpu-cuda: ## Run the CUDA (T2) GPU suite on hardware (deps must be installed; see development/testing.md) env -u CUDA_HOME -u CUDA_PATH DEVMM_GPU=cuda uv run --no-sync \ pytest tests/test_cuda_gpu.py tests/test_integrations_gpu.py -# Coverage thresholds (see docs/testing.md): >= 90% overall, >= 95% on the +# Coverage thresholds (see development/testing.md): >= 90% overall, >= 95% on the # core domain model + DLPack layer. Residual uncovered lines carry reasoned # `# pragma: no cover` / exclusions (see pyproject.toml). coverage: ## Enforce the coverage thresholds @@ -52,16 +52,16 @@ verify: ## What the agent runs before claiming done @./scripts/verify.sh # Gates are cumulative, so every gate-N aliases the current full verify gate -# (see docs/adr/0002-task-runner-make-over-just.md). +# (see development/adr/0002-task-runner-make-over-just.md). gate-%: verify @echo "gate-$*: green" gate-all: verify ## Cumulative gate: lint + mypy --strict + tests + packaging @echo "gate-all: green" -# Every release check that runs without GPU hardware (see docs/testing.md); +# Every release check that runs without GPU hardware (see development/testing.md); # the GPU suites run on their own runners -# (docs/adr/0003-gpu-suite-waiver-for-0.1.0.md). +# (development/adr/0003-gpu-suite-waiver-for-0.1.0.md). release-gate: verify coverage test-devmode ## Release gate: verify + coverage + dev-mode suite @echo "release-gate: green" diff --git a/README.md b/README.md index e68132a..c4c0f33 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,7 @@ allocation strategies, and exposes every allocation as a zero-copy **DLPack ≥ 1.0** producer consumable by any Array-API library (NumPy, CuPy, PyTorch, JAX). Zero required dependencies; `ctypes` only. -The authoritative design is [`work/devmm-design.md`](work/devmm-design.md); +The authoritative design is [`development/devmm-design.md`](development/devmm-design.md); the API reference with executable examples is [`docs/api.md`](docs/api.md). ## Usage @@ -78,8 +78,8 @@ This repository follows the **agent-agnostic harness** convention: stanzas (skills, subagents, slash commands, hooks). - Shared agent assets (skills, subagents, and slash commands) live under `.agents/` and are symlinked into `.claude/` and `.opencode/`. -- Per-feature specs go under `work/-/`. -- Architecture decisions are in `docs/adr/` (Michael Nygard format, append-only). +- Per-feature specs go under `development/work/-/`. +- Architecture decisions are in `development/adr/` (Michael Nygard format, append-only). See `AGENTS.md` for the full conventions. diff --git a/development/README.md b/development/README.md new file mode 100644 index 0000000..c95132c --- /dev/null +++ b/development/README.md @@ -0,0 +1,44 @@ +# development/ — repo process memory + +Everything agents and developers need to work on this repo: conventions, +decisions, and the per-feature work-document lifecycle. This tree is +**repo-internal** — never published, never linked from a documentation site. +User-facing documentation, if the project has any, lives in `docs/` (kept for +that purpose alone, the way `src/` is kept for sources) and is written for +readers, not rewritten from these files mechanically. + +| File / folder | What | +| --- | --- | +| [`harness-usage.md`](harness-usage.md) | how to drive the agent harness (Claude Code & OpenCode): phases, subagents, hooks, document liveness | +| [`architecture.md`](architecture.md) | orientation: system structure, boundaries, invariants | +| [`style.md`](style.md) | code style, comments, commit messages, changelog | +| [`glossary.md`](glossary.md) | the project's ubiquitous language: domain terms used in specs, code, and conversation | +| [`testing.md`](testing.md) | test layering, gate commands, gate-output reading rules | +| [`tool-bootstrap.md`](tool-bootstrap.md) | toolchain install and new-machine setup | +| [`adr/`](adr/) | architecture decision records (append-only) + the decision register | +| `work/-/` | one folder per feature: `spec.md` / `plan.md` / `tasks.md` / `report.md` (+ gitignored `scratch.md`) | + +Scaffold markers, used consistently across these files (and the example +`work/` unit): + +- `_Fill in: …_` — a block for you to replace with real content, deleting + the marker. `rg 'Fill in:' development/` lists what is still scaffold. +- `` — inline: replace the bracketed text, keep what's around + it. The brackets are never given backticks of their own, and they stay + bare inside a backticked path too (`work/-/`). That is the + form the role subagents emit in their output formats, so a scaffold and a + freshly written document look alike — at the price of a Markdown preview + swallowing a bare `` as an unknown HTML tag, which these + read-as-text files accept. +- `> blockquote` — a durable note about how the document itself works; + it stays. + +These documents are living but **trunk-gated**: agents follow them and +propose changes via a dedicated PR — never by silently editing them in the +middle of a feature. Two files are **registers**, not prose docs, and accrete +mid-feature through their own reviewable channels: the decision register in +[`adr/README.md`](adr/README.md) (one row per escalated decision, in the same +PR as its `DECISION-PENDING:` marker) and [`glossary.md`](glossary.md) (entries +promoted from a reviewed spec's Glossary section at `/spec` wrap-up). Which +files freeze, and when: +[`harness-usage.md`](harness-usage.md#document-liveness). diff --git a/development/adr/0001-record-architecture-decisions.md b/development/adr/0001-record-architecture-decisions.md index bb50dac..408d794 100644 --- a/development/adr/0001-record-architecture-decisions.md +++ b/development/adr/0001-record-architecture-decisions.md @@ -17,7 +17,7 @@ are the lightest-weight option that satisfies: ## Decision -Use Michael Nygard's ADR format. One file per decision in `docs/adr/`, named +Use Michael Nygard's ADR format. One file per decision in `development/adr/`, named `NNNN-kebab-title.md` (zero-padded to 4 digits). Sections: **Status, Context, Decision, Consequences**. Supersession is recorded by a new ADR that references the old one, not by editing history. diff --git a/development/adr/0002-task-runner-make-over-just.md b/development/adr/0002-task-runner-make-over-just.md index 1f825a1..9141544 100644 --- a/development/adr/0002-task-runner-make-over-just.md +++ b/development/adr/0002-task-runner-make-over-just.md @@ -7,7 +7,7 @@ Accepted ## Context The devmm implementation plan -([`work/devmm-implementation-plan.md`](../../work/devmm-implementation-plan.md), +([`development/devmm-implementation-plan.md`](../devmm-implementation-plan.md), §0) fixes the repository tooling and names **`just`** as the command runner, with `justfile` targets `just lint`, `just typecheck`, `just test`, `just gate N`, and `just gate-all` driving the phased build. @@ -43,7 +43,7 @@ targets map onto `make` targets: `make verify` remains the cumulative gate the agent hooks call; `make gate-all` runs the full release-gate sequence. The `gate-N` / `gate-all` / `typecheck` -targets are added in build phase p00 (`work/2026-07-p00-scaffold/`), not before. +targets are added in build phase p00 (`development/work/2026-07-p00-scaffold/`), not before. This is a deviation from the implementation plan's stated tooling. The plan text is not amended (it remains the historical build order); this ADR is the record of diff --git a/development/adr/0003-gpu-suite-waiver-for-0.1.0.md b/development/adr/0003-gpu-suite-waiver-for-0.1.0.md index 392e3a5..9804f3d 100644 --- a/development/adr/0003-gpu-suite-waiver-for-0.1.0.md +++ b/development/adr/0003-gpu-suite-waiver-for-0.1.0.md @@ -9,7 +9,7 @@ T3-only waiver clause. ## Context The 0.1.0 release gate -([`work/devmm-implementation-plan.md`](../../work/devmm-implementation-plan.md), +([`development/devmm-implementation-plan.md`](../devmm-implementation-plan.md), Phase 12) requires: - **Gate 2 (T2)**: the CUDA GPU job green — the MR conformance suite over @@ -71,7 +71,7 @@ term above. ROCm (T3) remains unexecuted. `torch` 2.11.0+cu128, `numba` 0.66.0 + `numba-cuda` 0.30.4, `filecheck` 1.0.3. Numba's JIT toolchain aligned on the cu12 wheels (`nvidia-cuda-nvcc-cu12` / `nvidia-nvjitlink-cu12` both 12.8.93, `CUDA_HOME` unset) — see - [`docs/testing.md`](../testing.md#running-the-cuda-gpu-suite-on-hardware). + [`development/testing.md`](../testing.md#running-the-cuda-gpu-suite-on-hardware). - **Result**: `tests/test_cuda_gpu.py` + `tests/test_integrations_gpu.py` — **41 passed, 1 skipped** (the stream-race misorder canary is best-effort and did not manifest). Reproduce with `make test-gpu-cuda`. diff --git a/development/adr/README.md b/development/adr/README.md index 958fdf2..13a6972 100644 --- a/development/adr/README.md +++ b/development/adr/README.md @@ -3,6 +3,24 @@ One file per architecturally significant decision, named `NNNN-kebab-title.md` (zero-padded to 4 digits, append-only). +## When to write one + +Write an ADR only when **all three** are true: + +1. **Hard to reverse** — the cost of changing your mind later is meaningful. +2. **Surprising without context** — a future reader will wonder "why did they + do it this way?" +3. **The result of a real trade-off** — there were genuine alternatives and you + picked one for specific reasons. + +Most decisions fail at least one test and need no ADR: reversible, obvious, or +uncontested choices belong in the code, the commit message, or the plan — not +here. When in doubt, leave it out; you can add the ADR the day the trade-off +actually bites. A new datastore, wire protocol, or auth model usually clears all +three bars; a library swap you could undo in an afternoon usually does not. For +a one-line operational fact that clears none of the bars, a decision-register +row alone is enough (see "ADR or register row?" below). + ## Format (Michael Nygard) ```markdown @@ -21,9 +39,38 @@ One file per architecturally significant decision, named ``` -Supersession adds a new ADR that references the old one; it does not edit -the old file's content. The full historical record is the value. +Supersession adds a new ADR that references the old one; the only edit the +old file receives is flipping its Status line to `Superseded by ADR M` — +its body is never rewritten. The full historical record is the value. Optional: install [`adr-tools`](https://github.com/npryce/adr-tools) (single shell-script binary) and use `adr new ""` to scaffold the next file with the correct number. + +## Decision register + +One row per decision that an agent escalated (a `DECISION-PENDING:` line in a +feature's `report.md`) or that a human granted outside an ADR (a tolerance, a +pin, a one-line operational fact). Rows are appended and their Status flipped +in place (`pending` → `accepted` / `rejected`); nothing else is edited. + +Contract: every `DECISION-PENDING:` line lands in the same PR as its register +row. After merge the report freezes, so the marker line stays as history and +**this register alone** records the outcome — to see what is still open, scan +the table for `pending` rows, not the reports. The reviewer checks the +contract per change: a `DECISION-PENDING:` line in the diff without a row +here is a defect. (The marker is the colon form, `DECISION-PENDING:`; +mentions of the token without the colon are prose, not markers.) + +| ID | Date | Decision | Status | Source | Evidence | +|---|---|---|---|---|---| +| 2026-07-p12.1 | 2026-07-17 | Waive the T2 (CUDA) and T3 (ROCm) hardware gates for the 0.1.0 release; CPU CI plus the fake-api suites stand in. Maintainer sign-off extended the p12 spec's T3-only clause to T2. | accepted | [ADR 0003](0003-gpu-suite-waiver-for-0.1.0.md) | `0900acf` | +| 2026-07-p12.2 | 2026-07-17 | Coverage thresholds for the release gate: ≥ 90% overall, ≥ 95% on `devmm._core` + `devmm._dlpack`, enforced by the `gate-coverage` CI leg. | accepted | [`development/testing.md`](../testing.md#coverage-targets) | `97cc5c9` | + +(ID = `<feature-slug>.<k>`, e.g. `2026-07-user-auth.1`. Source = the report +or ADR that raised it. Evidence = the PR/commit that settled it.) + +**ADR or register row?** If the decision shapes structure — of the code, the +repo, or the process — and someone will later ask *why*, write an ADR and add +a register row whose Source points at it. If it is a one-line operational +fact, a register row alone is enough. diff --git a/development/architecture.md b/development/architecture.md index afdc1cf..fe0f6a6 100644 --- a/development/architecture.md +++ b/development/architecture.md @@ -10,7 +10,7 @@ allocators (rmm, hipMM, CuPy pools, libc) rather than implementing allocation strategies, and exposes every allocation as a zero-copy **DLPack ≥ 1.0** producer consumable by any Array-API library. `ctypes` is used only for building DLPack C structs/capsules and for raw C-ABI runtime interop. The layered design (and its -rationale) is authoritative in [`work/devmm-design.md`](../work/devmm-design.md). +rationale) is authoritative in [`development/devmm-design.md`](devmm-design.md). ## Module map diff --git a/development/glossary.md b/development/glossary.md new file mode 100644 index 0000000..8790164 --- /dev/null +++ b/development/glossary.md @@ -0,0 +1,28 @@ +# Glossary — the project's ubiquitous language + +One entry per domain term, as the domain experts use it. Code, specs, +plans, and conversation all use these exact terms: when a term changes +here, the corresponding code identifiers are renamed in the same change, +and when the code needs a concept this file lacks, that's a missing +entry, not a private invention. + +Entry format — term, domain meaning, code name only if it must differ: + +``` +- **Term** — what it means in the domain, one or two sentences. + (code: `TermInCode`, only when it can't match the term itself) +``` + +This file is a **register**, like the decision register in +[`adr/README.md`](adr/README.md): it accretes mid-feature through one +channel only. New terms arrive through a feature spec's **Glossary** +section (`/spec`, Product Owner): the spec proposes, the user reviews +the spec, and the reviewed entries are promoted here verbatim at +`/spec` wrap-up — the Reviewer checks each new entry against the spec +that proposed it. Renames and meaning changes are not register +traffic: those are a dedicated PR, trunk-gated like the rest of +`development/` (see [`README.md`](README.md)). + +## Terms + +_None yet. The first feature spec will propose some._ diff --git a/development/harness-usage.md b/development/harness-usage.md index 7e15bd0..18821ee 100644 --- a/development/harness-usage.md +++ b/development/harness-usage.md @@ -8,14 +8,14 @@ hooks) for different tasks — and how to phrase prompts so the existing `.agent configuration is used without restating conventions every time. See also: [`AGENTS.md`](../AGENTS.md) (agent instructions), -[`CLAUDE.md`](../CLAUDE.md) (Claude Code specifics), [`docs/style.md`](style.md), -[`docs/testing.md`](testing.md), [`.agents/README.md`](../.agents/README.md) +[`CLAUDE.md`](../CLAUDE.md) (Claude Code specifics), [`development/style.md`](style.md), +[`development/testing.md`](testing.md), [`.agents/README.md`](../.agents/README.md) (supported agents & how to add one). ## The shared model The canonical definitions live once under `.agents/` and are symlinked into each -tool's config directory, so the same roles, commands, and skill serve every +tool's config directory, so the same roles, commands, and skills serve every tool. **You edit the files under `.agents/`, never the symlinks.** ``` @@ -24,6 +24,11 @@ tool. **You edit the files under `.agents/`, never the symlinks.** .agents/skills/ ─┴─► .claude/skills/ .opencode/skills/ # skills ``` +`.agents/skills/` holds the four **role playbooks** (`{product-owner,architect, +developer,reviewer}-playbook`) plus the shared `design-principles` ground rules +they all link. Each role subagent reads its own playbook as part of its +instructions; the `verify` skill is separate — see below. + The design is a **gated four-phase loop** — one slash command per phase, each backed by a single-purpose subagent that **stops for human review before the next phase starts**: @@ -35,7 +40,7 @@ idea ──/spec──▶ spec.md ──/plan──▶ plan.md + tasks.md ── ``` Plus a read-only `explorer` agent any phase can call for codebase Q&A. Each -phase writes fixed artifacts under `work/<YYYY-MM>-<slug>/`. +phase writes fixed artifacts under `development/work/<YYYY-MM>-<slug>/`. The five subagents and their access: @@ -52,7 +57,8 @@ The five subagents and their access: ### 1. Manual (you type it) - **Slash commands:** `/spec <slug>`, `/plan [dir]`, `/build [dir]`, `/verify`. -- **Skill by name:** "use the verify skill". +- **Skill by name:** "use the architect-playbook skill", "use the verify + skill". - **Subagent by name:** Claude Code — "use the explorer subagent to find where X is wired up"; OpenCode — `@explorer find where X is wired up` (the filename is the agent id). @@ -64,6 +70,8 @@ your prompt matches that language, the capability fires without you naming it: - The `verify` skill fires on "verify", "is this ready", "ready to commit", "check this", or after any non-trivial edit. +- The role playbooks fire on method cues — "how should I design this", "weigh + the alternatives", "is this good" — independently of the slash commands. - `explorer` fires on "find where X is implemented", "how does Y work", "what calls Z". - The role agents fire on their phase cues (see [per-phase prompting](#writing-prompts-so-the-right-phase-config-is-used)). @@ -99,7 +107,7 @@ Claude Code only). You cannot prompt around the hooks: Permissions allowlist the build tool, read-only git (`status/diff/log/show`), and `rg/ls/cat/head/tail`; destructive operations are denied. -`.claude/rules/` holds path-scoped rule fragments (currently empty). Default to +`.claude/rules/` holds path-scoped rule fragments (comment hygiene ships there). Default to **plan mode** (`shift-tab`) for non-trivial work. ## OpenCode specifics @@ -107,7 +115,7 @@ and `rg/ls/cat/head/tail`; destructive operations are denied. OpenCode reads `.opencode/opencode.jsonc`, which sets: - **`instructions`** — loads [`AGENTS.md`](../AGENTS.md), - [`docs/architecture.md`](architecture.md), [`docs/style.md`](style.md) as + [`development/architecture.md`](architecture.md), [`development/style.md`](style.md) as always-on context. - **`default_agent: "build"`** — the session starts in the full-access `build` primary. OpenCode has two built-in **primary** agents, cycled with **Tab**: @@ -165,21 +173,32 @@ language and the correct agent + output format is selected automatically. - **Trigger words:** "new feature", "spec out", "I want to add…", or `/spec <slug>`. - **What you get:** `spec.md` with Problem / Goal / Users & stakeholders / - Success criteria / Non-goals / Open questions. WHAT and WHY only — no file - paths or libraries. + Success criteria / Non-goals / Constraints / Glossary / Open questions. WHAT + and WHY only — no file paths or libraries. - **Prompt tips:** Give the user-facing intent and at least one observable - success condition. Don't prescribe implementation — the PO strips it. Expect - _one_ clarifying question if ambiguous, then it stops for your review. + success condition. Don't prescribe implementation — the PO strips it. +- **Expect a dialogue, not one round.** The PO looks up whatever the repo can + answer, then stops on the single highest-value question it still needs you + for — with its own recommended answer attached, so "yes, go with that" is a + valid reply. Answer it and the loop resumes; it writes the spec once nothing + blocking is left (`/spec` caps this at five rounds). Repeated questions are + the design working, not the agent stuck. ### Phase 2 — Plan (architect) -- **Trigger:** `/plan` after the spec is reviewed (defaults to most recent `work/*`). +- **Trigger:** `/plan` after the spec is reviewed (defaults to most recent `development/work/*`). - **What you get:** `plan.md` (Architecture-decisions block + numbered phases, each ≤1 day with explicit Tests and Exit criteria) and a mirrored checkbox `tasks.md`. - **Prompt tips:** Run only once the spec is approved. New dependency/persistence/protocol choices are surfaced in the **Architecture decisions** block and flagged "ADR needed". It writes no code and stops. +- **Expect spike round-trips.** When the plan would rest on an untested + assumption, the architect hands back a `SPIKE-REQUEST:` and the main agent + runs a throwaway experiment (three rounds max), folding the findings into + the plan's **Spike findings** section — so `/plan` may take a few + hand-backs before `plan.md` appears. Spike code is disposable; only the + findings survive. ### Phase 3 — Build (developer) @@ -197,11 +216,14 @@ language and the correct agent + output format is selected automatically. ### Phase 4 — Verify / review (reviewer) - **Trigger:** `/verify`. -- **What you get:** a **GO / NEEDS-WORK** verdict across three axes — spec +- **What you get:** a **GO / NEEDS-WORK** verdict across four axes — spec conformance (each criterion has observable evidence), plan conformance (no - undocumented detours), implementation quality — plus a citation-rich defect - list (`path/file.ext:LINE`). It runs the gate; a red gate is automatic - NEEDS-WORK. + undocumented detours), implementation quality, and report honesty + (`report.md` must match the code; undeclared deviations are defects) — plus + a citation-rich defect list (`path/file.ext:LINE`) ranked MAJOR / MINOR / + INFO. It re-runs the gate itself (it never takes the Developer's word), and + it reads `plan.md`'s **Review checklist** as extra instructions. A red gate + or any MAJOR is automatic NEEDS-WORK. - **Loop back:** NEEDS-WORK → `/build` to fix; GO → ship/next phase. The reviewer is read-only — it never fixes, only reports. @@ -215,6 +237,24 @@ These are distinct: - **/verify command** = full reviewer pass against spec + plan + diff with a GO verdict. Use at a phase boundary. +## Document liveness + +Which harness files may still change, and when they freeze. "Frozen" means +content-frozen: fixing a broken link in a sanctioned cleanup is fine; changing +what the document *says* is not. + +| File | Liveness | +| --- | --- | +| `AGENTS.md`, `CLAUDE.md`, `development/*.md` | living, **trunk-gated**: changed via a dedicated PR (or an explicit maintainer request), never silently mid-feature. The two **registers** (last rows) accrete by their own contracts instead | +| `development/work/*/spec.md` | frozen once reviewed — scope changes get a new spec revision, noted in `report.md` | +| `development/work/*/plan.md` | frozen once `/build` starts — if the plan is wrong, hand back to `/plan`; don't edit it mid-build | +| `development/work/*/tasks.md` | living during build | +| `development/work/*/report.md` | frozen at merge — **never retro-edited**; new findings go in the report of the feature that finds them | +| `development/adr/NNNN-*.md` | frozen once accepted, except the Status line (supersede with a new ADR) | +| `development/adr/README.md` decision register | **register** — rows appended mid-feature (each `DECISION-PENDING:` marker lands with its row in the same PR), Status flipped in place | +| `development/glossary.md` | **register** — entries appended mid-feature only by promotion from a reviewed spec's **Glossary** section at `/spec` wrap-up (the Reviewer checks each new entry against that spec); renames and meaning changes are trunk-gated like prose | +| `development/work/*/scratch.md` | dead on completion (gitignored) | + ## Conventions the agents already know (don't re-specify) These are enforced by docs + hooks; restating them in prompts is noise: @@ -225,11 +265,17 @@ These are enforced by docs + hooks; restating them in prompts is noise: (see `scripts/verify.sh`). Keep the fast loop (`make test`) under ~60s; slow suites belong in CI. - **Tests are the spec** — a behaviour change means changing/adding a test first. -- **Commits** — Conventional Commits 1.0.0 in the **PR title** (squash-merge); branch commits can be freeform. See [`docs/style.md#commit-messages`](style.md#commit-messages). -- **ADRs** — any new dependency/persistence/protocol/auth decision gets an ADR in - `docs/adr/` (append-only). The architect flags these. +- **Commits** — Conventional Commits 1.0.0 in the **PR title** (squash-merge); branch commits can be freeform. See [`development/style.md#commit-messages`](style.md#commit-messages). +- **ADRs** — a significant decision may warrant an ADR in `development/adr/` + (append-only); check the criteria in [`development/adr/README.md`](adr/README.md) + before writing one — trivial or reversible choices get none. The architect flags + candidates. - **Working memory** — `scratch.md` is gitignored; promote durable notes into - spec/plan/ADR/docs. + spec/plan/report/ADR/docs. `report.md` is the durable account of what + actually happened (deviations, dead ends, follow-ups) and freezes at merge. +- **Decision escalation** — a decision beyond the agent's authority becomes a + `DECISION-PENDING:` line in `report.md` plus a register row in + [`development/adr/README.md`](adr/README.md), never a silent local fix. ## Quick-start cheatsheet diff --git a/development/style.md b/development/style.md index b13394c..4136574 100644 --- a/development/style.md +++ b/development/style.md @@ -124,7 +124,7 @@ When the project keeps a `CHANGELOG.md`, follow - Mark breaking changes (`### Removed (breaking)`, or a `!` per the commit convention) and add an `### Upgrade notes` block when an upgrade needs manual action. -- Link the ADR when the change has one (see [`docs/adr/`](adr/)). +- Link the ADR when the change has one (see [`development/adr/`](adr/)). - On release, rename `[Unreleased]` to the version + date and add the compare link. diff --git a/development/testing.md b/development/testing.md index 2da0c87..91d1b52 100644 --- a/development/testing.md +++ b/development/testing.md @@ -6,6 +6,20 @@ - **Fast loop**: `make test` — must finish in <60s. Add slow suites under `make test-all`. +## Reading gate output + +- **Passed counts may only grow.** A drop in passed tests that the diff + doesn't explain line-by-line (an intentional, declared removal) means: + stop, don't commit, report the failure verbatim — the full failure block, + not a summary. +- **Any new skip is a finding to explain,** in the feature's `report.md`, + not to ignore. +- **Never narrow the gate to make it pass.** No `-x`, `-k`, `--ignore`, + no marker edits, no deleted or weakened assertions. If you cannot go + green within your task's scope, report honestly and stop. +- The GPU suites are the one sanctioned exception: they skip by design off + hardware (`development/adr/0003`), so their skips are expected, not findings. + ## Layering The CPU memory resources make the **whole DLPack protocol testable without a diff --git a/development/work/2026-07-p00-scaffold/plan.md b/development/work/2026-07-p00-scaffold/plan.md index a66fb08..f1b27df 100644 --- a/development/work/2026-07-p00-scaffold/plan.md +++ b/development/work/2026-07-p00-scaffold/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 0 — Scaffold delta & CI skeleton -> **Context.** Step 00 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 0). Design sections: §2. This is the first step; no prior step is required. +> **Context.** Step 00 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 0). Design sections: §2. This is the first step; no prior step is required. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p00-scaffold/spec.md b/development/work/2026-07-p00-scaffold/spec.md index d746617..adf14de 100644 --- a/development/work/2026-07-p00-scaffold/spec.md +++ b/development/work/2026-07-p00-scaffold/spec.md @@ -1,6 +1,6 @@ # Phase 0 — Scaffold delta & CI skeleton -> **Context.** Step 00 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 0). Design sections: §2. This is the first step; no prior step is required. +> **Context.** Step 00 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 0). Design sections: §2. This is the first step; no prior step is required. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ The repo has a stub package tree, `pyproject.toml`, `py.typed` and a smoke test A green cumulative gate (lint + strict types + tests + wheel-in-bare-venv) runs locally and in CI on the T0 matrix and a T1 job, with the public-API snapshot and zero-dependency tests wired in. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - Building the wheel, installing it into an environment with no other packages, and importing it succeeds. diff --git a/development/work/2026-07-p01-dtypes-device/plan.md b/development/work/2026-07-p01-dtypes-device/plan.md index 5bdfa82..5647401 100644 --- a/development/work/2026-07-p01-dtypes-device/plan.md +++ b/development/work/2026-07-p01-dtypes-device/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 1 — dtypes & device -> **Context.** Step 01 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 1). Design sections: §3.1, §3.7. This step depends on **p00-scaffold** landing first. +> **Context.** Step 01 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 1). Design sections: §3.1, §3.7. This step depends on **p00-scaffold** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p01-dtypes-device/spec.md b/development/work/2026-07-p01-dtypes-device/spec.md index 12a7959..dc009ca 100644 --- a/development/work/2026-07-p01-dtypes-device/spec.md +++ b/development/work/2026-07-p01-dtypes-device/spec.md @@ -1,6 +1,6 @@ # Phase 1 — dtypes & device -> **Context.** Step 01 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 1). Design sections: §3.1, §3.7. This step depends on **p00-scaffold** landing first. +> **Context.** Step 01 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 1). Design sections: §3.1, §3.7. This step depends on **p00-scaffold** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ Nothing yet describes element types or device identity, so no other module can n `DType` and `Device`/`DeviceType` exist, map exactly onto DLPack codes, and construct from Array-API strings and duck-typed NumPy dtypes with no NumPy import at module scope. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - Every dtype alias maps to its exact `(code, bits, lanes)` from `dlpack.h` and reports the right itemsize. diff --git a/development/work/2026-07-p02-layout/plan.md b/development/work/2026-07-p02-layout/plan.md index d010ee8..9786e6f 100644 --- a/development/work/2026-07-p02-layout/plan.md +++ b/development/work/2026-07-p02-layout/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 2 — layout policies & resolution -> **Context.** Step 02 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 2). Design sections: §3.6. This step depends on **p01-dtypes-device** landing first. +> **Context.** Step 02 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 2). Design sections: §3.6. This step depends on **p01-dtypes-device** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p02-layout/spec.md b/development/work/2026-07-p02-layout/spec.md index 035873f..f8b206f 100644 --- a/development/work/2026-07-p02-layout/spec.md +++ b/development/work/2026-07-p02-layout/spec.md @@ -1,6 +1,6 @@ # Phase 2 — layout policies & resolution -> **Context.** Step 02 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 2). Design sections: §3.6. This step depends on **p01-dtypes-device** landing first. +> **Context.** Step 02 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 2). Design sections: §3.6. This step depends on **p01-dtypes-device** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ There is no way to turn a (shape, dtype) into concrete strides and a byte size, `Layout`/`LayoutPolicy` and the shipped policies produce concrete, provably non-overlapping, in-bounds strides for any supported shape. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - For every shipped policy and small shape, all index offsets are distinct, non-negative, and within `required_nbytes` (exhaustive). diff --git a/development/work/2026-07-p03-stream-memory-resource/plan.md b/development/work/2026-07-p03-stream-memory-resource/plan.md index bd64054..24981c0 100644 --- a/development/work/2026-07-p03-stream-memory-resource/plan.md +++ b/development/work/2026-07-p03-stream-memory-resource/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 3 — stream, memory resource & recording fixture -> **Context.** Step 03 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 3). Design sections: §3.2, §3.3. This step depends on **p02-layout** landing first. +> **Context.** Step 03 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 3). Design sections: §3.2, §3.3. This step depends on **p02-layout** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p03-stream-memory-resource/spec.md b/development/work/2026-07-p03-stream-memory-resource/spec.md index 1626c4f..f748903 100644 --- a/development/work/2026-07-p03-stream-memory-resource/spec.md +++ b/development/work/2026-07-p03-stream-memory-resource/spec.md @@ -1,6 +1,6 @@ # Phase 3 — stream, memory resource & recording fixture -> **Context.** Step 03 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 3). Design sections: §3.2, §3.3. This step depends on **p02-layout** landing first. +> **Context.** Step 03 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 3). Design sections: §3.2, §3.3. This step depends on **p02-layout** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ There is no allocation interface and no test double to exercise allocator behavi The `DeviceMemoryResource` ABC, `Stream`/`CpuStream`, the adaptor stack, and a `RecordingMemoryResource` fixture exist and are self-tested. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - `RecordingMemoryResource` produces deterministic fake pointers and detects double-free, foreign-free and size-mismatch. diff --git a/development/work/2026-07-p04-cpu-mrs/plan.md b/development/work/2026-07-p04-cpu-mrs/plan.md index d203c72..669ab23 100644 --- a/development/work/2026-07-p04-cpu-mrs/plan.md +++ b/development/work/2026-07-p04-cpu-mrs/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 4 — CPU memory resources & conformance suite -> **Context.** Step 04 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 4). Design sections: §5.1. This step depends on **p03-stream-memory-resource** landing first. +> **Context.** Step 04 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 4). Design sections: §5.1. This step depends on **p03-stream-memory-resource** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p04-cpu-mrs/spec.md b/development/work/2026-07-p04-cpu-mrs/spec.md index 19aab90..88e018f 100644 --- a/development/work/2026-07-p04-cpu-mrs/spec.md +++ b/development/work/2026-07-p04-cpu-mrs/spec.md @@ -1,6 +1,6 @@ # Phase 4 — CPU memory resources & conformance suite -> **Context.** Step 04 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 4). Design sections: §5.1. This step depends on **p03-stream-memory-resource** landing first. +> **Context.** Step 04 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 4). Design sections: §5.1. This step depends on **p03-stream-memory-resource** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ No memory resource actually allocates real memory, so nothing can be written, re `BytearrayMemoryResource` and `MallocMemoryResource` allocate real, correctly aligned CPU memory and both pass one reusable MR conformance suite on every OS. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - A known pattern written into a returned pointer reads back byte-exact; adjacent allocations never alias. diff --git a/development/work/2026-07-p05-buffer-registry/plan.md b/development/work/2026-07-p05-buffer-registry/plan.md index 8108dbf..a0066c4 100644 --- a/development/work/2026-07-p05-buffer-registry/plan.md +++ b/development/work/2026-07-p05-buffer-registry/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 5 — buffer & registry -> **Context.** Step 05 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 5). Design sections: §3.4, §3.5. This step depends on **p04-cpu-mrs** landing first. +> **Context.** Step 05 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 5). Design sections: §3.4, §3.5. This step depends on **p04-cpu-mrs** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p05-buffer-registry/spec.md b/development/work/2026-07-p05-buffer-registry/spec.md index fdd31b8..fdf804c 100644 --- a/development/work/2026-07-p05-buffer-registry/spec.md +++ b/development/work/2026-07-p05-buffer-registry/spec.md @@ -1,6 +1,6 @@ # Phase 5 — buffer & registry -> **Context.** Step 05 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 5). Design sections: §3.4, §3.5. This step depends on **p04-cpu-mrs** landing first. +> **Context.** Step 05 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 5). Design sections: §3.4, §3.5. This step depends on **p04-cpu-mrs** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ Raw pointers from an MR have no ownership, lifetime safety net, or default-selec `DeviceBuffer` owns an allocation with a GC safety net and idempotent free, and the registry resolves a per-device current MR with contextvar scoping. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - `free()` deallocates exactly once on the allocation stream by default (explicit stream when given); a second `free()` is a no-op; use-after-free raises. diff --git a/development/work/2026-07-p06-dlpack-abi/plan.md b/development/work/2026-07-p06-dlpack-abi/plan.md index d2b5c39..050dad5 100644 --- a/development/work/2026-07-p06-dlpack-abi/plan.md +++ b/development/work/2026-07-p06-dlpack-abi/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 6 — DLPack ABI mirrors & compiled oracle -> **Context.** Step 06 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 6). Design sections: §7.1. This step depends on **p05-buffer-registry** landing first. +> **Context.** Step 06 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 6). Design sections: §7.1. This step depends on **p05-buffer-registry** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p06-dlpack-abi/spec.md b/development/work/2026-07-p06-dlpack-abi/spec.md index 8fcc961..96048fa 100644 --- a/development/work/2026-07-p06-dlpack-abi/spec.md +++ b/development/work/2026-07-p06-dlpack-abi/spec.md @@ -1,6 +1,6 @@ # Phase 6 — DLPack ABI mirrors & compiled oracle -> **Context.** Step 06 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 6). Design sections: §7.1. This step depends on **p05-buffer-registry** landing first. +> **Context.** Step 06 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 6). Design sections: §7.1. This step depends on **p05-buffer-registry** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ The exporter needs ctypes structs that match `dlpack.h` byte-for-byte; any silen ctypes mirrors of every DLPack struct provably match a compiled `dlpack.h` on T1 and committed per-platform snapshots on T0. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - On a machine with a C compiler, every `sizeof`/`offsetof` from a compiled `dlpack.h` matches the ctypes structs. diff --git a/development/work/2026-07-p07-export-tensor-empty/plan.md b/development/work/2026-07-p07-export-tensor-empty/plan.md index 25ba919..4bdf846 100644 --- a/development/work/2026-07-p07-export-tensor-empty/plan.md +++ b/development/work/2026-07-p07-export-tensor-empty/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 7 — export, Tensor & empty()/empty_like() (critical) -> **Context.** Step 07 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 7). Design sections: §3.8, §7. This step depends on **p06-dlpack-abi** landing first. +> **Context.** Step 07 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 7). Design sections: §3.8, §7. This step depends on **p06-dlpack-abi** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p07-export-tensor-empty/spec.md b/development/work/2026-07-p07-export-tensor-empty/spec.md index 4b408ab..df999f3 100644 --- a/development/work/2026-07-p07-export-tensor-empty/spec.md +++ b/development/work/2026-07-p07-export-tensor-empty/spec.md @@ -1,6 +1,6 @@ # Phase 7 — export, Tensor & empty()/empty_like() (critical) -> **Context.** Step 07 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 7). Design sections: §3.8, §7. This step depends on **p06-dlpack-abi** landing first. +> **Context.** Step 07 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 7). Design sections: §3.8, §7. This step depends on **p06-dlpack-abi** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ Nothing yet produces a DLPack capsule or a user-facing tensor; this is where mem `empty()`/`empty_like()` produce a `Tensor` that NumPy (and array-api-strict) consume zero-copy, with a deleter chain that never leaks or frees memory a consumer still holds. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - For hypothesis (shape, dtype, policy) on CPU, `np.from_dlpack(t)` matches dtype/shape/values; padded layouts give byte-correct strides and mutation round-trips (genuine zero-copy). diff --git a/development/work/2026-07-p08-runtimes-cpu/plan.md b/development/work/2026-07-p08-runtimes-cpu/plan.md index 945309b..50fbfea 100644 --- a/development/work/2026-07-p08-runtimes-cpu/plan.md +++ b/development/work/2026-07-p08-runtimes-cpu/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 8 — runtimes: base, discovery, CPU -> **Context.** Step 08 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 8). Design sections: §4. This step depends on **p07-export-tensor-empty** landing first. +> **Context.** Step 08 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 8). Design sections: §4. This step depends on **p07-export-tensor-empty** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p08-runtimes-cpu/spec.md b/development/work/2026-07-p08-runtimes-cpu/spec.md index 0b6fdc1..01995cf 100644 --- a/development/work/2026-07-p08-runtimes-cpu/spec.md +++ b/development/work/2026-07-p08-runtimes-cpu/spec.md @@ -1,6 +1,6 @@ # Phase 8 — runtimes: base, discovery, CPU -> **Context.** Step 08 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 8). Design sections: §4. This step depends on **p07-export-tensor-empty** landing first. +> **Context.** Step 08 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 8). Design sections: §4. This step depends on **p07-export-tensor-empty** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ The registry's default MR is a placeholder and there is no runtime discovery, so Runtime discovery + the CPU runtime exist; `empty(device="cpu")` with no explicit MR uses the runtime default, completing the CPU story. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - `runtime_names()` reports availability without importing heavyweight modules; `available_runtimes()` constructs only passing probes. diff --git a/development/work/2026-07-p09-cuda/plan.md b/development/work/2026-07-p09-cuda/plan.md index 43ab01d..cff6bc5 100644 --- a/development/work/2026-07-p09-cuda/plan.md +++ b/development/work/2026-07-p09-cuda/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 9 — CUDA runtime & MRs [GPU] -> **Context.** Step 09 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 9). Design sections: §4, §5.2. This step depends on **p08-runtimes-cpu** landing first. +> **Context.** Step 09 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 9). Design sections: §4, §5.2. This step depends on **p08-runtimes-cpu** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p09-cuda/spec.md b/development/work/2026-07-p09-cuda/spec.md index 8a02b44..2f2a005 100644 --- a/development/work/2026-07-p09-cuda/spec.md +++ b/development/work/2026-07-p09-cuda/spec.md @@ -1,6 +1,6 @@ # Phase 9 — CUDA runtime & MRs [GPU] -> **Context.** Step 09 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 9). Design sections: §4, §5.2. This step depends on **p08-runtimes-cpu** landing first. +> **Context.** Step 09 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 9). Design sections: §4, §5.2. This step depends on **p08-runtimes-cpu** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ devmm cannot allocate on NVIDIA GPUs, and GPU control-flow logic must be verifia A CUDA runtime and `CudaRuntimeMR`/`RmmMR` built on an injected `api`, with all control flow unit-tested on T0 and full round-trips on T2. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - On T0 with a `FakeCudartApi`: alloc/free call sequences are exact; `async_alloc="auto"` selects `cudaMallocAsync` iff supported; every nonzero status maps to the right exception; `make_stream_wait` emits create->record->wait->destroy; `activate_device` restores on exception. diff --git a/development/work/2026-07-p10-rocm/plan.md b/development/work/2026-07-p10-rocm/plan.md index 322bbaa..2c66521 100644 --- a/development/work/2026-07-p10-rocm/plan.md +++ b/development/work/2026-07-p10-rocm/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 10 — ROCm runtime & MRs [GPU] -> **Context.** Step 10 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 10). Design sections: §4.2, §5.3. This step depends on **p09-cuda** landing first. +> **Context.** Step 10 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 10). Design sections: §4.2, §5.3. This step depends on **p09-cuda** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p10-rocm/spec.md b/development/work/2026-07-p10-rocm/spec.md index 7fcb22e..b5fa6b6 100644 --- a/development/work/2026-07-p10-rocm/spec.md +++ b/development/work/2026-07-p10-rocm/spec.md @@ -1,6 +1,6 @@ # Phase 10 — ROCm runtime & MRs [GPU] -> **Context.** Step 10 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 10). Design sections: §4.2, §5.3. This step depends on **p09-cuda** landing first. +> **Context.** Step 10 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 10). Design sections: §4.2, §5.3. This step depends on **p09-cuda** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ devmm cannot allocate on AMD GPUs, and the CUDA/ROCm FFI logic should not be dup A shared FFI shim underpins both platforms; `HipRuntimeMR`/`HipmmMR` work and the platform probe disambiguates hipMM's `rmm` module. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - The Phase-9 fake-api suite, re-parametrized for HIP symbol names/error tables, passes on T0 (the shared shim pays off). diff --git a/development/work/2026-07-p11-integrations/plan.md b/development/work/2026-07-p11-integrations/plan.md index a285c5b..75566e8 100644 --- a/development/work/2026-07-p11-integrations/plan.md +++ b/development/work/2026-07-p11-integrations/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 11 — third-party MRs & integrations -> **Context.** Step 11 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 11). Design sections: §5.2-§5.4, §6. This step depends on **p10-rocm** landing first. +> **Context.** Step 11 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 11). Design sections: §5.2-§5.4, §6. This step depends on **p10-rocm** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p11-integrations/spec.md b/development/work/2026-07-p11-integrations/spec.md index efaeca4..b3d0136 100644 --- a/development/work/2026-07-p11-integrations/spec.md +++ b/development/work/2026-07-p11-integrations/spec.md @@ -1,6 +1,6 @@ # Phase 11 — third-party MRs & integrations -> **Context.** Step 11 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 11). Design sections: §5.2-§5.4, §6. This step depends on **p10-rocm** landing first. +> **Context.** Step 11 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 11). Design sections: §5.2-§5.4, §6. This step depends on **p10-rocm** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ devmm cannot yet consume ecosystem allocators as MRs nor install a devmm MR into The consume + provide arrows for CuPy, Numba, NumPy and rmm work, each `install()` is reversible, and composing both directions for one library is refused. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - Composing a consume+provide pair for the same library raises; every `install()` returns an uninstaller/context manager that restores prior state, including on exception. diff --git a/development/work/2026-07-p12-conformance-docs-release/plan.md b/development/work/2026-07-p12-conformance-docs-release/plan.md index a40aa8b..c04d357 100644 --- a/development/work/2026-07-p12-conformance-docs-release/plan.md +++ b/development/work/2026-07-p12-conformance-docs-release/plan.md @@ -1,6 +1,6 @@ # Plan — Phase 12 — public conformance, docs & release gate -> **Context.** Step 12 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 12). Design sections: §9, §10. This step depends on **p11-integrations** landing first. +> **Context.** Step 12 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 12). Design sections: §9, §10. This step depends on **p11-integrations** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Architecture decisions diff --git a/development/work/2026-07-p12-conformance-docs-release/spec.md b/development/work/2026-07-p12-conformance-docs-release/spec.md index 1a2a24a..56f66e9 100644 --- a/development/work/2026-07-p12-conformance-docs-release/spec.md +++ b/development/work/2026-07-p12-conformance-docs-release/spec.md @@ -1,6 +1,6 @@ # Phase 12 — public conformance, docs & release gate -> **Context.** Step 12 of the devmm v0.1 build. Authoritative spec: [`work/devmm-design.md`](../devmm-design.md); build order: [`work/devmm-implementation-plan.md`](../devmm-implementation-plan.md) (Phase 12). Design sections: §9, §10. This step depends on **p11-integrations** landing first. +> **Context.** Step 12 of the devmm v0.1 build. Authoritative spec: [`development/devmm-design.md`](../../devmm-design.md); build order: [`development/devmm-implementation-plan.md`](../../devmm-implementation-plan.md) (Phase 12). Design sections: §9, §10. This step depends on **p11-integrations** landing first. > The overall v0.1 goals, non-goals and release gate live in the design doc and implementation plan; this folder is scoped to a single phase. ## Problem @@ -10,7 +10,7 @@ Third-party MR/runtime authors have no supported way to prove conformance, the l The conformance suites are public, docs are executed as tests, and a release gate maps every design contract to a test before tagging 0.1.0. ## Users & stakeholders -Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`work/devmm-design.md`](../devmm-design.md). +Python developers building on devmm (this phase's consumers are the later phases and the test suite) and the grAItools maintainers who sign off against [`development/devmm-design.md`](../../devmm-design.md). ## Success criteria - `devmm.testing.mr_conformance(mr_factory)` and `devmm.testing.dlpack_conformance(device)` are public, documented, and self-tested. diff --git a/development/work/2026-07-p12-conformance-docs-release/tasks.md b/development/work/2026-07-p12-conformance-docs-release/tasks.md index ca5733c..56f38bb 100644 --- a/development/work/2026-07-p12-conformance-docs-release/tasks.md +++ b/development/work/2026-07-p12-conformance-docs-release/tasks.md @@ -19,5 +19,5 @@ one is verified. ## Gate & handoff - [x] Update the public-API snapshot: + `testing.mr_conformance`, `testing.dlpack_conformance` (public). - [x] Update `tests/traceability.md` for the contracts covered here. -- [x] Phase gate green (Release gate green; tag `v0.1.0`.). T0 legs green locally (`make release-gate`); T2/T3 run under the waiver in `docs/adr/0003-gpu-suite-waiver-for-0.1.0.md` (needs maintainer sign-off); tagging `v0.1.0` is the orchestrator's step. +- [x] Phase gate green (Release gate green; tag `v0.1.0`.). T0 legs green locally (`make release-gate`); T2/T3 run under the waiver in `development/adr/0003-gpu-suite-waiver-for-0.1.0.md` (needs maintainer sign-off); tagging `v0.1.0` is the orchestrator's step. - [x] Hand off to `/verify` (Reviewer) before the next step. diff --git a/pyproject.toml b/pyproject.toml index 646ec11..61d0d1b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -45,7 +45,7 @@ test = ["numpy", "array-api-strict"] # so installing it from here would pull a PyPI build and a conflicting nvidia-* # set that step 2 of the recipe then replaces. Install it separately, and on a # host with a newer system CUDA unset `CUDA_HOME` so Numba links these wheels — -# see docs/testing.md for the full recipe. +# see development/testing.md for the full recipe. gpu-test-cuda = [ "devmm[cuda]; python_full_version < '3.15'", "cupy-cuda12x; python_full_version < '3.15'", diff --git a/src/devmm/__init__.py b/src/devmm/__init__.py index 59b847e..b9bc2fd 100644 --- a/src/devmm/__init__.py +++ b/src/devmm/__init__.py @@ -2,7 +2,7 @@ A uniform, pure-Python interface for allocating and managing device memory across CPU/CUDA/ROCm, exposing allocations as DLPack >= 1.0 producers. -See work/devmm-design.md for the full architecture. +See development/devmm-design.md for the full architecture. This module holds the public re-exports only (design §2): the factories (`empty`, `empty_like`), the domain model (`Device`, `Stream`, `Tensor`, diff --git a/src/devmm/_runtimes/_hostcopy.py b/src/devmm/_runtimes/_hostcopy.py index 0847f96..73e46ab 100644 --- a/src/devmm/_runtimes/_hostcopy.py +++ b/src/devmm/_runtimes/_hostcopy.py @@ -2,7 +2,7 @@ The helpers stage host bytes in ctypes storage and move them through the runtime's `memcpy` primitive (design §3.5, §4.1). The ctypes lives here so -the core domain model imports none of it (docs/style.md). +the core domain model imports none of it (development/style.md). """ from __future__ import annotations diff --git a/tests/test_import_hygiene.py b/tests/test_import_hygiene.py index 10e794f..71b6b84 100644 --- a/tests/test_import_hygiene.py +++ b/tests/test_import_hygiene.py @@ -1,5 +1,5 @@ """Import hygiene: the core dtype/device modules duck-type NumPy and must not -pull it into `sys.modules` on import (design §3.7; docs/style.md, "never +pull it into `sys.modules` on import (design §3.7; development/style.md, "never import numpy in core"). Checked in a subprocess so the parent test session's own imports cannot mask a violation. """ diff --git a/tests/test_public_api.py b/tests/test_public_api.py index c4dc134..2c0c098 100644 --- a/tests/test_public_api.py +++ b/tests/test_public_api.py @@ -5,7 +5,7 @@ memory resources and bridges from — to its call signature (for callables and classes) or its type name (for everything else). Growing or changing the surface requires a matching change in the design doc -(`work/devmm-design.md`) before a snapshot is updated. +(`development/devmm-design.md`) before a snapshot is updated. """ from __future__ import annotations diff --git a/tests/traceability.md b/tests/traceability.md index 747110d..839da41 100644 --- a/tests/traceability.md +++ b/tests/traceability.md @@ -1,7 +1,7 @@ # Traceability — contracts to tests Maps enforced contracts to the tests that pin them. One row per contract; -extend this table as phases land (see `work/devmm-implementation-plan.md`). +extend this table as phases land (see `development/devmm-implementation-plan.md`). | Contract | Source | Test | | --- | --- | --- | @@ -12,14 +12,14 @@ extend this table as phases land (see `work/devmm-implementation-plan.md`). | Every dtype alias reports the right itemsize | design §3.7 | `tests/test_dtypes.py::test_alias_itemsize` | | `DType.from_string` accepts Array-API dtype strings, rejects everything else | design §3.7 | `tests/test_dtypes.py::test_from_string_returns_alias`, `tests/test_dtypes.py::test_from_string_rejects_unknown` | | `DType.from_any` duck-types NumPy dtypes via `.kind`/`.itemsize`; unsupported kinds/objects raise | design §3.7 | `tests/test_dtypes.py::test_from_any_duck_typed_numpy_like`, `tests/test_dtypes.py::test_from_any_rejects_unsupported_kind`, `tests/test_dtypes.py::test_from_any_rejects_non_dtype_objects` | -| `DType` is frozen and hashes/compares by value | docs/style.md, value objects | `tests/test_dtypes.py::test_dtype_is_frozen`, `tests/test_dtypes.py::test_dtype_is_hashable_and_equal_by_value` | +| `DType` is frozen and hashes/compares by value | development/style.md, value objects | `tests/test_dtypes.py::test_dtype_is_frozen`, `tests/test_dtypes.py::test_dtype_is_hashable_and_equal_by_value` | | NumPy differential: itemsize equality and `np.dtype` round-trip for every alias with a counterpart | design §3.7, §9 | `tests/test_dtypes_numpy.py::test_itemsize_matches_numpy`, `tests/test_dtypes_numpy.py::test_numpy_dtype_round_trips_to_alias` | | `DeviceType` values are the exact DLPack `DLDeviceType` codes | design §3.1 | `tests/test_device.py::test_device_type_codes_match_dlpack` | | `Device.from_string` parses well-formed strings; bare name means index 0 | design §3.1 | `tests/test_device.py::test_from_string_parses_valid`, `tests/test_device.py::test_from_string_bare_name_defaults_to_index_zero` | | `Device.from_string` rejects malformed strings with `ValueError` | design §3.1 | `tests/test_device.py::test_from_string_rejects_malformed`, `tests/test_device.py::test_from_string_rejects_malformed_examples` | -| `Device` round-trips through `str()`; frozen, hashable, equal by value | design §3.1; docs/style.md | `tests/test_device.py::test_from_string_format_round_trip`, `tests/test_device.py::test_device_hash_and_equality`, `tests/test_device.py::test_device_is_frozen` | +| `Device` round-trips through `str()`; frozen, hashable, equal by value | design §3.1; development/style.md | `tests/test_device.py::test_from_string_format_round_trip`, `tests/test_device.py::test_device_hash_and_equality`, `tests/test_device.py::test_device_is_frozen` | | `Device.__dlpack_device__() == (int(type), index)` | design §3.1 | `tests/test_device.py::test_dlpack_device_is_code_index_pair` | -| Importing devmm core never pulls `numpy` into `sys.modules` | design §3.7; docs/style.md | `tests/test_import_hygiene.py::test_core_import_does_not_import_numpy` | +| Importing devmm core never pulls `numpy` into `sys.modules` | design §3.7; development/style.md | `tests/test_import_hygiene.py::test_core_import_does_not_import_numpy` | | Zero runtime dependencies: wheel installs and imports in a bare venv | design §8 | `tests/test_packaging.py::test_wheel_imports_in_bare_venv` | | Wheel metadata declares no unconditional `Requires-Dist` | design §8 | `tests/test_packaging.py::test_wheel_declares_no_runtime_dependencies` | | Wheel ships only `devmm/` (incl. `py.typed`), never `tests/` | plan p00, Risks | `tests/test_packaging.py::test_wheel_packages_only_devmm` | @@ -28,7 +28,7 @@ extend this table as phases land (see `work/devmm-implementation-plan.md`). | `Aligned` line pitch is `unit_stride_alignment`-divisible with minimal padding; `required_nbytes % base_alignment == 0` | design §3.6 | `tests/test_layout.py::test_aligned_pitch_divisible_and_padding_minimal` | | `layout.base_alignment <= policy.base_alignment` (and pitch-padding analogue); `layout.policy is policy` | design §3.6 | `tests/test_layout.py::test_layout_alignment_bounded_by_policy_and_provenance` | | `DeviceOptimal` dispatches host- vs GPU-resident alignments within its declared bounds | design §3.6 | `tests/test_layout.py::test_device_optimal_dispatches_per_device`, `tests/test_layout.py::test_device_optimal_treats_host_resident_memory_as_cpu` | -| Policies and layouts are frozen, hashable dict keys, equal by value | design §3.6; docs/style.md | `tests/test_layout.py::test_field_backed_policies_are_frozen`, `tests/test_layout.py::test_stateless_policies_reject_attribute_injection`, `tests/test_layout.py::test_layout_is_frozen`, `tests/test_layout.py::test_policies_are_hashable_dict_keys`, `tests/test_layout.py::test_layouts_are_hashable_dict_keys` | +| Policies and layouts are frozen, hashable dict keys, equal by value | design §3.6; development/style.md | `tests/test_layout.py::test_field_backed_policies_are_frozen`, `tests/test_layout.py::test_stateless_policies_reject_attribute_injection`, `tests/test_layout.py::test_layout_is_frozen`, `tests/test_layout.py::test_policies_are_hashable_dict_keys`, `tests/test_layout.py::test_layouts_are_hashable_dict_keys` | | `Layout.validate()` rejects overlapping/negative/zero/non-derivable strides, rank mismatches, undersized or non-integer `required_nbytes`; accepts sound hand-built layouts | design §3.6 | `tests/test_layout.py::test_validate_rejects`, `tests/test_layout.py::test_validate_accepts_hand_built_padded_f_order` | | `is_contiguous` reports dense offset envelopes; gapped hand-built strides are not contiguous | design §3.6 | `tests/test_layout.py::test_policy_layouts_are_dense_over_their_envelope`, `tests/test_layout.py::test_padded_layouts_are_dense_over_their_padded_envelope`, `tests/test_layout.py::test_hand_built_gapped_layouts_are_not_contiguous` | | Layout edge cases: ndim=0, zero/one extents, huge extents with exact integer-only `required_nbytes`/strides | design §3.6 | `tests/test_layout.py::test_scalar_layout_has_empty_strides_and_one_element`, `tests/test_layout.py::test_zero_extent_shapes_need_no_bytes`, `tests/test_layout.py::test_extent_one_dims_keep_cumulative_strides`, `tests/test_layout.py::test_huge_extents_stay_exact_ints` | @@ -98,14 +98,14 @@ extend this table as phases land (see `work/devmm-implementation-plan.md`). | `runtime_names()` probes without importing heavyweight modules (subprocess); `available_runtimes()` constructs only probe-passing runtimes | design §4.1; spec p08 | `tests/test_runtimes.py::TestDiscovery::test_runtime_names_probes_without_loading_heavy_modules`, `tests/test_runtimes.py::TestDiscovery::test_available_runtimes_constructs_only_passing_probes`, `tests/test_runtimes.py::TestDiscovery::test_available_runtimes_loads_the_cpu_runtime` | | The CPU runtime is always discovered; `runtime_for` accepts `Device`/`DeviceType`/str, caches the loaded runtime, and raises `RuntimeUnavailableError` with an actionable message for unsupported device types | design §4.1; spec p08 | `tests/test_runtimes.py::TestDiscovery::test_cpu_runtime_is_always_discovered`, `tests/test_runtimes.py::TestDiscovery::test_runtime_for_accepts_device_devicetype_and_string`, `tests/test_runtimes.py::TestDiscovery::test_runtime_for_caches_the_loaded_runtime`, `tests/test_runtimes.py::TestDiscovery::test_runtime_for_unsupported_device_raises_actionably` | | Entry-point runtimes (`devmm.runtimes` group) are discovered after built-ins, loadable, and shadowed by same-named built-ins | design §4.1; spec p08 | `tests/test_runtimes.py::TestEntryPoints::test_entry_point_runtime_is_discovered_after_builtins`, `tests/test_runtimes.py::TestEntryPoints::test_entry_point_runtime_is_loadable`, `tests/test_runtimes.py::TestEntryPoints::test_builtin_names_shadow_entry_points` | -| Discovery scans entry points at most once per process (spec table cached; hot paths never pay a metadata re-scan) | design §4.1; docs/testing.md, determinism | `tests/test_runtimes.py::TestDiscovery::test_discovery_scans_entry_points_at_most_once` | +| Discovery scans entry points at most once per process (spec table cached; hot paths never pay a metadata re-scan) | design §4.1; development/testing.md, determinism | `tests/test_runtimes.py::TestDiscovery::test_discovery_scans_entry_points_at_most_once` | | A runtime whose loader raises `RuntimeUnavailableError` is skipped: `available_runtimes()` returns the rest, and `runtime_for` for other device types still resolves (or raises the actionable no-runtime message, never the broken loader's error) | design §4.1 | `tests/test_runtimes.py::TestEntryPoints::test_available_runtimes_skips_runtimes_whose_loader_fails`, `tests/test_runtimes.py::TestEntryPoints::test_runtime_for_is_not_poisoned_by_an_unrelated_failing_loader` | | `DEVMM_RUNTIME` forces the named runtime (probe skipped, other runtimes excluded); a bogus value raises `RuntimeUnavailableError` naming the variable and the registered runtimes | design §4.2; spec p08 | `tests/test_runtimes.py::TestEnvOverride::test_forces_the_named_runtime`, `tests/test_runtimes.py::TestEnvOverride::test_forcing_skips_the_probe`, `tests/test_runtimes.py::TestEnvOverride::test_bogus_value_raises_with_the_registered_names`, `tests/test_runtimes.py::TestEnvOverride::test_bogus_value_fails_runtime_for_too`, `tests/test_runtimes.py::TestEnvOverride::test_override_excludes_other_runtimes` | | CPU runtime SPI: identity, `device_count`, `MallocMemoryResource` default MR, `CpuStream` factories, `wrap_stream` contract (pass-through, null handle, `__cuda_stream__`, rejections), HOST_TO_HOST-only `memcpy`, no-op `make_stream_wait`/`activate_device`, non-CPU devices rejected | design §4.1, §5.1; spec p08 | `tests/test_runtimes.py::TestCpuRuntime` (all tests) | | `CopyKind` values are the CUDA/HIP `cudaMemcpyKind` codes verbatim | design §4.1 | `tests/test_runtimes.py::TestCpuRuntime::test_copy_kind_values_are_the_cuda_hip_codes` | | `empty()` with no explicit MR uses the runtime default on CPU, and the round-trip suite re-runs through that path | design §3.4, §3.8, §4.1; spec p08 | `tests/test_runtimes.py::TestRuntimeDefaultPath::test_empty_with_no_mr_uses_the_runtime_default`, `tests/test_runtimes.py::TestRuntimeDefaultPath::test_default_path_round_trips_through_numpy`, `tests/test_dlpack_roundtrip.py::test_numpy_round_trip_matches_dtype_shape_and_values` (`runtime-default` case) | | `__dlpack__` consumer-stream handoff routes through `runtime.make_stream_wait` when a runtime serves the device, falling back to a producer-stream synchronize when none does | design §7.3, §4.1; spec p08 | `tests/test_runtimes.py::TestEntryPoints::test_dlpack_handoff_routes_through_the_runtime`, `tests/test_dlpack_refusals.py::test_consumer_stream_handoff_orders_against_the_producer_stream`, `tests/test_dlpack_refusals.py::test_none_stream_means_the_legacy_default_and_orders` | -| `DeviceBuffer` host copies route through the runtime's `memcpy` primitive (`_core/buffer.py` imports no ctypes) | design §3.5, §4.1; docs/style.md | `tests/test_buffer.py::TestHostCopies` (all tests), `tests/test_runtimes.py::TestCpuRuntime::test_memcpy_moves_host_bytes` | +| `DeviceBuffer` host copies route through the runtime's `memcpy` primitive (`_core/buffer.py` imports no ctypes) | design §3.5, §4.1; development/style.md | `tests/test_buffer.py::TestHostCopies` (all tests), `tests/test_runtimes.py::TestCpuRuntime::test_memcpy_moves_host_bytes` | | Public API snapshot grows `available_runtimes`, `runtime_names`, `runtime_for` | plan p08, Public-API snapshot | `tests/test_public_api.py::test_all_matches_snapshot`, `tests/test_public_api.py::test_exported_member_signatures_match_snapshot` | | GPU discovery keys off the platform driver library (CUDA: nvcuda/libcuda; ROCm: libamdhip64 — never the ambiguous `rmm` module name), probe only, no runtime-library load; specs register in cpu < cuda < rocm order; a passing probe with an unloadable runtime library is skipped without poisoning other runtimes; `DEVMM_RUNTIME=cuda|rocm` forces the named runtime | design §4.1, §4.2; spec p09/p10 | `tests/test_gpu_runtime.py::TestGpuDiscovery::test_specs_are_registered_in_platform_order`, `tests/test_gpu_runtime.py::TestGpuDiscovery::test_probe_passes_when_the_platform_driver_loads`, `tests/test_gpu_runtime.py::TestGpuDiscovery::test_probe_fails_when_no_driver_library_loads`, `tests/test_gpu_runtime.py::TestGpuDiscovery::test_unloadable_runtime_is_skipped_with_an_actionable_message`, `tests/test_gpu_runtime.py::TestGpuDiscovery::test_env_override_forces_the_runtime`, `tests/test_gpu_runtime.py::TestGpuDiscovery::test_loaded_runtime_resolves_platform_devices`, `tests/test_runtimes.py::TestDiscovery::test_runtime_names_probes_without_loading_heavy_modules` | | `runtime_for`'s failure message labels the spec list "registered:" (a listed runtime may still fail to load) | design §4.1; spec p09 | `tests/test_gpu_runtime.py::TestGpuDiscovery::test_runtime_for_failure_labels_the_spec_list_registered` | From e245a65d24adbc519959cb6890178a5ba4b52b07 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Enrique=20Gonz=C3=A1lez=20Paredes?= <enriqueg@cscs.ch> Date: Sun, 26 Jul 2026 17:00:58 +0200 Subject: [PATCH 03/10] fix(harness): explain every Bash-hook refusal and repair the register contract MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review of #3 found the hardened PreToolUse guard bricks a session on any host without jq: it fails closed, jq is not installed by the bootstrap, and the denial covers the `apt-get install jq` that would fix it. Verified — with jq off PATH, `make verify` returns exit 2. The pre-update wiring failed *open* instead, allowing destructive commands through unchecked, so neither form was right. Keep fail-closed and make it self-explanatory: - The guard checks for jq up front and names it as the reason, with the install command; the other two deny paths explain themselves too. - block-destructive.sh reports which deny-list pattern matched, instead of a bare exit 2 that reads as an unexplained refusal and invites a reword-retry loop. - ensure-toolchain.sh warns at SessionStart when jq is absent, so the problem surfaces before the first Bash call rather than as a mystery denial. Warning only — it must not abort the uv bootstrap. - tool-bootstrap.md states jq is required, not merely standard. The decision-register contract was self-defeating: it keyed on the literal `DECISION-PENDING:` text, which this PR itself adds 15 times as documentation, so `/verify` would have raised a dozen fabricated MAJOR defects while a real escalation hid among the quotes. A marker is now defined by location — its own line inside a development/work/*/report.md — with the scan command to match. Also: register rows sourced from an ADR or a human grant are marker-less by construction and no longer read as out-of-scope (both seeded rows are of that kind); the Developer is named owner of the paired row, the only role that can write it; and a PR with no work unit has the spec/plan/register axes marked n/a rather than failed. Remaining: gate-output rules now point at development/testing.md, which is authoritative and records the sanctioned DEVMM_GPU skips; the Developer gets the Architect's scratch.md clobber guard; development/README.md drops a scaffold-marker section describing artifacts this repo no longer has and no longer claims development/ is unpublished (the sdist ships it, the wheel does not); README.md uses absolute links, as it is also the PyPI project page. Gate: make verify green — 900 passed, 75 skipped, unchanged. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- .agents/hooks/block-destructive.sh | 10 +++++++++- .agents/hooks/ensure-toolchain.sh | 8 ++++++++ .agents/subagents/architect.md | 4 +++- .agents/subagents/developer.md | 28 ++++++++++++++++++++++------ .agents/subagents/reviewer.md | 22 ++++++++++++++++++---- .claude/settings.json | 2 +- CHANGELOG.md | 19 ++++++++++++++++--- README.md | 8 ++++++-- development/README.md | 29 +++++++++++++++++------------ development/adr/README.md | 24 +++++++++++++++++++----- development/tool-bootstrap.md | 9 +++++++-- 11 files changed, 126 insertions(+), 37 deletions(-) diff --git a/.agents/hooks/block-destructive.sh b/.agents/hooks/block-destructive.sh index 474c59d..d671a0c 100755 --- a/.agents/hooks/block-destructive.sh +++ b/.agents/hooks/block-destructive.sh @@ -11,4 +11,12 @@ # .opencode/opencode.jsonc restate these patterns by hand — keep in sync. # # See .agents/README.md for the single-source-of-truth rationale. -grep -qE 'rm -rf|push --force|reset --hard|DROP TABLE' && exit 2 || exit 0 +# The matched pattern goes to stderr: a bare `exit 2` reaches the agent as an +# unexplained refusal, which it answers by rewording and retrying the same +# command. Naming the pattern makes the refusal actionable in one turn. +match=$(grep -oE 'rm -rf|push --force|reset --hard|DROP TABLE' | head -n 1) +if [ -n "$match" ]; then + echo "block-destructive: refused — matches the deny-list pattern '$match'." >&2 + exit 2 +fi +exit 0 diff --git a/.agents/hooks/ensure-toolchain.sh b/.agents/hooks/ensure-toolchain.sh index 743bf86..17341b3 100755 --- a/.agents/hooks/ensure-toolchain.sh +++ b/.agents/hooks/ensure-toolchain.sh @@ -18,6 +18,14 @@ # # See .agents/README.md for the single-source-of-truth rationale. +# jq is a hard requirement of the PreToolUse Bash guard, not a convenience: it +# parses the tool input, and the guard fails closed, so without jq every Bash +# call is denied — including the one that would install it. Report that here, at +# session start, instead of letting the first tool call fail opaquely. Warn only: +# a missing jq must not abort the uv bootstrap below. +command -v jq >/dev/null 2>&1 || \ + echo "ensure-toolchain: WARNING — jq not found. The PreToolUse Bash guard denies every command without it; install jq (apt-get install jq / brew install jq). See development/tool-bootstrap.md." >&2 + command -v uv >/dev/null 2>&1 && exit 0 # A prior install may sit in $HOME/.local/bin without being on this shell's PATH yet. diff --git a/.agents/subagents/architect.md b/.agents/subagents/architect.md index bbb2004..b2772a2 100644 --- a/.agents/subagents/architect.md +++ b/.agents/subagents/architect.md @@ -77,7 +77,9 @@ Produce `plan.md` and mirror it into a checkbox `tasks.md` in the same - `scratch.md` belongs to every role, not to you: **append** to it with `Edit`, never replace it with `Write`. Overwriting it destroys spike findings and Developer hand-back notes, and it is gitignored, so what - you clobber is gone. + you clobber is gone. Use `Write` only to create it when it does not + exist yet — the file is nobody's deliverable, so whoever needs it + first makes it. ## Output format diff --git a/.agents/subagents/developer.md b/.agents/subagents/developer.md index d6e9dbc..e00e581 100644 --- a/.agents/subagents/developer.md +++ b/.agents/subagents/developer.md @@ -47,7 +47,14 @@ phase boundary. - When a decision exceeds your authority — loosening a test tolerance or assertion, adding a dependency, changing behaviour the spec froze — write a `DECISION-PENDING:` line in `report.md` and stop. Do not - resolve it locally. + resolve it locally. **You also own its register row**: append the + matching `pending` row to the table in `development/adr/README.md` in + the same change as the marker, then stop. You are the only role that + can — the Reviewer is read-only and the Architect and Product Owner may + not write outside the feature directory — and the Reviewer counts a + marker with no row as a defect. Appending that one row is the sole + sanctioned exception to leaving `development/` alone; flip its Status + once the human answers. - Work **one phase at a time**. Do not begin phase N+1 until phase N's tests pass and its `tasks.md` boxes are ticked. - Write the test **first** when the plan calls for behaviour change — @@ -62,15 +69,24 @@ phase boundary. - Never silently skip, disable, or `@ignore` a failing test. If a test must be skipped, draft an ADR under `development/adr/` and ask before proceeding. -- When reading gate output: passed counts may only grow. A drop you - cannot attribute to your own intentional, declared test removal means - stop, don't commit, and report the failure verbatim. Never add +- When reading gate output, follow + [`development/testing.md`](../../development/testing.md#reading-gate-output) — + it is authoritative and names this project's sanctioned skips (the + `DEVMM_GPU` suites skip by design off hardware, so they are not + findings). In short: passed counts may only grow; a drop you cannot + attribute to your own intentional, declared test removal means stop, + don't commit, and report the failure verbatim; never add `-x`/`-k`/`--ignore`-style narrowing, or edit markers or assertions, - to make the gate pass. Any new skip is a finding to explain in - `report.md`, not to ignore. + to make the gate pass; any *new* skip is a finding to explain in + `report.md`. - Never edit anything under `*/generated/`. - Never run destructive Git (`push --force`, `reset --hard origin/*`, history rewrites on shared branches). +- `scratch.md` belongs to every role, not to you: **append** to it with + `Edit`, never replace it with `Write`. Overwriting it destroys spike + findings and Architect hand-back notes, and it is gitignored, so what + you clobber is gone. Use `Write` only to create it when it does not + exist yet. - If you discover the plan is wrong or missing a phase, stop and hand back to the Architect with a 2-3 sentence note in `scratch.md`. Do not silently re-plan. diff --git a/.agents/subagents/reviewer.md b/.agents/subagents/reviewer.md index 04ed630..d51b76f 100644 --- a/.agents/subagents/reviewer.md +++ b/.agents/subagents/reviewer.md @@ -61,11 +61,18 @@ top of the verdict rules below, which it never overrides. `report.md`, or the body of an existing ADR is an automatic MAJOR defect unless the plan explicitly called for it. The two registers have their own scope contracts instead: a decision-register row must - pair with a `DECISION-PENDING:` line in this diff, and a new + pair with an escalation marker in this diff *unless* its Source is an + ADR or a direct human grant (both are sanctioned marker-less rows — + see `development/adr/README.md`), and a new `development/glossary.md` entry must match a term in this feature's reviewed spec **Glossary** section — a glossary edit with no matching spec term, or any rename or meaning change of an existing entry, is out of scope mid-feature (MAJOR). +- **A change with no work unit** — a harness or tooling PR with no + `development/work/<slug>/` directory — is reviewed on the axes that + apply (quality, gate, honesty of the PR description). The spec- and + plan-conformance axes and both register contracts are `n/a`, not + failures; do not manufacture defects from their absence. - **Never trust the Developer's narrative.** Re-run the gate yourself; re-derive every claim you can check (line numbers, test names, measured values) instead of quoting the hand-off summary. @@ -108,12 +115,19 @@ top of the verdict rules below, which it never overrides. 4. **Report honesty** — `report.md` must match the code. Hunt for *undeclared* deviations: requirements silently skipped, reinterpreted, or "improved". An inaccurate report claim is a - defect even when the code is correct. Every `DECISION-PENDING:` - line **added by this diff** must come with a row in the decision + defect even when the code is correct. Every escalation marker + **added by this diff** must come with a row in the decision register (`development/adr/README.md`) in the same change; markers in already-merged reports are history, judged by the register alone. + A marker counts only on its own line inside a + `development/work/*/report.md` — the colon form quoted in these + instructions, `AGENTS.md`, the PR template or the harness guide is + documentation, so check report files, not the whole tree. - Run the project's verification gate. A failing gate is an automatic - NEEDS-WORK with the failure cited verbatim. When reading gate output: + NEEDS-WORK with the failure cited verbatim. Read the output per + [`development/testing.md`](../../development/testing.md#reading-gate-output), + which is authoritative and names this project's sanctioned skips (the + `DEVMM_GPU` suites skip by design off hardware — not defects). In short: passed counts may only grow (a drop the diff doesn't explain is a defect); any *new* skip is a defect to explain, not to ignore; any `-x`/`-k`/`--ignore`-style narrowing or marker/assertion edit that diff --git a/.claude/settings.json b/.claude/settings.json index 49387a4..ee8865c 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -46,7 +46,7 @@ "hooks": [ { "type": "command", - "command": "S=\"${CLAUDE_PROJECT_DIR:-.}/.agents/hooks/block-destructive.sh\"; [ -r \"$S\" ] || exit 2; c=$(jq -r '.tool_input.command // empty') || exit 2; [ -n \"$c\" ] || exit 2; printf '%s' \"$c\" | sh \"$S\"" + "command": "S=\"${CLAUDE_PROJECT_DIR:-.}/.agents/hooks/block-destructive.sh\"; [ -r \"$S\" ] || { echo \"PreToolUse: $S missing or unreadable — denying Bash.\" >&2; exit 2; }; command -v jq >/dev/null 2>&1 || { echo 'PreToolUse: jq not found, so the destructive-command guard cannot read the tool input — denying Bash. Install jq (apt-get install jq / brew install jq) from a shell outside the agent, then retry.' >&2; exit 2; }; c=$(jq -r '.tool_input.command // empty') || { echo 'PreToolUse: jq could not parse the hook payload — denying Bash.' >&2; exit 2; }; [ -n \"$c\" ] || { echo 'PreToolUse: hook payload carried no .tool_input.command — denying Bash.' >&2; exit 2; }; printf '%s' \"$c\" | sh \"$S\"" } ] } diff --git a/CHANGELOG.md b/CHANGELOG.md index b01b8d6..08f89b8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -17,11 +17,24 @@ and this project adheres to ### Changed - Repository layout: the agent-facing docs, ADRs and per-feature work units moved - from `docs/` and `work/` into `development/`, and `docs/` is now reserved for - user documentation (`docs/api.md`). Harness updated to template v0.6.0, which - adds `report.md` per work unit, a decision register in + from `docs/` and `work/` into `development/`; `docs/` is now reserved for user + documentation (`docs/api.md`). +- `README.md`: links to in-repo files are absolute, so they resolve on the PyPI + project page as well as on GitHub. +- Agent harness updated to copier template v0.6.0: adds `report.md` per work + unit, a decision register in [`development/adr/README.md`](development/adr/README.md), a [glossary](development/glossary.md), and role playbook skills. +- `development/tool-bootstrap.md`: `jq` is documented as required (not merely + standard) when working through Claude Code — the `PreToolUse` Bash guard + parses tool input with it and fails closed. + +### Fixed + +- `.agents/hooks/block-destructive.sh` names the deny-list pattern it matched on + stderr, and the `PreToolUse` Bash guard explains each refusal instead of + exiting 2 silently — a missing `jq` previously denied every Bash call, giving + no indication why and no way to recover in-session. ### Fixed diff --git a/README.md b/README.md index c4c0f33..f8ddf76 100644 --- a/README.md +++ b/README.md @@ -7,8 +7,12 @@ allocation strategies, and exposes every allocation as a zero-copy **DLPack ≥ 1.0** producer consumable by any Array-API library (NumPy, CuPy, PyTorch, JAX). Zero required dependencies; `ctypes` only. -The authoritative design is [`development/devmm-design.md`](development/devmm-design.md); -the API reference with executable examples is [`docs/api.md`](docs/api.md). +The authoritative design is +[`development/devmm-design.md`](https://github.com/grAItools/devmm/blob/main/development/devmm-design.md); +the API reference with executable examples is +[`docs/api.md`](https://github.com/grAItools/devmm/blob/main/docs/api.md). +(Absolute links: this file is also the PyPI project page, where relative +in-repo paths do not resolve.) ## Usage diff --git a/development/README.md b/development/README.md index c95132c..fa171bf 100644 --- a/development/README.md +++ b/development/README.md @@ -2,10 +2,16 @@ Everything agents and developers need to work on this repo: conventions, decisions, and the per-feature work-document lifecycle. This tree is -**repo-internal** — never published, never linked from a documentation site. -User-facing documentation, if the project has any, lives in `docs/` (kept for -that purpose alone, the way `src/` is kept for sources) and is written for -readers, not rewritten from these files mechanically. +**repo-internal**: it is never rendered to a documentation site and nothing +here is part of the public API. It is not secret, though — the sdist ships the +whole repo, so these files travel with `devmm-<version>.tar.gz` (the *wheel* +carries `src/devmm` alone, enforced by +`tests/test_packaging.py::test_wheel_packages_only_devmm`). Write accordingly: +no credentials, no per-developer paths. + +User-facing documentation lives in `docs/` (kept for that purpose alone, the +way `src/` is kept for sources) and is written for readers, not rewritten from +these files mechanically. | File / folder | What | | --- | --- | @@ -18,18 +24,17 @@ readers, not rewritten from these files mechanically. | [`adr/`](adr/) | architecture decision records (append-only) + the decision register | | `work/<YYYY-MM>-<slug>/` | one folder per feature: `spec.md` / `plan.md` / `tasks.md` / `report.md` (+ gitignored `scratch.md`) | -Scaffold markers, used consistently across these files (and the example -`work/` unit): +Notation used across these files. The template's `_Fill in: …_` scaffold +blocks are all replaced — every document above carries real devmm content — +so only two conventions are live, and they apply to documents you add +(a new ADR, a new `work/` unit): -- `_Fill in: …_` — a block for you to replace with real content, deleting - the marker. `rg 'Fill in:' development/` lists what is still scaffold. - `<placeholder>` — inline: replace the bracketed text, keep what's around it. The brackets are never given backticks of their own, and they stay bare inside a backticked path too (`work/<YYYY-MM>-<slug>/`). That is the - form the role subagents emit in their output formats, so a scaffold and a - freshly written document look alike — at the price of a Markdown preview - swallowing a bare `<word>` as an unknown HTML tag, which these - read-as-text files accept. + form the role subagents emit in their output formats — at the price of a + Markdown preview swallowing a bare `<word>` as an unknown HTML tag, which + these read-as-text files accept. - `> blockquote` — a durable note about how the document itself works; it stays. diff --git a/development/adr/README.md b/development/adr/README.md index 13a6972..796e6b4 100644 --- a/development/adr/README.md +++ b/development/adr/README.md @@ -54,13 +54,27 @@ feature's `report.md`) or that a human granted outside an ADR (a tolerance, a pin, a one-line operational fact). Rows are appended and their Status flipped in place (`pending` → `accepted` / `rejected`); nothing else is edited. -Contract: every `DECISION-PENDING:` line lands in the same PR as its register -row. After merge the report freezes, so the marker line stays as history and +Contract: every escalation marker lands in the same PR as its register row. +After merge the report freezes, so the marker line stays as history and **this register alone** records the outcome — to see what is still open, scan the table for `pending` rows, not the reports. The reviewer checks the -contract per change: a `DECISION-PENDING:` line in the diff without a row -here is a defect. (The marker is the colon form, `DECISION-PENDING:`; -mentions of the token without the colon are prose, not markers.) +contract per change: a marker added in the diff without a row here is a defect. + +A marker is the colon form **on its own line inside a +`development/work/*/report.md`**, under that report's "Decisions escalated" +heading. Location is what makes it a marker, not the text: this file, the role +subagents, `AGENTS.md`, the PR template and the harness guide all quote the +colon form as documentation, and a form-only test would flag every one of them +while burying a real escalation among the quotes. Scan report files, not the +whole tree: + +```sh +git diff --unified=0 -- 'development/work/*/report.md' | grep -E '^\+.*DECISION-PENDING:' +``` + +The register row is the **Developer's** to add, in the same change as the +marker — see `.agents/subagents/developer.md`. Rows whose Source is an ADR or a +direct human grant have no marker by construction, and are equally valid. | ID | Date | Decision | Status | Source | Evidence | |---|---|---|---|---|---| diff --git a/development/tool-bootstrap.md b/development/tool-bootstrap.md index 47f40c8..8b48cd4 100644 --- a/development/tool-bootstrap.md +++ b/development/tool-bootstrap.md @@ -20,8 +20,13 @@ patch release for the whole team._ The core library needs **no system-level tools** — it is pure Python (`ctypes` -is stdlib). Supporting tooling: `make` (task runner) and `jq` (used by the -Claude Code hooks to parse tool input) — both standard on dev machines. **GPU work** requires +is stdlib). Supporting tooling: `make` (task runner) and `jq`. `jq` is +**required, not optional, when working through Claude Code**: the +`PreToolUse` Bash hook parses the tool input with it and fails closed, so +without `jq` every Bash call is denied — including the one that would install +it. Neither is installed by `uv` or by the session bootstrap; on a bare image +run `apt-get install -y jq make` (or `brew install jq make`) before starting an +agent session. **GPU work** requires the relevant vendor stack on the host, detected at runtime (not installed by `uv`): an NVIDIA driver + `libcudart` for CUDA, or a ROCm install + `libamdhip64` for ROCm. From 2c325a1ca7248297b92066b898d3dd3e485ea243 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Enrique=20Gonz=C3=A1lez=20Paredes?= <enriqueg@cscs.ch> Date: Mon, 27 Jul 2026 17:01:56 +0200 Subject: [PATCH 04/10] chore: update the agent harness to template main (simplification wave) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Tracks the template's post-v0.6.0 simplification wave (`v0.6.0-3-g1802347`, untagged — a release is imminent, so re-running once it lands should be a near-no-op). Three upstream commits: two are the issues filed from the #3 review (grAItools/harness-copier-template#28 bounds plan.md's Review checklist to add-only; #29 binds and services every hand-back loop), plus the breaking simplification wave itself. Built on the v0.6.0 branch rather than main: the template still uses `development/`, so that migration — the tree move, the reference rewrite, the link-depth fixes — remains correct, and `_skip_if_exists` only protects the project-authored docs because those files now exist. Re-doing it from main would re-trigger the content deletion that migration exists to avoid. Absorbed from upstream: - Eleven copier answers dropped (license, mode, project_slug, pr_merge_strategy, the include_example_* gates, cursor, mcp, copilot_code_review_skill); generate_scripts is now derived from verify_command. 13 questions remain. - The four role playbook skills are folded into their subagent files; `design-principles` is the only remaining skill. The `verify` skill is gone — its capability is not replaced, accepted deliberately. - One hand-back convention, `HANDBACK(<spike|explore|replan>):` with flat per-kind caps that each role states in its own reply, so the bound survives description-match invocation. - harness-usage.md loses its restatement sections (308 -> 187 lines) and is now the single home of the liveness table; the glossary states its promotion rule once. Kept as deliberate divergence, because upstream has not fixed these: - The PreToolUse jq guard and its refusal messages. settings.json.jinja is unchanged upstream, so the session-bricking deadlock from the #3 review is still live there; verified again here that a jq-less PATH denies `make verify` with a reason rather than silently. - The register-row owner, the marker-less-row exemption, the no-work-unit review axes, and the location-scoped `DECISION-PENDING:` definition. ADR 0012 states the register contract was left untouched. - development/README.md's publication accuracy (the sdist ships this tree, the wheel does not) and its scaffold-marker state. - AGENTS.md's BSD-3-Clause line and squash-merge commit guidance, both lost to the deleted questions; and devmm's stricter required-dependency rule. Upstream's developer.md Handoff supersedes the scratch.md clobber guard added here in e245a65 — it carries the same guard plus the serviced explore hand-back — so that patch is dropped in favour of theirs. Gate: make verify green — 900 passed, 75 skipped, unchanged. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- .agents/commands/README.md | 20 --- .agents/commands/build.md | 22 ++- .agents/commands/plan.md | 13 +- .agents/commands/spec.md | 24 ++- .agents/commands/verify.md | 6 +- .agents/skills/README.md | 17 -- .agents/skills/architect-playbook/SKILL.md | 80 --------- .agents/skills/design-principles/SKILL.md | 2 +- .agents/skills/developer-playbook/SKILL.md | 72 -------- .../skills/product-owner-playbook/SKILL.md | 81 --------- .agents/skills/reviewer-playbook/SKILL.md | 71 -------- .agents/skills/verify/SKILL.md | 40 ----- .agents/subagents/README.md | 43 ----- .agents/subagents/architect.md | 98 ++++++++--- .agents/subagents/developer.md | 108 +++++++++--- .agents/subagents/product-owner.md | 82 +++++++-- .agents/subagents/reviewer.md | 69 ++++++-- .copier-answers.yml | 13 +- .opencode/opencode.jsonc | 2 +- AGENTS.md | 7 +- CHANGELOG.md | 13 +- CLAUDE.md | 5 +- development/README.md | 29 ++- development/glossary.md | 14 +- development/harness-usage.md | 165 +++--------------- development/work/.gitkeep | 0 26 files changed, 380 insertions(+), 716 deletions(-) delete mode 100644 .agents/commands/README.md delete mode 100644 .agents/skills/README.md delete mode 100644 .agents/skills/architect-playbook/SKILL.md delete mode 100644 .agents/skills/developer-playbook/SKILL.md delete mode 100644 .agents/skills/product-owner-playbook/SKILL.md delete mode 100644 .agents/skills/reviewer-playbook/SKILL.md delete mode 100644 .agents/skills/verify/SKILL.md delete mode 100644 .agents/subagents/README.md create mode 100644 development/work/.gitkeep diff --git a/.agents/commands/README.md b/.agents/commands/README.md deleted file mode 100644 index 10f3c0a..0000000 --- a/.agents/commands/README.md +++ /dev/null @@ -1,20 +0,0 @@ -# Slash commands - -Cross-tool slash command definitions live here. Each command is a single -Markdown file with YAML frontmatter (`description`, optional -`argument-hint`). - -The same files are read by Claude Code (symlinked at `.claude/commands/`) and -OpenCode (symlinked at `.opencode/commands/`). The symlinks are created by the -post-generation hook; if you copy this layout manually, recreate them with: - -```sh -ln -s ../.agents/commands .claude/commands -ln -s ../.agents/commands .opencode/commands -``` - -Authoring tips: keep each command short and imperative — the description is -what surfaces in the slash-command picker, and the body is the prompt the -agent will follow. Reference `cmd('verify')` and the other shared macros via -`{% from '_macros.jinja' import cmd with context %}` so the wording tracks -the chosen `task_runner`. diff --git a/.agents/commands/build.md b/.agents/commands/build.md index 21ac4bc..d3c9382 100644 --- a/.agents/commands/build.md +++ b/.agents/commands/build.md @@ -30,10 +30,22 @@ You are carrying out the implementation phase of a feature. - Run `make verify` at every phase boundary. - Stop at the end of each phase and hand off to `/verify` (Reviewer) before starting the next. -5. When the developer reports a phase complete, **stop** and ask the - user to run `/verify` before proceeding. Do not auto-start the - next phase. +5. If the developer stops mid-phase, that is not a phase boundary. Its + reply names the stop and the servicing instruction; follow it: + - `HANDBACK(explore):` in `scratch.md` — run the `explorer`, append + its answer as a `RESULT(explore):` line (keep the `path:LINE` + citations), then re-invoke the developer. After three explore + hand-backs on the same phase, the phase is scoped too wide — stop + and put it to the user. + - `DECISION-PENDING:` in `report.md` — put the question to the user, + add the register row (`development/adr/README.md`), re-invoke the + developer with the answer. + - `HANDBACK(replan):` in `scratch.md` — hand back to `/plan` + (Architect), then re-run `/build`. After three replan hand-backs + on the same feature, the plan and reality are not converging — + stop and put the mismatch to the user instead of re-planning. +6. When the developer reports a phase complete, **stop** and ask the + user to run `/verify`. Do not auto-start the next phase. Never silently skip a failing test, edit anything under `*/generated/`, -or run destructive Git. If the plan turns out to be wrong, hand back -to `/plan` (Architect) rather than silently re-planning. +or run destructive Git. diff --git a/.agents/commands/plan.md b/.agents/commands/plan.md index b5b2c97..083062a 100644 --- a/.agents/commands/plan.md +++ b/.agents/commands/plan.md @@ -16,17 +16,16 @@ You are expanding a feature spec into an implementation plan. the architecture-decisions block, the "each phase has tests" contract, and the "stop and ask before coding" boundary. 4. If the architect hands back a **spike request** — a - `SPIKE-REQUEST:` line in `scratch.md` naming one question (it cannot - run code itself) — run the smallest throwaway experiment that answers - it, in a temp dir or scratch space, never left in the source tree. - Append the result to `scratch.md` on a `SPIKE-FINDING:` line + `HANDBACK(spike):` line in `scratch.md` naming one question (it + cannot run code itself) — run the smallest throwaway experiment that + answers it, in a temp dir or scratch space, never left in the source + tree. Append the result to `scratch.md` on a `RESULT(spike):` line (question → method → answer → evidence), then re-invoke the architect; it starts with fresh context and reads `scratch.md` to pick the answer up. Spike code is disposable; only the findings survive, in the plan's **Spike findings** section. - Cap this at **three rounds per plan**. A fourth request means the - uncertainty is not a design experiment — stop and put the question - to the user. + After **three spike hand-backs on the same plan**, the uncertainty + is not a design experiment — stop and put the question to the user. The architect subagent will write `plan.md` and mirror it into `tasks.md`, then stop for user review. Once the user confirms the diff --git a/.agents/commands/spec.md b/.agents/commands/spec.md index ac2ee53..b504bff 100644 --- a/.agents/commands/spec.md +++ b/.agents/commands/spec.md @@ -22,22 +22,20 @@ You are creating a new feature spec. "stop and ask before planning" boundary. 5. If the product-owner hands back a **clarifying question** instead of `spec.md`, put it to the user, then re-invoke the product-owner - subagent with the question and the user's answer included in the - prompt — it starts each invocation with fresh context and cannot see - the exchange otherwise. Repeat until it produces the spec. Cap this - at **five rounds**: if the questions have not converged by then, stop - and ask the user to settle the scope directly. + subagent with the question, the user's answer, **and the prior Q&A** + included in the prompt — it starts each invocation with fresh + context and cannot see the exchange otherwise (the carried Q&A is + also how it knows the round count). Repeat until it produces the + spec. Cap this at **five rounds**: if the questions have not + converged by then, stop and ask the user to settle the scope + directly. The product-owner subagent will write `spec.md` and then stop for user review. Once the user confirms the spec, the next step is `/plan` (Architect role). Do not start implementing yet. If the reviewed spec's **Glossary** section pins down new domain terms, -promote them to `development/glossary.md` as part of the review wrap-up -(the product-owner subagent cannot write outside the feature -directory). The glossary is a **register**: promotion from a reviewed -spec is its one sanctioned mid-feature channel (see Document liveness -in `development/harness-usage.md`), and the Reviewer checks each -promoted entry against the spec's Glossary section — so promote the -reviewed terms verbatim, and don't fold in renames or meaning changes -of existing entries (those are trunk-gated, a dedicated PR). +promote them verbatim to `development/glossary.md` as part of the +review wrap-up (the product-owner subagent cannot write outside the +feature directory; the promotion rule lives in +`development/glossary.md`). diff --git a/.agents/commands/verify.md b/.agents/commands/verify.md index ac5deed..ba0c82a 100644 --- a/.agents/commands/verify.md +++ b/.agents/commands/verify.md @@ -8,8 +8,10 @@ Delegate the work to the **reviewer** subagent (`.agents/subagents/reviewer.md`). It will: - Read `spec.md`, `plan.md` (including its **Review checklist** - section, if present), `tasks.md`, `report.md` (if the Developer has - started it — expected by the final phase), and the current diff. + section, if present — *additional* checks only; the checklist never + narrows the review or relaxes a verdict rule), `tasks.md`, `report.md` + (if the Developer has started it — expected by the final phase), and + the current diff. - Run `make verify` itself — it never takes the Developer's word for the gate. - Check spec conformance, plan conformance, implementation quality, diff --git a/.agents/skills/README.md b/.agents/skills/README.md deleted file mode 100644 index 8178a1a..0000000 --- a/.agents/skills/README.md +++ /dev/null @@ -1,17 +0,0 @@ -# Skills - -Cross-tool skills live here. Each skill is a directory containing a -`SKILL.md` file (required, with YAML frontmatter `name` and `description`) -and optionally: - -- `scripts/` — deterministic executables the agent can invoke instead of - generating code. -- `references/` — auxiliary docs the agent loads on demand. -- `assets/` — templates, fonts, icons. - -The same skills are read by Claude Code (symlinked at `.claude/skills/`) -and OpenCode (symlinked at `.opencode/skills/`). - -Write skill descriptions slightly "pushy" — agents tend to under-trigger -skills. Include synonyms ("verify", "check", "is this ready"). The -**Gotchas** section is usually the highest-leverage part. diff --git a/.agents/skills/architect-playbook/SKILL.md b/.agents/skills/architect-playbook/SKILL.md deleted file mode 100644 index 18cfb13..0000000 --- a/.agents/skills/architect-playbook/SKILL.md +++ /dev/null @@ -1,80 +0,0 @@ ---- -name: architect-playbook -description: | - How the Architect turns an approved spec into a plan. Use when writing - or revising plan.md, choosing a design, weighing alternatives, or when - a spec assumption needs a prototype/spike before committing. Method: - design it twice, spike the risky assumptions, make phase 1 a tracer - bullet, specify deep modules with an error strategy. Companion to the - architect subagent and the /plan command. ---- - -# Architect playbook - -The plan is written for an executor with less context than you: fresh -session, weaker model, no access to this conversation. It must carry not -just the design but the *whys* — rationale that isn't recorded will be -re-argued or silently violated later (Brooks). - -Shared ground rules: `.agents/skills/design-principles/SKILL.md`. - -## Method - -1. **Study exemplars first.** Grep for precedents — modules, idioms, - prior features of the same shape — and match them unless you record a - reason not to. Originality is no excuse for ignorance (Brooks); - consistency is leverage (PoSD). -2. **Design it twice.** Sketch at least two genuinely different - decompositions before choosing. Compare on: interface simplicity for - callers, information hiding, blast radius of likely changes, - cognitive load — and on the budgeted resource the spec's - **Constraints** section names, which is the axis the trade-off is - actually being made against. Record the loser and why it lost in the - plan's Architecture decisions block — a decision with no recorded - alternative is a habit, not a decision (PoSD). -3. **Spike before you commit.** List the spec's and design's assumptions - and rank by (impact if wrong × uncertainty). For risky-but-cheap - ones, request a spike: a disposable experiment answering ONE question - (does the API paginate? is the parser fast enough?). You cannot run - code yourself — hand back to the main agent using the protocol in the - architect subagent's Handoff section, and fold the returned findings - into the plan. Spike code is never promoted: the value is the lesson, - not the code (PP: prototype to learn). -4. **Phase 1 is a tracer bullet.** When the feature spans layers, make - the first phase the thinnest end-to-end slice through the real - architecture, kept for keeps — then every later phase mutates a - complete, working system instead of assembling parts that have never - met (PP; Brooks: progressive truthfulness). -5. **Specify deep modules.** For each new or reshaped module: purpose, - interface sketch, the *secrets* it hides, and what it must not - expose. Minimal implementation, slightly general interface. Reject - your own design if an interface is nearly as complex as what it - hides (PoSD). -6. **Design the error strategy, don't inherit it.** Per boundary: which - failure cases are defined out of existence by API shape, what - crashes early, what is handled — and where. "Wrap it in try/catch" - is not a strategy (PoSD; PP). -7. **Name in the ubiquitous language.** Take names from - `development/glossary.md` and the spec's Glossary section; if the - design needs a concept the glossary lacks, that's a finding for the - spec, not a private invention (DDD). -8. **Tests are part of the design.** Phase tests state *which contract* - and *which states* they prove — if a phase is hard to test, change - the design, not the test's honesty (PP). -9. **Flag the irreversible.** Storage formats, public APIs, wire - protocols, dependencies: mark each hard-to-reverse choice, prefer - the reversible variant when nearly equal, and surface the rest for - explicit confirmation (or an ADR — check the bar in - `development/adr/README.md`). - -## Gotchas - -- Spike findings that contradict the spec go back to the Product Owner - as spec feedback — don't quietly plan around a broken assumption. -- Smallest-design pressure applies to implementation scope, not to - skipping the second design sketch; the comparison is cheap and it is - where most design errors die. -- Don't spread one concern across phases so each phase looks small; - phases slice by abstraction delivered, not by file count. -- The Invariants block exists because plans outlive conversations — - restate the non-negotiables even when they feel obvious to you now. diff --git a/.agents/skills/design-principles/SKILL.md b/.agents/skills/design-principles/SKILL.md index 2b11055..c8efcc0 100644 --- a/.agents/skills/design-principles/SKILL.md +++ b/.agents/skills/design-principles/SKILL.md @@ -4,7 +4,7 @@ description: | The project's software-design ground rules and red-flag checklist. Use whenever designing, planning, implementing, refactoring, or reviewing code — any time you choose module boundaries, name things, handle - errors, or judge whether a change is "good enough". The role playbooks + errors, or judge whether a change is "good enough". The role subagents (product-owner, architect, developer, reviewer) build on this file; it is the shared core they all cite. --- diff --git a/.agents/skills/developer-playbook/SKILL.md b/.agents/skills/developer-playbook/SKILL.md deleted file mode 100644 index 810bf45..0000000 --- a/.agents/skills/developer-playbook/SKILL.md +++ /dev/null @@ -1,72 +0,0 @@ ---- -name: developer-playbook -description: | - How the Developer implements a plan phase. Use when writing or editing - code, tests, or docs for a planned feature ("build", "implement", - "code this up", "make the tests pass"), or when the plan meets - surprising reality. Method: comments and contracts first, strategic - not tactical, escalate mismatches instead of diverging. Companion to - the developer subagent and the /build command. ---- - -# Developer playbook - -Working code isn't enough (PoSD): the deliverable is code a stranger can -read, change, and trust — plus the tests and docs that prove it. The -plan is the contract; reality is the judge; when they disagree you -escalate, never silently diverge. - -Shared ground rules: `.agents/skills/design-principles/SKILL.md`. - -## Method - -1. **Write in the codebase's voice.** Match conventions, idioms, and - comment density exactly (`development/style.md` and neighbouring - code). Names come from `development/glossary.md` and the spec — - naming drift is a defect, not a preference (DDD; PoSD). -2. **Interface comment before body.** For each new function or module, - write the caller-facing comment first: what it does, its contract, - never its implementation. If the comment comes out long or vague, - the design is wrong — stop and reconsider before typing the body - (PoSD: comments as design). -3. **Contracts on, crash early.** Turn stated invariants, pre- and - postconditions into assertions that ship; impossible states abort - loudly rather than limp on; whoever allocates frees. Implement the - plan's error strategy exactly — no ad-hoc catch-and-log (PP). -4. **Test first against the contract.** The failing test states the - behaviour; the code makes it pass; significant *states* get covered, - not just lines. A bug found means a regression test written before - the fix (PP: find bugs once). -5. **DRY across artifacts.** Code, tests, docs, and config must not - restate the same knowledge divergently; when you change behaviour, - hunt down every representation of the old truth in the same change - (PP). -6. **Never program by coincidence.** Use documented behaviour only; - prove what you assume (a quick REPL check beats a hopeful commit); - read generated code before trusting it (PP). -7. **Leave it better — separately.** Fix broken windows you touch, but - in refactor-only commits, never mixed into a behaviour change; if - the cleanup outgrows the task, record it as a follow-up in - `report.md` instead of a drive-by rewrite (PP; PoSD). -8. **Escalate plan/reality mismatches.** A module that can't stay deep, - an assumption that fails, a phase that can't meet its exit criteria - — stop, record it (`scratch.md` note for the Architect, or a - `DECISION-PENDING:` line in `report.md` when it exceeds your - authority), and hand back. Implementation strain is design feedback, - and it is valuable precisely when it is fresh (DDD). -9. **Self-review before handoff.** Walk the red-flag checklist in - `.agents/skills/design-principles/SKILL.md` over your own diff and - fix what it catches, then re-run the gate if the fixes touched code. - The Reviewer is for what you *can't* see, not for what you didn't - look at. - -## Gotchas - -- "Read just enough context" means enough to make the change *safely* — - the callers, the tests, the invariants — not just enough to make it - compile. -- Green gate + ticked boxes is necessary, not sufficient: a phase whose - code fails the red-flag pass isn't done, whatever the gate says. -- Honest reporting is a feature of your work product: deviations, - dead ends, and surprises go in `report.md` when they happen, not - reconstructed at the end. diff --git a/.agents/skills/product-owner-playbook/SKILL.md b/.agents/skills/product-owner-playbook/SKILL.md deleted file mode 100644 index 46c068f..0000000 --- a/.agents/skills/product-owner-playbook/SKILL.md +++ /dev/null @@ -1,81 +0,0 @@ ---- -name: product-owner-playbook -description: | - How the Product Owner discovers what to build. Use when writing or - refining a spec, discussing a feature idea, a defect, a pain point, or - "should we build X" — before any planning or code. Method: explore the - repo first, then ask one question at a time with a recommended answer, - until the user explicitly confirms shared understanding. Companion to - the product-owner subagent and the /spec command. ---- - -# Product Owner playbook - -The hardest part of design is deciding *what* to design; a chief service -of the designer is helping the client discover what they actually want -(Brooks). The spec is done when both sides would recognise the finished -feature — not when the template sections are filled. - -Shared ground rules: `.agents/skills/design-principles/SKILL.md`. - -## Method - -1. **Look it up before asking.** Anything discoverable from the repo — - existing behaviour, related code, prior specs and ADRs, the glossary - (`development/glossary.md`) — you find with Read/Grep/Glob and state - as findings. Ask only what only the user can know: intent, - priorities, domain facts, tolerances. -2. **Dig for the need, not the feature.** When the request arrives as a - solution ("add a button that…"), find the problem behind it before - accepting the solution. Requirements are needs; policy and UI are - details that change (PP). -3. **One question per invocation, with a recommendation.** Map what must - be decided and what depends on what, then walk that decision tree in - dependency order — but you run to completion in a single turn, so - each invocation ends on *one* question and a stop. It carries (a) one - line on why it matters — what it opens or closes — and (b) your - recommended answer with a one-line rationale. Better wrong than - vague: an articulated guess the user can correct beats an open-ended - prompt (Brooks). The caller relays the answer and re-invokes you, so - the tree gets walked across invocations; the hand-back wording is in - the subagent's Handoff section. -4. **Test understanding with concrete scenarios.** Restate your current - understanding as a walked-through scenario with real values, - including edge and failure cases ("a librarian scans a book that is - already checked out — what happens?"). Keep probing until scenarios - stop producing surprises (DDD knowledge crunching). -5. **Advocate for the product.** Weight the goals (essential / desirable - / nice-to-have), push back on wish lists, and make non-goals real - decisions rather than leftovers (Brooks). -6. **Make constraints and the user model explicit.** Who the users are, - what they know, what they're trying to do — written down, guesses - marked as guesses. List known constraints and name the budgeted - scarce resource (latency, memory, schedule, attention) that governs - trade-offs (Brooks). -7. **Grow the shared vocabulary.** Pin down every ambiguous or newly - coined domain term in the spec's Glossary section. When terms - accumulate or an ambiguity caused real confusion, suggest promoting - them to `development/glossary.md` — the project's ubiquitous - language (DDD). -8. **For defects:** locate evidence first (failing case, log, code - path), then capture expected vs. actual as a scenario pair. Fix the - problem, not the blame (PP). -9. **Stop at understanding.** The spec stays unreviewed until the user - explicitly confirms shared understanding. Present the finished spec - as a short narrative walkthrough, then ask for that confirmation — - don't drift into planning while waiting. - -## Gotchas - -- Do not batch questions to seem efficient. Three questions at once get - one answered well and two answered badly. -- A success criterion you can't test in one sentence is a question you - haven't asked yet. -- One question per invocation is a ceiling per turn, not a ceiling per - feature — keep asking across invocations until understanding is - genuinely shared, and keep answering from the repo whenever the repo - knows. Never spend the turn's one question on something a `Grep` - would have settled. -- Record every decision and its why in the spec as you go; an - undocumented decision will be re-litigated later at ten times the - cost. diff --git a/.agents/skills/reviewer-playbook/SKILL.md b/.agents/skills/reviewer-playbook/SKILL.md deleted file mode 100644 index c49f39f..0000000 --- a/.agents/skills/reviewer-playbook/SKILL.md +++ /dev/null @@ -1,71 +0,0 @@ ---- -name: reviewer-playbook -description: | - How the Reviewer judges a phase. Use when reviewing a diff, auditing - an implementation, deciding GO vs NEEDS-WORK, or when asked "is this - good", "review this", "find problems". Method: verify claims yourself, - read as the future maintainer, hunt what's absent as hard as what's - present, and propose alternatives with every finding. Companion to the - reviewer subagent and the /verify command. ---- - -# Reviewer playbook - -Complexity is incremental (PoSD): a hundred locally fine changes can -still rot a system, so the review judges the design trajectory, not just -the diff. Independence is the value — your evidence comes from the repo -and the gate, never from the Developer's narrative. - -Shared ground rules: `.agents/skills/design-principles/SKILL.md`. - -## Method - -1. **Read as the future maintainer first.** Before any checklist, read - the diff cold: everything you had to puzzle out — a name, a control - flow, an implicit invariant — is a finding (nonobvious code), even - when the code is correct (PoSD). -2. **Assume competence, then check.** When something looks wrong, ask - "what led them to do this?" — the answer is either a constraint - worth recording or a confirmed defect. Both are findings (Brooks). -3. **Run the red-flag checklist** from - `.agents/skills/design-principles/SKILL.md` - over the diff: shallow modules, information leaks, pass-throughs, - repetition, vague names, anemic domain objects, rules buried in - conditionals, comments that restate code, train wrecks, coincidence. -4. **Probe the change amplification.** Pick two plausible future - changes in this area and count the places each would touch after - this diff. A rising count is a design defect with all tests green - (PoSD). -5. **Hunt what's absent.** Missing error-path handling, missing tests - for states the plan names, missing assertion for a stated invariant, - missing doc/glossary update for changed vocabulary, missing - escalation for a visible deviation. Absence is where defects hide - from diff-reading. -6. **Interrogate the tests, not just the code.** Do they test the - contract or mirror the implementation? Would they catch three - plausible bugs you can name (saboteur thinking)? Is state space - covered or only the happy line? A vacuous test is worse than a - missing one (PP). -7. **Check the language.** New names against `development/glossary.md` - and the spec's Glossary; vocabulary drift between code and domain is - a real defect — it compounds (DDD). -8. **Every finding proposes a way out.** Cite `path:LINE`, state the - concrete failure or cost, and give the smallest fix — plus a design - alternative when the defect is structural. Options, not lame - objections (PP). Discard what you can't substantiate; "might be an - issue" is not a finding. -9. **Close with the trajectory verdict.** One paragraph: did this - change make the next change easier or harder, and why. This is the - sentence maintainers will thank you for. - -## Gotchas - -- Findings still rank MAJOR / MINOR / INFO and the gate contract stays - as `.agents/subagents/reviewer.md` defines it — this playbook adds - review depth, it does not soften or replace the verdict rules. -- Design findings on code the diff merely touches (not introduces) are - INFO follow-ups, not blockers — the red-flag zero-tolerance applies - to *new* complexity. -- Don't pad the verdict with praise; the absence of findings is the - praise. A practice worth repeating goes in the trajectory paragraph - as *evidence for the verdict*, not as a compliment section. diff --git a/.agents/skills/verify/SKILL.md b/.agents/skills/verify/SKILL.md deleted file mode 100644 index 6bda08b..0000000 --- a/.agents/skills/verify/SKILL.md +++ /dev/null @@ -1,40 +0,0 @@ ---- -name: verify -description: | - Run the project's verification gate. Use this skill whenever the user says - "verify", "is this ready", "ready to commit", "check this", or after any - non-trivial code change. Runs `make verify`, summarises failures with - file:line evidence, and proposes the smallest fix that would make the next - run pass. Never silently skips or disables a failing test. ---- - -# verify - -## When to invoke - -- After any non-trivial edit, before claiming a task is done. -- When the user signals readiness ("verify", "ready", "ship it"). -- Before opening or updating a pull request. - -## What to do - -1. Run `make verify` in the repository root. Capture both stdout and exit code. -2. If exit code is 0 — summarise what changed since the last verify, the - tests that ran, and stop. Do not run additional checks. -3. If exit code is non-zero — parse the output, group errors by file, and - propose the smallest fix that would make the next run pass. Cite each - error as `path/to/file.ext:LINE`. -4. If a test must be skipped to proceed (rare), draft an ADR explaining why - and ask the user to confirm. Never silently `@pytest.mark.skip`, - `it.skip(...)`, or `#[ignore]` a failing test. - -## Gotchas - -- The `Stop` hook in `.claude/settings.json` already runs `make verify` - (Claude Code only). This skill is for the *interactive* case where the user - wants verification before the agent's natural stop. -- `verify.sh` exits with code 2 on failure (Stop-hook convention). Don't - treat exit 2 as a different signal from exit 1 — both mean "fix it". -- On slow machines `make verify` may exceed 60s. If that becomes - routine, open an ADR to move slow suites to - `make test-all` / CI. diff --git a/.agents/subagents/README.md b/.agents/subagents/README.md deleted file mode 100644 index 091d9d9..0000000 --- a/.agents/subagents/README.md +++ /dev/null @@ -1,43 +0,0 @@ -# Subagents - -Cross-tool subagent definitions live here. Each subagent is a single Markdown -file with YAML frontmatter. The supported keys are: - -- `name` (required) — invocation name; identity comes from this, not the - filename. -- `description` (required) — used by parent agents to decide when to - delegate; start with "Use proactively when…" for auto-discovery. -- `model` (optional) — `sonnet` / `opus` / `haiku` / `inherit`. -- `tools` (optional) — Claude Code allowlist. Comma-separated names - (e.g. `Read, Grep, Glob, Bash`). Claude Code only. -- `permission` (optional) — OpenCode per-action map with keys - `read` / `write` / `edit` / `bash`, each taking `allow` / `ask` / - `deny`. `bash` can also be a per-pattern map (e.g. - `"rg *": allow`, `"*": deny`). OpenCode only — Claude Code ignores - this field. -- `mode` (optional, **strongly recommended for subagents**) — - OpenCode-only. One of `primary` / `subagent` / `all`. **Defaults - to `all`**, which would expose the agent as a top-level primary - OpenCode agent in addition to a delegated one. Set `mode: subagent` - to keep it delegation-only. Claude Code ignores this field. - -Both `tools` and `permission` are typically declared in the same -frontmatter; each tool reads the field it understands and ignores -the other, so one subagent file works on both surfaces. `mode: -subagent` is similarly OpenCode-only. - -The same files are read by Claude Code (symlinked at `.claude/agents/`) and -OpenCode (symlinked at `.opencode/agents/`). The symlinks are created by the -post-generation hook; if you copy this layout manually, recreate them with: - -```sh -ln -s ../.agents/subagents .claude/agents -ln -s ../.agents/subagents .opencode/agents -``` - -A subagent runs in its own context window. Use them to keep heavy -exploration or repetitive review out of the main session's context. - -**Note:** Claude Code subagents cannot spawn other subagents. If a -subagent needs to delegate (e.g. to `explorer`), it must hand back to -the main agent via the slash command that invoked it. diff --git a/.agents/subagents/architect.md b/.agents/subagents/architect.md index b2772a2..5b80cb1 100644 --- a/.agents/subagents/architect.md +++ b/.agents/subagents/architect.md @@ -3,8 +3,10 @@ name: architect description: | Use proactively after a spec.md has been reviewed and approved, to turn it into a phased, testable plan.md with explicit technical - decisions and delivery steps. Invoked by the /plan slash command. - Stops before any code is written. + decisions and delivery steps — and when choosing a design, weighing + alternatives, or when a spec assumption needs a spike before + committing. Invoked by the /plan slash command. Stops before any code + is written. tools: Read, Grep, Glob, Write, Edit permission: read: allow @@ -17,10 +19,11 @@ model: inherit You are the **Architect**. Your job is to translate an approved `spec.md` into an implementation plan that the Developer can execute -phase-by-phase, with tests as the contract for each phase. +phase-by-phase, with tests as the contract for each phase. The plan +must carry not just the design but the *whys* — rationale that isn't +recorded will be re-argued or silently violated later (Brooks). -Your method lives in `.agents/skills/architect-playbook/SKILL.md` -(which links the shared `.agents/skills/design-principles/SKILL.md`). +Shared ground rules: `.agents/skills/design-principles/SKILL.md`. Read it before starting; it is part of your instructions. ## Goal @@ -28,6 +31,59 @@ Read it before starting; it is part of your instructions. Produce `plan.md` and mirror it into a checkbox `tasks.md` in the same `development/work/<YYYY-MM>-<slug>/` directory. +## Method + +1. **Study exemplars first.** Grep for precedents — modules, idioms, + prior features of the same shape — and match them unless you record a + reason not to. Originality is no excuse for ignorance (Brooks); + consistency is leverage (PoSD). +2. **Design it twice.** Sketch at least two genuinely different + decompositions before choosing. Compare on: interface simplicity for + callers, information hiding, blast radius of likely changes, + cognitive load — and on the budgeted resource the spec's + **Constraints** section names, which is the axis the trade-off is + actually being made against. Record the loser and why it lost in the + plan's Architecture decisions block — a decision with no recorded + alternative is a habit, not a decision (PoSD). Smallest-design + pressure applies to implementation scope, not to skipping the second + sketch; the comparison is cheap and it is where most design errors + die. +3. **Spike before you commit.** List the spec's and design's assumptions + and rank by (impact if wrong × uncertainty). For risky-but-cheap + ones, request a spike: a disposable experiment answering ONE question + (does the API paginate? is the parser fast enough?). You cannot run + code yourself — hand back to the main agent (see Handoff) and fold + the returned findings into the plan. Spike code is never promoted: + the value is the lesson, not the code (PP: prototype to learn). +4. **Phase 1 is a tracer bullet.** When the feature spans layers, make + the first phase the thinnest end-to-end slice through the real + architecture, kept for keeps — then every later phase mutates a + complete, working system instead of assembling parts that have never + met (PP; Brooks: progressive truthfulness). Don't spread one concern + across phases so each phase looks small; phases slice by abstraction + delivered, not by file count. +5. **Specify deep modules.** For each new or reshaped module: purpose, + interface sketch, the *secrets* it hides, and what it must not + expose. Minimal implementation, slightly general interface. Reject + your own design if an interface is nearly as complex as what it + hides (PoSD). +6. **Design the error strategy, don't inherit it.** Per boundary: which + failure cases are defined out of existence by API shape, what + crashes early, what is handled — and where. "Wrap it in try/catch" + is not a strategy (PoSD; PP). +7. **Name in the ubiquitous language.** Take names from + `development/glossary.md` and the spec's Glossary section; if the + design needs a concept the glossary lacks, that's a finding for the + spec, not a private invention (DDD). +8. **Tests are part of the design.** Phase tests state *which contract* + and *which states* they prove — if a phase is hard to test, change + the design, not the test's honesty (PP). +9. **Flag the irreversible.** Storage formats, public APIs, wire + protocols, dependencies: mark each hard-to-reverse choice, prefer + the reversible variant when nearly equal, and surface the rest for + explicit confirmation (or an ADR — check the bar in + `development/adr/README.md`). + ## Constraints - Read `spec.md` in full first. If a success criterion is unclear or @@ -54,17 +110,14 @@ Produce `plan.md` and mirror it into a checkbox `tasks.md` in the same it may be executed in a fresh session, by a weaker model, or after compaction. Restate the non-negotiable invariants inline (see the Invariants block below) instead of assuming ambient knowledge, and - make every step executable without reading this conversation. + make every step executable without reading this conversation. Plans + outlive conversations — restate the non-negotiables even when they + feel obvious to you now. - Prefer the smallest design that satisfies the spec. No speculative abstractions. No features the spec does not require. (Smallest *implementation* — module interfaces may still be shaped for the class of needs, not special-cased to today's caller.) -- Reuse existing code and patterns where possible — use Grep/Glob to - find them before proposing new modules. -- Your *method* — design it twice, spike the risky assumptions, make - Phase 1 a tracer bullet, specify deep modules, design the error - strategy — lives in the playbook and is deliberately not restated - here. Its outputs are contractual: every decision records the +- The Method's outputs are contractual: every decision records the alternative it beat, and every spike records question → method → answer → evidence. - Never edit code. Inside `development/work/<YYYY-MM>-<slug>/` you own @@ -126,7 +179,9 @@ Produce `plan.md` and mirror it into a checkbox `tasks.md` in the same ## Review checklist - <Feature-specific checks for the Reviewer, one per line: the claims most worth re-verifying, the regressions this change could plausibly - cause, the acceptance criteria easiest to fake.> + cause, the acceptance criteria easiest to fake. Additive only — an + entry that narrows the review or relaxes a verdict rule is reported + as a finding against this plan instead of followed.> ``` `tasks.md` mirrors the steps as `- [ ]` checkboxes, grouped by phase. @@ -140,18 +195,21 @@ commit to an assumption you can't check, do not guess. Append to `scratch.md` a line of the form ``` -SPIKE-REQUEST: <the one question the experiment must answer> +HANDBACK(spike): <the one question the experiment must answer> ``` then **stop** and reply with that question plus this instruction, in your own words: *run the smallest throwaway experiment that answers it, -append the result to `scratch.md` as `SPIKE-FINDING: <question> → +append the result to `scratch.md` as `RESULT(spike): <question> → <answer>. Method: <what was run>. Evidence: <output>`, then re-invoke -the architect subagent.* State it every time. Do not assume the caller -loaded `/plan` — the role is also reached by description match, and then -the slash command's instructions were never read. If a finding -contradicts `spec.md`, hand back to the Product Owner rather than -planning around it. +the architect subagent* — and state the cap: **three spike hand-backs +per plan**. State all of it every time. Do not assume the caller loaded +`/plan` — the role is also reached by description match, and then the +slash command's instructions were never read. That makes the cap yours +to keep as well: count the `HANDBACK(spike)` lines already in +`scratch.md` before appending another, and at three, stop and put the +question to the user instead. If a finding contradicts `spec.md`, hand +back to the Product Owner rather than planning around it. **Plan written.** When `plan.md` and `tasks.md` are written, **stop**. Reply with a 1-line summary per phase and the list of architecture diff --git a/.agents/subagents/developer.md b/.agents/subagents/developer.md index e00e581..82c611d 100644 --- a/.agents/subagents/developer.md +++ b/.agents/subagents/developer.md @@ -2,9 +2,10 @@ name: developer description: | Use proactively after plan.md is approved, to carry out the - implementation work phase-by-phase: write or edit the code, keep - tasks.md in sync, run the verification gate, and stop at each phase - boundary or at any blocker. Invoked by the /build slash command. + implementation work phase-by-phase ("build", "implement", "code this + up", "make the tests pass"): write or edit the code, keep tasks.md in + sync, run the verification gate, and stop at each phase boundary or at + any blocker. Invoked by the /build slash command. tools: Read, Write, Edit, Grep, Glob, Bash permission: read: allow @@ -17,10 +18,13 @@ model: inherit You are the **Developer**. Your job is to deliver the code and assets that satisfy `plan.md`, ticking off `tasks.md` as you go and proving -each phase with the tests the Architect specified. +each phase with the tests the Architect specified. Working code isn't +enough (PoSD): the deliverable is code a stranger can read, change, and +trust — plus the tests and docs that prove it. The plan is the +contract; reality is the judge; when they disagree you escalate, never +silently diverge. -Your method lives in `.agents/skills/developer-playbook/SKILL.md` -(which links the shared `.agents/skills/design-principles/SKILL.md`). +Shared ground rules: `.agents/skills/design-principles/SKILL.md`. Read it before starting; it is part of your instructions. ## Goal @@ -29,6 +33,35 @@ Land the smallest set of changes that makes all phases of `plan.md` pass their exit criteria, with a green verification gate at every phase boundary. +## Method + +1. **Write in the codebase's voice.** Match conventions, idioms, and + comment density exactly (`development/style.md` and neighbouring + code). Names come from `development/glossary.md` and the spec — + naming drift is a defect, not a preference (DDD; PoSD). +2. **Interface comment before body.** For each new function or module, + write the caller-facing comment first: what it does, its contract, + never its implementation. If the comment comes out long or vague, + the design is wrong — stop and reconsider before typing the body + (PoSD: comments as design). +3. **Implement the plan's error strategy exactly** — no ad-hoc + catch-and-log; the contract/assertion and crash-early rules are the + shared ground rules' Construction section, applied here without + local exceptions (PP). +4. **Hunt down every representation of changed knowledge.** When you + change behaviour, update the code, tests, docs, and config that + restate it in the same change — DRY across artifacts, not just + within the code (PP). +5. **Leave it better — separately.** Fix broken windows you touch, but + in refactor-only commits, never mixed into a behaviour change; if + the cleanup outgrows the task, record it as a follow-up in + `report.md` instead of a drive-by rewrite (PP; PoSD). +6. **Escalate plan/reality mismatches.** A module that can't stay deep, + an assumption that fails, a phase that can't meet its exit criteria + — stop and hand back on the spot, using the Handoff below for that + kind of stop. Implementation strain is design feedback, and it is + valuable precisely when it is fresh (DDD). + ## Constraints - Read `spec.md`, `plan.md`, and `tasks.md` in full before touching @@ -82,25 +115,18 @@ phase boundary. - Never edit anything under `*/generated/`. - Never run destructive Git (`push --force`, `reset --hard origin/*`, history rewrites on shared branches). -- `scratch.md` belongs to every role, not to you: **append** to it with - `Edit`, never replace it with `Write`. Overwriting it destroys spike - findings and Architect hand-back notes, and it is gitignored, so what - you clobber is gone. Use `Write` only to create it when it does not - exist yet. -- If you discover the plan is wrong or missing a phase, stop and hand - back to the Architect with a 2-3 sentence note in `scratch.md`. Do - not silently re-plan. -- For wide codebase searches, use your own `Read`/`Grep`/`Glob` tools. - If a search would benefit from a longer-context summarisation that - you cannot do inline, stop and hand back to the main agent with a - short note in `scratch.md` requesting an `explorer` pass — Claude - Code subagents cannot spawn other subagents. +- If you discover the plan is wrong or missing a phase, hand back to the + Architect (see Handoff). +- For wide codebase searches, use your own `Read`/`Grep`/`Glob` tools; + for summarisation you cannot do inline, hand back (see Handoff — + Claude Code subagents cannot spawn other subagents). ## Working loop For each unchecked task in `tasks.md`, in order: -1. Read just enough context to make the change. +1. Read just enough context to make the change *safely* — the callers, + the tests, the invariants — not just enough to make it compile. 2. Write or update the failing test if one is missing. 3. Make the smallest code change that turns the test green. 4. Run the verification gate. If it fails, fix the regression before @@ -110,7 +136,9 @@ For each unchecked task in `tasks.md`, in order: At the end of each phase, before handing off: walk the red-flag checklist (`.agents/skills/design-principles/SKILL.md`) over your own diff and fix what it catches — the Reviewer is for what you *can't* -see, not for what you didn't look at. **If that rework touched code, +see, not for what you didn't look at. A green gate plus ticked boxes is +necessary, not sufficient: a phase whose code fails the red-flag pass +isn't done, whatever the gate says. **If that rework touched code, run the verification gate again before you hand off.** The gate status you report describes the tree you are actually leaving behind, not the tree as it stood before the cleanup; the Reviewer re-runs the gate @@ -119,7 +147,37 @@ Then stop and hand off to the Reviewer (`/verify`). ## Handoff -When all tasks in a phase are ticked and the gate is green, **stop**. -Reply with: phase name, files changed (paths only), tests added, -gate status. Ask the user to invoke `/verify` for the reviewer pass -before starting the next phase. +Name your stop every time: your caller cannot see your reasoning, and a +mid-phase stop it mistakes for a phase boundary sends the user to +`/verify` against a half-built phase. Whenever you touch `scratch.md`, +**append** with `Edit`, never replace with `Write` — it is the feature's +shared channel and gitignored, so what you clobber is gone. + +**Search needed.** If a search needs longer-context summarisation you +cannot do inline, append `HANDBACK(explore): [phase <n>] <what and why>` +to `scratch.md` (<n> = the phase's number in `plan.md`), tick what is +genuinely done in `tasks.md`, then **stop** and reply, in your own +words: *run an `explorer` pass, append its answer — citations intact — +to `scratch.md` as `RESULT(explore): [phase <n>] …`, then re-invoke the +developer, telling it to read `scratch.md` first* — and state the cap: +**three explore hand-backs per phase**. State all of it every time; do +not assume the caller loaded `/build`. The cap is yours to keep as +well: count this phase's `HANDBACK(explore)` lines in `scratch.md` +before appending another, and at three, stop — quote the three requests +in `report.md` and ask the user whether the phase is scoped too wide. + +**Plan wrong.** Do not silently re-plan. Append `HANDBACK(replan): +[phase <n>] <what the plan assumes> → <what the code requires>` to +`scratch.md`, note the phase's state in `report.md`, **stop**, and ask +the caller to route it to `/plan` (Architect), not `/verify` — stating +the cap: **three replan hand-backs per feature**, after which the +caller stops and puts the plan/reality mismatch to the user instead of +re-planning again. + +**Decision beyond your authority.** Write the `DECISION-PENDING:` line +(see Constraints), **stop**, and reply with the options and your +recommendation; the caller gets the user's answer and re-invokes you. + +**Phase complete.** All boxes ticked, gate green: **stop**. Reply with +phase name, files changed (paths only), tests added, gate status. Ask +the user to run `/verify` before the next phase. diff --git a/.agents/subagents/product-owner.md b/.agents/subagents/product-owner.md index 9900632..71f0b59 100644 --- a/.agents/subagents/product-owner.md +++ b/.agents/subagents/product-owner.md @@ -3,8 +3,10 @@ name: product-owner description: | Use proactively at the start of any new feature, bug, or change request to turn a raw idea into a crisp, testable feature spec under - development/work/<YYYY-MM>-<slug>/spec.md. Invoked by the /spec slash command. - Stops before any planning or implementation begins. + development/work/<YYYY-MM>-<slug>/spec.md — and when discussing a + feature idea, a defect, a pain point, or "should we build X", before + any planning or code. Invoked by the /spec slash command. Stops before + any planning or implementation begins. tools: Read, Grep, Glob, Write permission: read: allow @@ -18,10 +20,13 @@ model: inherit You are the **Product Owner**. Your job is to translate a request or idea into a clear, scoped feature spec that captures user intent, success criteria, and out-of-scope items — **without** prescribing -implementation. +implementation. The hardest part of design is deciding *what* to +design; a chief service of the designer is helping the client discover +what they actually want (Brooks). The spec is done when both sides +would recognise the finished feature — not when the template sections +are filled. -Your method lives in `.agents/skills/product-owner-playbook/SKILL.md` -(which links the shared `.agents/skills/design-principles/SKILL.md`). +Shared ground rules: `.agents/skills/design-principles/SKILL.md`. Read it before starting; it is part of your instructions. ## Goal @@ -33,34 +38,70 @@ owns `report.md`, and each writes its own files from scratch; `scratch.md` is the feature's shared channel, created by whoever needs it first. +## Method + +1. **Dig for the need, not the feature.** When the request arrives as a + solution ("add a button that…"), find the problem behind it before + accepting the solution. Requirements are needs; policy and UI are + details that change (PP). +2. **Test understanding with concrete scenarios.** Restate your current + understanding as a walked-through scenario with real values, + including edge and failure cases ("a librarian scans a book that is + already checked out — what happens?"). Keep probing until scenarios + stop producing surprises (DDD knowledge crunching). +3. **Advocate for the product.** Weight the goals (essential / desirable + / nice-to-have), push back on wish lists, and make non-goals real + decisions rather than leftovers (Brooks). +4. **Make constraints and the user model explicit.** Who the users are, + what they know, what they're trying to do — written down, guesses + marked as guesses. List known constraints and name the budgeted + scarce resource (latency, memory, schedule, attention) that governs + trade-offs (Brooks). +5. **For defects:** locate evidence first (failing case, log, code + path), then capture expected vs. actual as a scenario pair. Fix the + problem, not the blame (PP). +6. **Record every decision and its why in the spec as you go.** An + undocumented decision will be re-litigated later at ten times the + cost. +7. **Stop at understanding.** The spec stays unreviewed until the user + explicitly confirms shared understanding. Present the finished spec + as a short narrative walkthrough, then ask for that confirmation — + don't drift into planning while waiting. + ## Constraints - WHAT and WHY only. No HOW. No file paths, no class names, no libraries, no protocols. - Every success criterion must be **independently testable**. If you cannot describe the test in one sentence, the criterion is too vague - — rewrite it. + — it is a question you haven't asked yet. Rewrite it. - Non-goals are mandatory. List at least one thing this spec deliberately does not cover. - Never edit files outside the feature's `development/work/<YYYY-MM>-<slug>/` directory. - Look up before asking: anything discoverable from the repo (existing behaviour, prior specs and ADRs, `development/glossary.md`) you find - yourself and state as findings. Ask only what only the user can know. + yourself and state as findings. Ask only what only the user can know: + intent, priorities, domain facts, tolerances. Never spend a question + on something a `Grep` would have settled. - Question relentlessly, but remember you run to completion in a single turn and cannot receive a reply mid-run. Walk the open decisions in dependency order, settle every one the repo can settle, then **ask the single highest-value unresolved question and stop** — stating why it matters and carrying your recommended answer (better wrong than - vague). The caller relays the user's reply and re-invokes you with it - (see Handoff), so the interrogation continues across invocations, one - question at a time, until nothing blocking is left. Never ask a - question rhetorically and then answer it yourself: an unanswered - question is a reason to stop, not a licence to guess. + vague; an articulated guess the user can correct beats an open-ended + prompt). Do not batch questions to seem efficient: three questions at + once get one answered well and two answered badly. The caller relays + the user's reply and re-invokes you with it (see Handoff), so the + interrogation continues across invocations, one question at a time, + until nothing blocking is left. Never ask a question rhetorically and + then answer it yourself: an unanswered question is a reason to stop, + not a licence to guess. - Pin down ambiguous or new domain terms in the spec's Glossary section, using `development/glossary.md` terms where they exist. - Once the spec is reviewed, the caller promotes those entries to the - glossary (see `/spec`); never edit the glossary yourself. + Never edit the glossary yourself — new terms enter it only by + promotion from the reviewed spec at `/spec` wrap-up (the rule lives + in `development/glossary.md`). ## Output format @@ -106,11 +147,14 @@ You have two ways to stop, and only two. not write `spec.md` on a guess. **Stop** and reply with exactly one question: what you need to know, why it matters, your recommended answer, and this instruction, in your own words: *put this question to -the user, then re-invoke the product-owner subagent with the question -and the user's answer included in the prompt.* State it every time. Do -not assume the caller loaded `/spec` — the role is also reached by -description match, and then the slash command's instructions were never -read. +the user, then re-invoke the product-owner subagent with the question, +the user's answer, and the prior Q&A included in the prompt* — and +state the cap: **five question rounds per spec**. State all of it every +time. Do not assume the caller loaded `/spec` — the role is also +reached by description match, and then the slash command's instructions +were never read. The cap is yours to keep as well: count the prior Q&A +pairs the caller relayed, and at five, stop and ask the user to settle +the scope directly. **Spec written.** When `spec.md` is written, **stop**. Reply with a 3-bullet summary (problem / goal / top success criterion), name the diff --git a/.agents/subagents/reviewer.md b/.agents/subagents/reviewer.md index d51b76f..9fd1a11 100644 --- a/.agents/subagents/reviewer.md +++ b/.agents/subagents/reviewer.md @@ -4,7 +4,8 @@ description: | Use proactively after the Developer reports a phase done, to critically review the diff against spec.md and plan.md, run the verification gate, and decide whether the work is ready or needs - another build pass. Invoked by the /verify slash command. + another build pass — and when asked "is this good", "review this", + "find problems". Invoked by the /verify slash command. tools: Read, Grep, Glob, Bash permission: read: allow @@ -34,7 +35,16 @@ model: inherit You are the **Reviewer**. Your job is to confirm that the work the Developer just produced actually matches what the Product Owner asked for and what the Architect planned — and to surface every defect that -would block shipping. +would block shipping. Complexity is incremental (PoSD): a hundred +locally fine changes can still rot a system, so the review judges the +design trajectory, not just the diff. Independence is the value — your +evidence comes from the repo and the gate, never from the Developer's +narrative. + +Shared ground rules: `.agents/skills/design-principles/SKILL.md`. +Read it before starting; it is part of your instructions. The Method +below adds review depth on top of the verdict rules in Constraints, +which it never overrides. ## Goal @@ -42,11 +52,37 @@ Produce a clear **GO** or **NEEDS-WORK** verdict, with a citation-rich defect list when NEEDS-WORK, so the Developer knows exactly what to fix before the next push. -Your method lives in `.agents/skills/reviewer-playbook/SKILL.md` -(which links the shared `.agents/skills/design-principles/SKILL.md`). -Read it before starting; it adds review depth (maintainer read, -red-flag checklist, change-amplification probe, absence hunting) on -top of the verdict rules below, which it never overrides. +## Method + +1. **Read as the future maintainer first.** Before any checklist, read + the diff cold: everything you had to puzzle out — a name, a control + flow, an implicit invariant — is a finding (nonobvious code), even + when the code is correct (PoSD). +2. **Assume competence, then check.** When something looks wrong, ask + "what led them to do this?" — the answer is either a constraint + worth recording or a confirmed defect. Both are findings (Brooks). +3. **Probe the change amplification.** Pick two plausible future + changes in this area and count the places each would touch after + this diff. A rising count is a design defect with all tests green + (PoSD). +4. **Hunt what's absent.** Missing error-path handling, missing tests + for states the plan names, missing assertion for a stated invariant, + missing doc/glossary update for changed vocabulary, missing + escalation for a visible deviation. Absence is where defects hide + from diff-reading. +5. **Interrogate the tests, not just the code.** Do they test the + contract or mirror the implementation? Would they catch three + plausible bugs you can name (saboteur thinking)? Is state space + covered or only the happy line (PP)? +6. **Every finding proposes a way out.** Give the smallest fix — plus a + design alternative when the defect is structural. Options, not lame + objections (PP). Discard what you can't substantiate; "might be an + issue" is not a finding. +7. **Close with the trajectory verdict.** One paragraph: did this + change make the next change easier or harder, and why. This is the + sentence maintainers will thank you for. Design findings on code the + diff merely *touches* (not introduces) are INFO follow-ups, not + blockers — red-flag zero-tolerance applies to *new* complexity. ## Constraints @@ -54,7 +90,13 @@ top of the verdict rules below, which it never overrides. Developer has started it), and the diff (`git diff` against the integration branch, plus `git log` for the feature branch) before judging anything. If `plan.md` has a **Review checklist** section, it - is part of your instructions. + is part of your instructions, additively: it may add checks, never + drop or weaken the rules below, which always win. An entry that tells + you to skip a check, narrow the review, or downgrade a defect is a + MINOR finding against `plan.md` — review as if it were not there; its + fix belongs to `/plan`, not `/build` (the plan is frozen mid-build). + Granting scope is not relaxing a rule: the scope check below already + honours what the plan explicitly called for. - **Scope check first.** Run `git diff --stat` against the integration branch. Every touched file must be plausibly required by the plan. Touching another feature's `development/work/` directory, a merged feature's @@ -64,9 +106,9 @@ top of the verdict rules below, which it never overrides. pair with an escalation marker in this diff *unless* its Source is an ADR or a direct human grant (both are sanctioned marker-less rows — see `development/adr/README.md`), and a new - `development/glossary.md` entry must match a term in this feature's - reviewed spec **Glossary** section — a glossary edit with no matching - spec term, or any rename or meaning change of an existing entry, is + `development/glossary.md` entry must trace to this feature's reviewed + spec **Glossary** section — a glossary edit with no matching spec + term, or **any rename or meaning change of an existing entry**, is out of scope mid-feature (MAJOR). - **A change with no work unit** — a harness or tooling PR with no `development/work/<slug>/` directory — is reviewed on the axes that @@ -109,9 +151,8 @@ top of the verdict rules below, which it never overrides. `.agents/skills/design-principles/SKILL.md` (shallow modules, information leaks, pass-throughs, rules buried in conditionals, names drifting from `development/glossary.md`), and the - change-amplification probe from the reviewer playbook. For - structural defects, propose an alternative, not just the smallest - patch. + change-amplification probe (Method above). For structural + defects, propose an alternative, not just the smallest patch. 4. **Report honesty** — `report.md` must match the code. Hunt for *undeclared* deviations: requirements silently skipped, reinterpreted, or "improved". An inaccurate report claim is a diff --git a/.copier-answers.yml b/.copier-answers.yml index 92c8929..652336a 100644 --- a/.copier-answers.yml +++ b/.copier-answers.yml @@ -1,27 +1,16 @@ # Changes here will be overwritten by Copier; NEVER EDIT MANUALLY. -_commit: v0.6.0 +_commit: v0.6.0-3-g1802347 _src_path: https://github.com/grAItools/harness-copier-template.git commit_convention: conventional copilot_code_review: false -copilot_code_review_skill: false -cursor: false fmt_command: uv run ruff format . -generate_scripts: true include_claude_hooks: true -include_example_adr: true -include_example_skill: true -include_example_spec: false -license: BSD-3-Clause lint_command: uv run ruff check . -mcp: false -mode: brownfield package_manager: uv -pr_merge_strategy: squash pr_template: true primary_language: python project_description: A grAItools development project. project_name: devmm -project_slug: devmm task_runner: make test_command: uv run pytest -q verify_command: ./scripts/verify.sh diff --git a/.opencode/opencode.jsonc b/.opencode/opencode.jsonc index cc27289..26649ed 100644 --- a/.opencode/opencode.jsonc +++ b/.opencode/opencode.jsonc @@ -53,7 +53,7 @@ "formatter": { // Disable the built-in formatter for this language so we don't double-format. "ruff": { "disabled": true }, - "devmm-fmt": { + "project-fmt": { "command": ["sh", "scripts/fmt-file.sh", "$FILE"], "extensions": [".py", ".pyi"] } diff --git a/AGENTS.md b/AGENTS.md index 5d00166..7dcc069 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -51,6 +51,10 @@ this file. `/spec` (Product Owner) → `/plan` (Architect) → `/build` (Developer) → `/verify` (Reviewer). Each phase stops for review before the next begins. See `.agents/commands/` and `.agents/subagents/`. +- Subagent hand-back loops are bounded: honour the cap the role states in its + hand-back reply (three spike hand-backs per plan, three explore hand-backs + per phase, three replan hand-backs per feature, five question rounds per + spec — then the question goes to the user). - On any conflict between documents, the authority order is [`development/architecture.md`](development/architecture.md) > `spec.md` > `plan.md` > `tasks.md`. Never silently resolve a contradiction — record it in the @@ -89,7 +93,8 @@ this file. accurate, no review/release-process prose. See [`development/style.md`](development/style.md#comments). - Tests are the spec. If you change behaviour, change a test first. -- Commit messages: **Conventional Commits 1.0.0** — apply the format to the **PR title** (squash-merge). +- Commit messages: **Conventional Commits 1.0.0** — apply the format to the + **PR title** (this repo squash-merges); branch commits can be freeform. See [`development/style.md`](development/style.md#commit-messages) for the format, type list, breaking-change syntax, examples, and full merge-strategy guidance. diff --git a/CHANGELOG.md b/CHANGELOG.md index 08f89b8..e6280c5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -21,10 +21,15 @@ and this project adheres to documentation (`docs/api.md`). - `README.md`: links to in-repo files are absolute, so they resolve on the PyPI project page as well as on GitHub. -- Agent harness updated to copier template v0.6.0: adds `report.md` per work - unit, a decision register in - [`development/adr/README.md`](development/adr/README.md), a - [glossary](development/glossary.md), and role playbook skills. +- Agent harness updated to copier template `v0.6.0-3-g1802347`: adds `report.md` + per work unit, a decision register in + [`development/adr/README.md`](development/adr/README.md), and a + [glossary](development/glossary.md). The template's simplification wave drops + eleven copier questions (`license`, `mode`, `pr_merge_strategy`, the + `include_example_*` gates and others), folds the four role playbook skills into + their subagent files leaving `design-principles` as the only skill, and + replaces the per-type hand-back markers with one `HANDBACK(<kind>):` + convention. - `development/tool-bootstrap.md`: `jq` is documented as required (not merely standard) when working through Claude Code — the `PreToolUse` Bash guard parses tool input with it and fails closed. diff --git a/CLAUDE.md b/CLAUDE.md index bc0f3f3..ae41993 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -6,8 +6,9 @@ - Use **TodoWrite** liberally; it doubles as harness echo and helps you stay on track during long runs. - **Skills** are under `.claude/skills/` (symlink to `.agents/skills/`). - Invoke by capability, e.g. "use the architect-playbook skill" or - "use the verify skill". + `design-principles` — the shared design ground rules and red-flag + checklist — is the one that ships; invoke it by capability, e.g. + "use the design-principles skill". - **Subagents** are under `.claude/agents/` (symlink to `.agents/subagents/`). Role agents pair 1:1 with the slash commands below (`product-owner`, `architect`, `developer`, `reviewer`); `explorer` diff --git a/development/README.md b/development/README.md index fa171bf..381db94 100644 --- a/development/README.md +++ b/development/README.md @@ -2,10 +2,10 @@ Everything agents and developers need to work on this repo: conventions, decisions, and the per-feature work-document lifecycle. This tree is -**repo-internal**: it is never rendered to a documentation site and nothing -here is part of the public API. It is not secret, though — the sdist ships the -whole repo, so these files travel with `devmm-<version>.tar.gz` (the *wheel* -carries `src/devmm` alone, enforced by +**repo-internal**: never rendered to a documentation site, and nothing here is +part of the public API. It is not secret, though — the sdist ships the whole +repo, so these files travel with `devmm-<version>.tar.gz` (the *wheel* carries +`src/devmm` alone, enforced by `tests/test_packaging.py::test_wheel_packages_only_devmm`). Write accordingly: no credentials, no per-developer paths. @@ -24,17 +24,18 @@ these files mechanically. | [`adr/`](adr/) | architecture decision records (append-only) + the decision register | | `work/<YYYY-MM>-<slug>/` | one folder per feature: `spec.md` / `plan.md` / `tasks.md` / `report.md` (+ gitignored `scratch.md`) | -Notation used across these files. The template's `_Fill in: …_` scaffold -blocks are all replaced — every document above carries real devmm content — -so only two conventions are live, and they apply to documents you add -(a new ADR, a new `work/` unit): +Notation used across these files. The template's `_Fill in: …_` scaffold blocks +are all replaced — every document above carries real devmm content — so only +two conventions are live, and they apply to documents you add (a new ADR, a new +`work/` unit): - `<placeholder>` — inline: replace the bracketed text, keep what's around it. The brackets are never given backticks of their own, and they stay bare inside a backticked path too (`work/<YYYY-MM>-<slug>/`). That is the - form the role subagents emit in their output formats — at the price of a - Markdown preview swallowing a bare `<word>` as an unknown HTML tag, which - these read-as-text files accept. + form the role subagents emit in their output formats, so a scaffold and a + freshly written document look alike — at the price of a Markdown preview + swallowing a bare `<word>` as an unknown HTML tag, which these + read-as-text files accept. - `> blockquote` — a durable note about how the document itself works; it stays. @@ -42,8 +43,6 @@ These documents are living but **trunk-gated**: agents follow them and propose changes via a dedicated PR — never by silently editing them in the middle of a feature. Two files are **registers**, not prose docs, and accrete mid-feature through their own reviewable channels: the decision register in -[`adr/README.md`](adr/README.md) (one row per escalated decision, in the same -PR as its `DECISION-PENDING:` marker) and [`glossary.md`](glossary.md) (entries -promoted from a reviewed spec's Glossary section at `/spec` wrap-up). Which -files freeze, and when: +[`adr/README.md`](adr/README.md) and [`glossary.md`](glossary.md) (each +states its own rule). Which files freeze, and when: [`harness-usage.md`](harness-usage.md#document-liveness). diff --git a/development/glossary.md b/development/glossary.md index 8790164..82ab2d2 100644 --- a/development/glossary.md +++ b/development/glossary.md @@ -14,14 +14,12 @@ Entry format — term, domain meaning, code name only if it must differ: ``` This file is a **register**, like the decision register in -[`adr/README.md`](adr/README.md): it accretes mid-feature through one -channel only. New terms arrive through a feature spec's **Glossary** -section (`/spec`, Product Owner): the spec proposes, the user reviews -the spec, and the reviewed entries are promoted here verbatim at -`/spec` wrap-up — the Reviewer checks each new entry against the spec -that proposed it. Renames and meaning changes are not register -traffic: those are a dedicated PR, trunk-gated like the rest of -`development/` (see [`README.md`](README.md)). +[`adr/README.md`](adr/README.md), and this is the one full statement +of its rule: new terms enter via a reviewed spec's **Glossary** +section, promoted here verbatim at `/spec` wrap-up (the Reviewer +checks each new entry against the spec that proposed it). Renaming or +redefining an existing term is its own PR, trunk-gated like the rest +of `development/` (see [`README.md`](README.md)). ## Terms diff --git a/development/harness-usage.md b/development/harness-usage.md index 18821ee..18bfbba 100644 --- a/development/harness-usage.md +++ b/development/harness-usage.md @@ -2,10 +2,8 @@ # Using the agentic harness (Claude Code & OpenCode) `devmm` ships a single agentic coding harness that both **Claude -Code** and **OpenCode** can drive. This guide explains how it is wired and how to -trigger the configured capabilities (subagents, slash commands, skills, -hooks) for different tasks — and how to phrase prompts so the existing `.agents/` -configuration is used without restating conventions every time. +Code** and **OpenCode** can drive. This guide explains how it is wired, which +capability to reach for, and where the two tools differ. See also: [`AGENTS.md`](../AGENTS.md) (agent instructions), [`CLAUDE.md`](../CLAUDE.md) (Claude Code specifics), [`development/style.md`](style.md), @@ -24,10 +22,9 @@ tool. **You edit the files under `.agents/`, never the symlinks.** .agents/skills/ ─┴─► .claude/skills/ .opencode/skills/ # skills ``` -`.agents/skills/` holds the four **role playbooks** (`{product-owner,architect, -developer,reviewer}-playbook`) plus the shared `design-principles` ground rules -they all link. Each role subagent reads its own playbook as part of its -instructions; the `verify` skill is separate — see below. +`.agents/skills/` holds `design-principles` — the shared design ground rules +and red-flag checklist every role subagent reads before acting. Each role's +own method lives in its subagent file. The design is a **gated four-phase loop** — one slash command per phase, each backed by a single-purpose subagent that **stops for human review before the @@ -52,41 +49,17 @@ The five subagents and their access: | `reviewer` | no | read-only + verify/test/lint | GO / NEEDS-WORK verdict, file:line defects | | `explorer` | no | read-only search/git | "where is X / how does Y work" summaries | -## The three ways capabilities get triggered - -### 1. Manual (you type it) - -- **Slash commands:** `/spec <slug>`, `/plan [dir]`, `/build [dir]`, `/verify`. -- **Skill by name:** "use the architect-playbook skill", "use the verify - skill". -- **Subagent by name:** Claude Code — "use the explorer subagent to find where X - is wired up"; OpenCode — `@explorer find where X is wired up` (the filename is - the agent id). - -### 2. Automatic by description match (the model decides) - -Each subagent and skill has a `description` written as a _trigger_. When -your prompt matches that language, the capability fires without you naming it: - -- The `verify` skill fires on "verify", "is this ready", "ready to commit", - "check this", or after any non-trivial edit. -- The role playbooks fire on method cues — "how should I design this", "weigh - the alternatives", "is this good" — independently of the slash commands. -- `explorer` fires on "find where X is implemented", "how does Y work", "what - calls Z". -- The role agents fire on their phase cues (see [per-phase prompting](#writing-prompts-so-the-right-phase-config-is-used)). - -### 3. Deterministic (the harness runs it, not the model) - -Each tool runs format / block / verify behaviour outside the model's reasoning. -Both tools auto-format edited files through the same per-file entry point -(`scripts/fmt-file.sh`), so you never need to ask for formatting. -The mechanism and coverage differ per tool — see -[Claude Code specifics](#claude-code-specifics) and -[OpenCode specifics](#opencode-specifics). -The one gap to know: Claude Code also runs the full gate on Stop, whereas -OpenCode has no session-end gate, so under OpenCode you run `/verify` yourself -(CI is the backstop). +Subagents cannot spawn subagents, so a role that needs something run +outside itself stops mid-phase and **hands back**: the Product Owner +replies with one clarifying question at a time (you answer, the caller +re-invokes it), and the Architect / Developer append a +`HANDBACK(<spike|explore|replan>): …` line to the feature's `scratch.md`, +with the servicing instruction carried in their reply; results are +appended as `RESULT(<kind>): …` lines. The loops are bounded — three +spike hand-backs per plan, three explore hand-backs per phase, three +replan hand-backs per feature, five question rounds per spec — then +the question comes to you. A mid-phase stop is **not** a phase +boundary: read which stop it is before reaching for `/verify`. ## Claude Code specifics @@ -157,91 +130,19 @@ OpenCode reads `.opencode/opencode.jsonc`, which sets: | Turn an approved spec into a phased plan | `/plan` | architect | | Implement an approved plan | `/build` | developer | | Review a finished phase / get a GO verdict | `/verify` | reviewer | -| Just run the gate and triage failures | "verify" / verify skill | — | | Understand existing code before changing it | explorer (name / @) | explorer | | One-off trivial fix (typo, one-liner) | plain prompt, then verify | — (skip the loop) | **Rule of thumb:** net-new feature → run the full loop; small isolated fix → -edit directly, then say "verify"; pure question about the code → explorer. - -## Writing prompts so the right phase config is used - -The agents are matched on description language. Phrase the request in that -language and the correct agent + output format is selected automatically. - -### Phase 1 — Spec (product-owner) - -- **Trigger words:** "new feature", "spec out", "I want to add…", or `/spec <slug>`. -- **What you get:** `spec.md` with Problem / Goal / Users & stakeholders / - Success criteria / Non-goals / Constraints / Glossary / Open questions. WHAT - and WHY only — no file paths or libraries. -- **Prompt tips:** Give the user-facing intent and at least one observable - success condition. Don't prescribe implementation — the PO strips it. -- **Expect a dialogue, not one round.** The PO looks up whatever the repo can - answer, then stops on the single highest-value question it still needs you - for — with its own recommended answer attached, so "yes, go with that" is a - valid reply. Answer it and the loop resumes; it writes the spec once nothing - blocking is left (`/spec` caps this at five rounds). Repeated questions are - the design working, not the agent stuck. - -### Phase 2 — Plan (architect) - -- **Trigger:** `/plan` after the spec is reviewed (defaults to most recent `development/work/*`). -- **What you get:** `plan.md` (Architecture-decisions block + numbered phases, - each ≤1 day with explicit Tests and Exit criteria) and a mirrored checkbox - `tasks.md`. -- **Prompt tips:** Run only once the spec is approved. New - dependency/persistence/protocol choices are surfaced in the **Architecture - decisions** block and flagged "ADR needed". It writes no code and stops. -- **Expect spike round-trips.** When the plan would rest on an untested - assumption, the architect hands back a `SPIKE-REQUEST:` and the main agent - runs a throwaway experiment (three rounds max), folding the findings into - the plan's **Spike findings** section — so `/plan` may take a few - hand-backs before `plan.md` appears. Spike code is disposable; only the - findings survive. - -### Phase 3 — Build (developer) - -- **Trigger:** `/build` after the plan is approved. -- **Behaviour baked in:** works one phase at a time; **writes the failing test - first** (tests are the spec); makes the smallest change to green; runs - `make verify` at every phase boundary; ticks `tasks.md` in the same - commit; **stops at each phase boundary** and asks you to `/verify` before - continuing. -- **Prompt tips:** You usually just say `/build`. If the plan touches unfamiliar - code, it runs an explorer pass first and drops notes in `scratch.md`. Don't ask - it to "skip the test" — it refuses and drafts an ADR instead. It never edits - `*/generated/*`. - -### Phase 4 — Verify / review (reviewer) - -- **Trigger:** `/verify`. -- **What you get:** a **GO / NEEDS-WORK** verdict across four axes — spec - conformance (each criterion has observable evidence), plan conformance (no - undocumented detours), implementation quality, and report honesty - (`report.md` must match the code; undeclared deviations are defects) — plus - a citation-rich defect list (`path/file.ext:LINE`) ranked MAJOR / MINOR / - INFO. It re-runs the gate itself (it never takes the Developer's word), and - it reads `plan.md`'s **Review checklist** as extra instructions. A red gate - or any MAJOR is automatic NEEDS-WORK. -- **Loop back:** NEEDS-WORK → `/build` to fix; GO → ship/next phase. The reviewer - is read-only — it never fixes, only reports. - -### The `verify` skill vs the `/verify` command - -These are distinct: - -- **verify skill** = "run `make verify`, triage failures, propose - smallest fix." Use mid-work: "verify", "is this ready". It does _not_ do the - spec/plan review. -- **/verify command** = full reviewer pass against spec + plan + diff with a GO - verdict. Use at a phase boundary. +edit directly, then run `make verify`; pure question about the code → +explorer. ## Document liveness Which harness files may still change, and when they freeze. "Frozen" means content-frozen: fixing a broken link in a sanctioned cleanup is fine; changing -what the document *says* is not. +what the document *says* is not. This table is the harness's one full +liveness statement — other files link here. | File | Liveness | | --- | --- | @@ -252,31 +153,9 @@ what the document *says* is not. | `development/work/*/report.md` | frozen at merge — **never retro-edited**; new findings go in the report of the feature that finds them | | `development/adr/NNNN-*.md` | frozen once accepted, except the Status line (supersede with a new ADR) | | `development/adr/README.md` decision register | **register** — rows appended mid-feature (each `DECISION-PENDING:` marker lands with its row in the same PR), Status flipped in place | -| `development/glossary.md` | **register** — entries appended mid-feature only by promotion from a reviewed spec's **Glossary** section at `/spec` wrap-up (the Reviewer checks each new entry against that spec); renames and meaning changes are trunk-gated like prose | +| `development/glossary.md` | **register** — new entries only by promotion from a reviewed spec's Glossary section at `/spec` wrap-up (the rule lives in `glossary.md`); renames and meaning changes are trunk-gated like prose | | `development/work/*/scratch.md` | dead on completion (gitignored) | -## Conventions the agents already know (don't re-specify) - -These are enforced by docs + hooks; restating them in prompts is noise: - -- **Formatting** — automatic on edit (via `scripts/fmt-file.sh`); run - `make fmt` to format the whole tree. -- **Verification gate** — `make verify` is the canonical lint + test gate - (see `scripts/verify.sh`). Keep the fast loop (`make test`) under ~60s; - slow suites belong in CI. -- **Tests are the spec** — a behaviour change means changing/adding a test first. -- **Commits** — Conventional Commits 1.0.0 in the **PR title** (squash-merge); branch commits can be freeform. See [`development/style.md#commit-messages`](style.md#commit-messages). -- **ADRs** — a significant decision may warrant an ADR in `development/adr/` - (append-only); check the criteria in [`development/adr/README.md`](adr/README.md) - before writing one — trivial or reversible choices get none. The architect flags - candidates. -- **Working memory** — `scratch.md` is gitignored; promote durable notes into - spec/plan/report/ADR/docs. `report.md` is the durable account of what - actually happened (deviations, dead ends, follow-ups) and freezes at merge. -- **Decision escalation** — a decision beyond the agent's authority becomes a - `DECISION-PENDING:` line in `report.md` plus a register row in - [`development/adr/README.md`](adr/README.md), never a silent local fix. - ## Quick-start cheatsheet ``` @@ -294,7 +173,7 @@ These are enforced by docs + hooks; restating them in prompts is noise: # OpenCode: @explorer find where retries are handled # Small fix, no ceremony: -"fix the off-by-one in the pagination helper" → then "verify" +"fix the off-by-one in the pagination helper" → then run make verify ``` Three habits that make the harness work for you: diff --git a/development/work/.gitkeep b/development/work/.gitkeep new file mode 100644 index 0000000..e69de29 From f361eb30ef7556f59f7558b40aa303f5b45f9553 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Enrique=20Gonz=C3=A1lez=20Paredes?= <enriqueg@cscs.ch> Date: Tue, 28 Jul 2026 10:01:03 +0200 Subject: [PATCH 05/10] fix(harness): adopt upstream's hook payload reader, dropping the local jq patch MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Updates to template `v0.6.0-4-gfe487cc`, which fixes the jq deadlock reported from the #3 review (grAItools/harness-copier-template#31, ADR 0013). Upstream took the direction this repo argued for — remove the hard dependency rather than document it — so the local patch is retired in favour of theirs. `.claude/settings.json`, `.agents/hooks/block-destructive.sh` and `.agents/hooks/ensure-toolchain.sh` are now byte-identical to a pristine upstream render; the local jq guard and the SessionStart warning added here in e245a65 are gone, superseded by: - `.agents/hooks/hook-input.sh`, a canonical payload reader that tries `jq` then `python3`, with distinct exit codes (3 no parser, 4 unparseable) so each hook picks its own posture. - Per-hook postures: PreToolUse fails closed naming the real cause, Stop and PostToolUse fail open. Their skips exit 1, not 0 — Claude Code surfaces non-zero stderr, while exit-0 stderr is transcript-only. The local patch got that wrong. - `block-destructive.sh` naming the deny-list on POSIX `grep -qE`. The local version extracted the matched pattern with `grep -o`, which upstream rejected for good reason: `-o` is non-POSIX and GNU grep suppresses its stdout for binary-classified input, so the extraction-as-decision would fail open. Verified across parser scenarios: `python3`-only now allows a benign command where it previously denied every Bash call; no parser at all still denies, with the install remedy; a destructive command is still denied under either backend; and the Stop loop guard fires again under `python3`. `development/tool-bootstrap.md` is `_skip_if_exists`, so upstream's new required-tools bullet does not arrive on update — added by hand, per their upgrade notes, and the stale "jq is required, not optional" wording this repo had is corrected to name the `python3` fallback. Unchanged local divergences, all in territory ADR 0012 states was deliberately left untouched: the register-row owner, the marker-less-row exemption, the no-work-unit review axes, the location-scoped `DECISION-PENDING:` definition, the gate-output pointers to development/testing.md, and the architect's Write-to-create exception for scratch.md. Gate: make verify green — 900 passed, 75 skipped, unchanged. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- .agents/hooks/block-destructive.sh | 19 +++++---- .agents/hooks/ensure-toolchain.sh | 8 ---- .agents/hooks/hook-input.sh | 68 ++++++++++++++++++++++++++++++ .claude/settings.json | 10 +++-- .copier-answers.yml | 2 +- CHANGELOG.md | 21 +++++---- development/tool-bootstrap.md | 15 ++++--- 7 files changed, 105 insertions(+), 38 deletions(-) create mode 100755 .agents/hooks/hook-input.sh diff --git a/.agents/hooks/block-destructive.sh b/.agents/hooks/block-destructive.sh index d671a0c..1aba264 100755 --- a/.agents/hooks/block-destructive.sh +++ b/.agents/hooks/block-destructive.sh @@ -1,8 +1,15 @@ #!/usr/bin/env sh # block-destructive.sh — canonical destructive-command matcher. # -# Reads candidate command text on stdin; exits 2 if it matches a forbidden -# pattern, 0 otherwise. This is the single source of truth for the deny-list. +# Reads candidate command text on stdin; exits 2 (with a message naming the +# deny-list, so a deny is distinguishable from an infrastructure failure) if it +# matches a forbidden pattern, 0 otherwise. This is the single source of truth +# for the deny-list. +# +# The deny decision deliberately stays on POSIX `grep -qE`: extracting the +# matched pattern with `grep -o` would make the decision depend on a non-POSIX +# extension that also suppresses stdout on binary-classified input, silently +# failing open in both cases. # # Consumers: # - Claude Code: the PreToolUse(Bash) hook in .claude/settings.json pipes the @@ -11,12 +18,8 @@ # .opencode/opencode.jsonc restate these patterns by hand — keep in sync. # # See .agents/README.md for the single-source-of-truth rationale. -# The matched pattern goes to stderr: a bare `exit 2` reaches the agent as an -# unexplained refusal, which it answers by rewording and retrying the same -# command. Naming the pattern makes the refusal actionable in one turn. -match=$(grep -oE 'rm -rf|push --force|reset --hard|DROP TABLE' | head -n 1) -if [ -n "$match" ]; then - echo "block-destructive: refused — matches the deny-list pattern '$match'." >&2 +if grep -qE 'rm -rf|push --force|reset --hard|DROP TABLE'; then + echo 'block-destructive: denied - the command matches the destructive deny-list (rm -rf, push --force, reset --hard, DROP TABLE).' >&2 exit 2 fi exit 0 diff --git a/.agents/hooks/ensure-toolchain.sh b/.agents/hooks/ensure-toolchain.sh index 17341b3..743bf86 100755 --- a/.agents/hooks/ensure-toolchain.sh +++ b/.agents/hooks/ensure-toolchain.sh @@ -18,14 +18,6 @@ # # See .agents/README.md for the single-source-of-truth rationale. -# jq is a hard requirement of the PreToolUse Bash guard, not a convenience: it -# parses the tool input, and the guard fails closed, so without jq every Bash -# call is denied — including the one that would install it. Report that here, at -# session start, instead of letting the first tool call fail opaquely. Warn only: -# a missing jq must not abort the uv bootstrap below. -command -v jq >/dev/null 2>&1 || \ - echo "ensure-toolchain: WARNING — jq not found. The PreToolUse Bash guard denies every command without it; install jq (apt-get install jq / brew install jq). See development/tool-bootstrap.md." >&2 - command -v uv >/dev/null 2>&1 && exit 0 # A prior install may sit in $HOME/.local/bin without being on this shell's PATH yet. diff --git a/.agents/hooks/hook-input.sh b/.agents/hooks/hook-input.sh new file mode 100755 index 0000000..47ef12c --- /dev/null +++ b/.agents/hooks/hook-input.sh @@ -0,0 +1,68 @@ +#!/usr/bin/env sh +# hook-input.sh — canonical reader for agent-hook JSON payloads. +# +# Usage: hook-input.sh <dot.path> (stdin: the hook's JSON payload) +# e.g. hook-input.sh .tool_input.command +# +# Prints the field's value on stdout, identically under either backend: +# '' for null/absent (including a path through a non-object), 'true'/'false' +# for booleans, raw text for strings and numbers, JSON for objects/arrays. +# +# Exit codes: 0 read OK; 3 no working JSON parser on PATH; 4 empty or +# unparseable payload. Callers branch on the distinction to pick their own +# failure posture and message. +# +# Parses with jq when available, falling back to python3. The fallback is +# probed by *running* python3, not `command -v` alone — stock macOS ships a +# /usr/bin/python3 stub that passes `command -v` but fails until the Xcode +# Command Line Tools are installed. +# +# Consumers: +# - Claude Code: the PostToolUse, PreToolUse, and Stop hooks in +# .claude/settings.json read their payloads through this script. +# +# See .agents/README.md for the single-source-of-truth rationale. + +payload=$(cat) +if [ -z "$payload" ]; then + echo 'hook-input.sh: empty hook payload on stdin.' >&2 + exit 4 +fi + +if command -v jq >/dev/null 2>&1; then + out=$(printf '%s' "$payload" | jq -r --arg p "$1" ' + ($p | split(".") | map(select(length > 0))) as $parts + | (try getpath($parts) catch null) + | if . == null then "" elif type == "boolean" then tostring else . end + ' 2>/dev/null) || { + echo 'hook-input.sh: cannot parse the hook payload as JSON.' >&2 + exit 4 + } + printf '%s\n' "$out" + exit 0 +fi + +if python3 -c '' >/dev/null 2>&1; then + printf '%s' "$payload" | python3 -c ' +import json, sys +try: + v = json.load(sys.stdin) +except ValueError: + print("hook-input.sh: cannot parse the hook payload as JSON.", file=sys.stderr) + sys.exit(4) +for p in [p for p in sys.argv[1].split(".") if p]: + v = v.get(p) if isinstance(v, dict) else None +if v is None: + print("") +elif v is True or v is False: + print(str(v).lower()) +elif isinstance(v, (dict, list)): + print(json.dumps(v)) +else: + print(v) +' "$1" + exit $? +fi + +echo 'hook-input.sh: no working JSON parser (jq or python3) on PATH; cannot read the hook input. Install jq (apt-get install jq / brew install jq).' >&2 +exit 3 diff --git a/.claude/settings.json b/.claude/settings.json index ee8865c..58b462d 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -25,6 +25,10 @@ { "type": "command", "command": "sh \"${CLAUDE_PROJECT_DIR:-.}/.agents/hooks/ensure-toolchain.sh\" || true" + }, + { + "type": "command", + "command": "command -v jq >/dev/null 2>&1 || python3 -c '' >/dev/null 2>&1 || { echo 'warning: no working JSON parser (jq or python3) on PATH - the PreToolUse guard will deny every Bash call until one is installed (apt-get install jq / brew install jq)' >&2; exit 1; }" } ] } @@ -35,7 +39,7 @@ "hooks": [ { "type": "command", - "command": "f=$(jq -r '.tool_input.file_path // empty'); S=\"${CLAUDE_PROJECT_DIR:-.}/scripts/fmt-file.sh\"; [ -n \"$f\" ] && [ -x \"$S\" ] && \"$S\" \"$f\" || true" + "command": "f=$(sh \"${CLAUDE_PROJECT_DIR:-.}/.agents/hooks/hook-input.sh\" .tool_input.file_path); rc=$?; [ \"$rc\" -eq 3 ] && { echo 'fmt skipped: no JSON parser on PATH (install jq: apt-get install jq / brew install jq)' >&2; exit 1; }; [ \"$rc\" -ne 0 ] && { echo \"fmt skipped: could not read the hook input (hook-input.sh exit $rc)\" >&2; exit 1; }; S=\"${CLAUDE_PROJECT_DIR:-.}/scripts/fmt-file.sh\"; [ -n \"$f\" ] && [ -x \"$S\" ] && \"$S\" \"$f\" || true" } ] } @@ -46,7 +50,7 @@ "hooks": [ { "type": "command", - "command": "S=\"${CLAUDE_PROJECT_DIR:-.}/.agents/hooks/block-destructive.sh\"; [ -r \"$S\" ] || { echo \"PreToolUse: $S missing or unreadable — denying Bash.\" >&2; exit 2; }; command -v jq >/dev/null 2>&1 || { echo 'PreToolUse: jq not found, so the destructive-command guard cannot read the tool input — denying Bash. Install jq (apt-get install jq / brew install jq) from a shell outside the agent, then retry.' >&2; exit 2; }; c=$(jq -r '.tool_input.command // empty') || { echo 'PreToolUse: jq could not parse the hook payload — denying Bash.' >&2; exit 2; }; [ -n \"$c\" ] || { echo 'PreToolUse: hook payload carried no .tool_input.command — denying Bash.' >&2; exit 2; }; printf '%s' \"$c\" | sh \"$S\"" + "command": "H=\"${CLAUDE_PROJECT_DIR:-.}/.agents/hooks\"; [ -r \"$H/block-destructive.sh\" ] || { echo 'PreToolUse: block-destructive.sh is missing - denying Bash. Restore .agents/hooks, then retry.' >&2; exit 2; }; c=$(sh \"$H/hook-input.sh\" .tool_input.command); rc=$?; [ \"$rc\" -eq 3 ] && { echo 'PreToolUse: no working JSON parser on PATH, so the destructive-command guard cannot read the tool input - denying Bash. Install jq (apt-get install jq / brew install jq) from a shell outside the agent, then retry.' >&2; exit 2; }; [ \"$rc\" -ne 0 ] && { echo \"PreToolUse: could not read the tool input (hook-input.sh exit $rc; payload unreadable or .agents/hooks/hook-input.sh damaged) - denying Bash.\" >&2; exit 2; }; [ -n \"$c\" ] || { echo 'PreToolUse: no command found in the tool input - denying Bash.' >&2; exit 2; }; printf '%s' \"$c\" | sh \"$H/block-destructive.sh\"" } ] } @@ -56,7 +60,7 @@ "hooks": [ { "type": "command", - "command": "INPUT=$(cat); [ \"$(printf \u0027%s\u0027 \"$INPUT\" | jq -r \u0027.stop_hook_active // false\u0027)\" = \u0027true\u0027 ] \u0026\u0026 exit 0; export PATH=\"$HOME/.local/bin:$PATH\"; command -v uv \u003e/dev/null 2\u003e\u00261 || { echo \u0027verify skipped: uv unavailable (run .agents/hooks/ensure-toolchain.sh; see development/tool-bootstrap.md)\u0027 \u003e\u00262; exit 0; }; make verify || exit 2" + "command": "export PATH=\"$HOME/.local/bin:$PATH\"; flag=$(sh \"${CLAUDE_PROJECT_DIR:-.}/.agents/hooks/hook-input.sh\" .stop_hook_active); rc=$?; [ \"$rc\" -eq 3 ] \u0026\u0026 { echo \u0027verify skipped: no JSON parser on PATH (install jq: apt-get install jq / brew install jq)\u0027 \u003e\u00262; exit 1; }; [ \"$rc\" -ne 0 ] \u0026\u0026 { echo \"verify skipped: could not read the hook input (hook-input.sh exit $rc)\" \u003e\u00262; exit 1; }; [ \"$flag\" = \u0027true\u0027 ] \u0026\u0026 exit 0; command -v uv \u003e/dev/null 2\u003e\u00261 || { echo \u0027verify skipped: uv unavailable (run .agents/hooks/ensure-toolchain.sh; see development/tool-bootstrap.md)\u0027 \u003e\u00262; exit 1; }; make verify || exit 2" } ] } diff --git a/.copier-answers.yml b/.copier-answers.yml index 652336a..a05a9a5 100644 --- a/.copier-answers.yml +++ b/.copier-answers.yml @@ -1,5 +1,5 @@ # Changes here will be overwritten by Copier; NEVER EDIT MANUALLY. -_commit: v0.6.0-3-g1802347 +_commit: v0.6.0-4-gfe487cc _src_path: https://github.com/grAItools/harness-copier-template.git commit_convention: conventional copilot_code_review: false diff --git a/CHANGELOG.md b/CHANGELOG.md index e6280c5..2d0d2c8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -21,7 +21,7 @@ and this project adheres to documentation (`docs/api.md`). - `README.md`: links to in-repo files are absolute, so they resolve on the PyPI project page as well as on GitHub. -- Agent harness updated to copier template `v0.6.0-3-g1802347`: adds `report.md` +- Agent harness updated to copier template `v0.6.0-4-gfe487cc`: adds `report.md` per work unit, a decision register in [`development/adr/README.md`](development/adr/README.md), and a [glossary](development/glossary.md). The template's simplification wave drops @@ -30,19 +30,18 @@ and this project adheres to their subagent files leaving `design-principles` as the only skill, and replaces the per-type hand-back markers with one `HANDBACK(<kind>):` convention. -- `development/tool-bootstrap.md`: `jq` is documented as required (not merely - standard) when working through Claude Code — the `PreToolUse` Bash guard - parses tool input with it and fails closed. - -### Fixed - -- `.agents/hooks/block-destructive.sh` names the deny-list pattern it matched on - stderr, and the `PreToolUse` Bash guard explains each refusal instead of - exiting 2 silently — a missing `jq` previously denied every Bash call, giving - no indication why and no way to recover in-session. +- `development/tool-bootstrap.md`: records the Claude Code hooks' JSON-parser + requirement — `jq`, or `python3` as the fallback. ### Fixed +- Claude Code hooks no longer depend on `jq` alone: they read their payloads via + `.agents/hooks/hook-input.sh` (`jq`, then `python3`) and branch on its exit + code. Previously a host without `jq` had every Bash call denied with no + explanation and no in-session way to recover, the `Stop` hook's + `stop_hook_active` loop guard silently defeated, and format-on-edit silently + disabled. With neither parser present the Bash guard still fails closed, now + naming the remedy, and `SessionStart` warns before the first denial. - `integrations.numba`: `DevmmEMMPlugin` passes the device pointer as a `ctypes.c_void_p`, the only type numba-cuda ≥ 0.30 converts to a driver pointer. diff --git a/development/tool-bootstrap.md b/development/tool-bootstrap.md index 8b48cd4..ef0062d 100644 --- a/development/tool-bootstrap.md +++ b/development/tool-bootstrap.md @@ -20,13 +20,14 @@ patch release for the whole team._ The core library needs **no system-level tools** — it is pure Python (`ctypes` -is stdlib). Supporting tooling: `make` (task runner) and `jq`. `jq` is -**required, not optional, when working through Claude Code**: the -`PreToolUse` Bash hook parses the tool input with it and fails closed, so -without `jq` every Bash call is denied — including the one that would install -it. Neither is installed by `uv` or by the session bootstrap; on a bare image -run `apt-get install -y jq make` (or `brew install jq make`) before starting an -agent session. **GPU work** requires +is stdlib). Supporting tooling: `make` (task runner), plus **a JSON parser for +the Claude Code hooks** — `jq`, or `python3` as the fallback +([`hook-input.sh`](../.agents/hooks/hook-input.sh) tries `jq` first). Any host +that can run this project already has `python3`, so this is normally satisfied; +with neither on PATH the hooks warn at session start and the +destructive-command guard denies every Bash call until one is installed +(`apt-get install jq` / `brew install jq`). `make` is not installed by `uv` +either — on a bare image, `apt-get install -y make`. **GPU work** requires the relevant vendor stack on the host, detected at runtime (not installed by `uv`): an NVIDIA driver + `libcudart` for CUDA, or a ROCm install + `libamdhip64` for ROCm. From 4a2dab3f7afb18151212a21046902be23c2f3f64 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Enrique=20Gonz=C3=A1lez=20Paredes?= <enriqueg@cscs.ch> Date: Tue, 28 Jul 2026 10:05:54 +0200 Subject: [PATCH 06/10] chore(harness): pin the template to the v0.7.0 release tag MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The simplification wave and the hook-payload fix are now released as v0.7.0, which is the same commit (fe487cc) this branch already tracked — so this records the tag in place of the pseudo-version and changes nothing else. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- .copier-answers.yml | 2 +- CHANGELOG.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/.copier-answers.yml b/.copier-answers.yml index a05a9a5..ebee7bd 100644 --- a/.copier-answers.yml +++ b/.copier-answers.yml @@ -1,5 +1,5 @@ # Changes here will be overwritten by Copier; NEVER EDIT MANUALLY. -_commit: v0.6.0-4-gfe487cc +_commit: v0.7.0 _src_path: https://github.com/grAItools/harness-copier-template.git commit_convention: conventional copilot_code_review: false diff --git a/CHANGELOG.md b/CHANGELOG.md index 2d0d2c8..4d53562 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -21,7 +21,7 @@ and this project adheres to documentation (`docs/api.md`). - `README.md`: links to in-repo files are absolute, so they resolve on the PyPI project page as well as on GitHub. -- Agent harness updated to copier template `v0.6.0-4-gfe487cc`: adds `report.md` +- Agent harness updated to copier template `v0.7.0`: adds `report.md` per work unit, a decision register in [`development/adr/README.md`](development/adr/README.md), and a [glossary](development/glossary.md). The template's simplification wave drops From 9f27b8f6ac0663a92993096d0b5d900b8df1be5c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Enrique=20Gonz=C3=A1lez=20Paredes?= <enriqueg@cscs.ch> Date: Tue, 28 Jul 2026 11:52:13 +0200 Subject: [PATCH 07/10] docs(harness): seed the glossary and refresh .agents/README.md MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses the two actionable findings from the #4 review. Glossary: it shipped empty while the same PR made it an authority — AGENTS.md lists it under "Where things live" and the Reviewer is told to flag names drifting from it, both pointing at a file with no terms. Worse, there was no route in: entries may only be promoted from a reviewed spec's Glossary section, and none of the 13 shipped work units has one, so the vocabulary already load-bearing in the code could never enter under the stated rule. Seeded from devmm-design.md §3 as an explicit one-time baseline, with the promotion rule left governing everything after it. All 22 code identifiers cited were checked against the package: 21 are in the public API, the rest resolves in src/. This also makes development/README.md's claim that every document in its table carries real devmm content true. .agents/README.md was 63 lines stale — missing the entire Layout section the simplification wave merged in from the four deleted per-directory READMEs. Root cause: copier's `_skip_if_exists` lists a bare `README.md`, which matches every README at any depth rather than only the root one, so this file has not been updated since the initial scaffold. The same glob silently skipped development/README.md during the v0.7.0 update, which is why that one had to be copied from a fresh render by hand. Refreshed from the v0.7.0 render, and corrected the Caveats bullet the review flagged: it listed `.agents/hooks/*` among files safe from `copier update`, while this PR makes block-destructive.sh and ensure-toolchain.sh byte-identical to the template render and takes hook-input.sh from it. Hook scripts are template-owned; local edits there get reverted. Not fixed here: the hook-input.sh backend-parity gap (objects, arrays and non-integral numbers differ between jq and python3). Confirmed but out of scope for a PR whose thesis is retiring local hook patches — reported upstream. Gate: make verify green — 900 passed, 75 skipped, unchanged. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- .agents/README.md | 78 +++++++++++++++++++++++++++++++++++++---- development/glossary.md | 45 +++++++++++++++++++++++- 2 files changed, 116 insertions(+), 7 deletions(-) diff --git a/.agents/README.md b/.agents/README.md index 68712ed..a90326f 100644 --- a/.agents/README.md +++ b/.agents/README.md @@ -18,7 +18,7 @@ agent back at one file prevents that structurally. | ------------------------------------------------------------------- | -------------------------------------------------- | --------------------------------------------------------------------------------- | | Claude Code | `CLAUDE.md` (first line `@AGENTS.md`) + `.claude/` | `CLAUDE.md`, `.claude/settings.json`, `.claude/{agents,commands,skills}` symlinks | | OpenCode | `.opencode/opencode.jsonc` `instructions` | `.opencode/opencode.jsonc`, `.opencode/{…}` symlinks | -| GitHub Copilot | native `AGENTS.md` (2026); legacy stub | `.github/copilot-instructions.md` → `AGENTS.md` (opt-in `copilot`) | +| GitHub Copilot | coding agent: native root `AGENTS.md`; code review: `.github/` files only | populated `.github/copilot-instructions.md` + `.github/instructions/` review rules + `.github/skills/code-review/` skill (opt-in `copilot_code_review`; code review does **not** read `AGENTS.md`) | | OpenAI Codex | native root `AGENTS.md` (32 KiB doc cap) | none needed | | Google Gemini CLI | `.gemini/settings.json` `context.fileName` | add `.gemini/settings.json` → `AGENTS.md` (see recipe below) | | Jules, Cursor, Windsurf, Roo Code, Zed, JetBrains Junie, Aider, Amp | native root `AGENTS.md` | none needed | @@ -35,26 +35,92 @@ agent back at one file prevents that structurally. ln -s ../../AGENTS.md .continue/rules/AGENTS.md ``` 3. **Needs a config key or a frontmatter'd file?** Add a thin pointer/config stub - that references `AGENTS.md`. Never copy instruction prose into it. Precedents: + that references `AGENTS.md`. Never copy instruction prose into it. Precedent: ```jsonc - // .github/copilot-instructions.md → one line: see ../AGENTS.md // .gemini/settings.json { "context": { "fileName": ["AGENTS.md", "GEMINI.md"] } } ``` 4. Put tool-specific guidance (not meant for every agent) in that tool's own file, not in `AGENTS.md`. +## Layout + +Everything here is a single source read by both tools. The symlinks are +created by the post-generation hook; if you copy this layout manually, +recreate them with: + +```sh +ln -s ../.agents/subagents .claude/agents && ln -s ../.agents/subagents .opencode/agents +ln -s ../.agents/commands .claude/commands && ln -s ../.agents/commands .opencode/commands +ln -s ../.agents/skills .claude/skills && ln -s ../.agents/skills .opencode/skills +``` + +### `subagents/` + +One Markdown file per role, with YAML frontmatter. Supported keys: + +- `name` (required) — invocation name; identity comes from this, not the + filename. +- `description` (required) — used by parent agents to decide when to + delegate; start with "Use proactively when…" for auto-discovery. +- `model` (optional) — `sonnet` / `opus` / `haiku` / `inherit`. +- `tools` (optional) — Claude Code allowlist, comma-separated names + (e.g. `Read, Grep, Glob, Bash`). Claude Code only. +- `permission` (optional) — OpenCode per-action map with keys + `read` / `write` / `edit` / `bash`, each taking `allow` / `ask` / `deny`; + `bash` can also be a per-pattern map (e.g. `"rg *": allow`, `"*": deny`). + OpenCode only — Claude Code ignores this field. +- `mode` (optional, **strongly recommended for subagents**) — OpenCode-only. + Set `mode: subagent` to keep the agent delegation-only; the default + (`all`) would also expose it as a top-level primary OpenCode agent. + +A subagent runs in its own context window — use them to keep heavy +exploration or repetitive review out of the main session's context. Claude +Code subagents **cannot spawn other subagents**: when a role needs something +run outside itself (an `explorer` pass, a spike experiment, a user's +answer), it stops and hands back to the main agent, carrying the request +and the resume instruction in its own reply — the role subagents' Handoff +sections define these protocols. The reply must be self-contained because a +subagent can be reached by description match as well as by its slash +command, and in the former case the command's instructions were never +loaded. + +### `commands/` + +One Markdown file per slash command, with YAML frontmatter (`description`, +optional `argument-hint`). Keep each command short and imperative — the +description is what surfaces in the slash-command picker, and the body is +the prompt the agent will follow. + +### `skills/` + +One directory per skill, containing a `SKILL.md` (required, with YAML +frontmatter `name` and `description`) and optionally `scripts/` +(deterministic executables), `references/` (docs loaded on demand), and +`assets/`. `design-principles/` is the skill that ships — the shared design +ground rules and red-flag checklist the role subagents read. Write skill +descriptions slightly "pushy" — agents tend to under-trigger skills — +and include synonyms. + ## Caveats - **Copier-managed.** This harness is generated from a Copier template (`gh:grAItools/harness-copier-template`; see `.copier-answers.yml`). Edits to template-owned files (`.claude/settings.json`, `.opencode/opencode.jsonc`, - `AGENTS.md`, the managed `.gitignore` block) can be reverted by `copier update`; - port durable changes upstream behind a per-agent toggle. Net-new files (this - README, `.agents/hooks/*`, `.gemini/settings.json`) are safe. + `AGENTS.md`, `.agents/hooks/*`, the managed `.gitignore` block) can be reverted + by `copier update`; port durable changes upstream behind a per-agent toggle. + Only files the template never ships (`.gemini/settings.json`) are safe. + The hook scripts are template-owned: `copier update` has rewritten + `ensure-toolchain.sh` and `block-destructive.sh` in place, and `hook-input.sh` + arrives from the template. - **Symlinks need `core.symlinks=true`.** On Windows checkouts without it, Git materializes a symlink as a text file containing the target path; prefer a stub there. The repo already relies on symlinks for `.claude/` / `.opencode/`. - **Destructive-command deny-list** is canonical in [`hooks/block-destructive.sh`](hooks/block-destructive.sh); OpenCode's deny globs are a hand-kept mirror (it cannot call a script). +- **Hook payload parsing** is canonical in + [`hooks/hook-input.sh`](hooks/hook-input.sh): the Claude Code hooks in + `.claude/settings.json` read their JSON input through it (`jq`, with a + `python3` fallback). With neither parser on PATH, the PreToolUse guard fails + closed with an explanatory message and SessionStart prints a warning. diff --git a/development/glossary.md b/development/glossary.md index 82ab2d2..71a0581 100644 --- a/development/glossary.md +++ b/development/glossary.md @@ -21,6 +21,49 @@ checks each new entry against the spec that proposed it). Renaming or redefining an existing term is its own PR, trunk-gated like the rest of `development/` (see [`README.md`](README.md)). +The terms below are the **initial baseline**, transcribed from +[`devmm-design.md`](devmm-design.md) §3 when the glossary was +introduced. The promotion rule above governs everything after it: the +0.1.0 vocabulary predates the rule and none of the shipped work units +has a spec **Glossary** section to promote from, so a one-time +baseline is the only way the register can describe the code it is +supposed to govern. Later terms come through a reviewed spec. + ## Terms -_None yet. The first feature spec will propose some._ +- **Device** — a specific piece of hardware that owns memory, identified by + a device type plus an ordinal (`cuda:0`). Every buffer, stream and memory + resource carries one explicitly; there is no ambient "current device" as a + correctness mechanism. (code: `Device`, `DeviceType`) +- **Stream** — an opaque, ordered execution queue that allocations are + sequenced against. Deliberately thin — a handle plus ordering primitives — + because devmm never launches kernels. (code: `Stream`) +- **Memory resource** — the central abstraction: an allocator that hands out + stream-ordered device memory, isomorphic to rmm's so the mental model + transfers. devmm *wraps* these rather than implementing allocation + strategies. (code: `DeviceMemoryResource`, concrete ones under `devmm.mrs`) +- **Adaptor** — a memory resource that wraps another to add behaviour + (statistics, logging, limiting, callbacks) without changing the allocation + contract. (code: `StatisticsAdaptor`, `LoggingAdaptor`, …) +- **Current memory resource** — the per-device default that `empty` and + friends allocate from when no resource is passed explicitly. + (code: `get_current_memory_resource` / `set_current_memory_resource`) +- **Device buffer** — an owning, untyped, stream-ordered allocation: bytes on + a device plus the resource that allocated them, which it keeps a strong + reference to so memory never outlives its allocator. (code: `DeviceBuffer`) +- **Layout** — a concrete, resolved arrangement of elements in a buffer: + strides, alignment, and the padding implied by them. (code: `Layout`) +- **Layout policy** — a callable that, given shape, dtype and device, + produces a `Layout`. The resulting layout keeps a reference to the policy + that made it. (code: `LayoutPolicy`; shipped ones `RowMajor`, `ColMajor`, + `Permuted`, `Aligned`, `DeviceOptimal`) +- **DType** — an element type mapping 1:1 onto DLPack's `(code, bits, lanes)` + triple, duck-typed from NumPy dtypes and Array API strings without + importing NumPy. (code: `DType`) +- **Tensor** — a minimal DLPack producer: dtype, shape, strides and offset + over a device buffer, exposing the two protocol methods. This is devmm's + export surface, not a compute type. (code: `Tensor`) +- **Device runtime** — the per-platform backend (CPU, CUDA, ROCm) that + supplies the raw allocation, copy and stream primitives a memory resource + builds on, discovered at runtime. (code: `DeviceRuntime`, `runtime_for`, + `available_runtimes`) From 17232e306d02f1bcc29c3d0aa6ceb2c75e03876e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Enrique=20Gonz=C3=A1lez=20Paredes?= <enriqueg@cscs.ch> Date: Tue, 28 Jul 2026 12:39:39 +0200 Subject: [PATCH 08/10] fix(harness): repair the register contract and correct four false doc claims MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses the locally-owned findings from the xhigh review of #4. The hook-script findings are upstream's and are reported there instead. The decision-register scan command shipped without a revision range, so `git diff` compared the worktree to the index and found nothing on any committed branch. Verified: a `DECISION-PENDING:` line committed to a report was missed by the documented command and caught by the range form. The check that was meant to make an unregistered escalation an automatic MAJOR silently passed instead — and since the report freezes at merge and the register alone records the outcome, the decision would have been lost with nothing marked pending. Register-row ownership was assigned twice, to different actors: `/build` step 5 has the caller add the row while servicing the hand-back, and the local text added in e245a65 claimed the Developer was the only role that could. Upstream's version is right and now covers what that patch was written for, so the local claim is removed rather than reconciled — which also retires its "sole sanctioned exception to leaving development/ alone", contradicted two lines later by the Developer's duty to draft ADRs under development/adr/. Also corrected, all claims this branch made false: - The no-work-unit review carve-out exempted the spec/plan axes and both register contracts but not the scope check, so a repo-layout PR touching every work unit — this one — tripped an automatic MAJOR nothing could clear. - The CHANGELOG said the pre-fix guard denied every Bash call. It failed *open*: the pipeline's status was the matcher's, so commands passed unchecked. The fail-closed behaviour was a v0.6.0 intermediate that never reached main. - harness-usage.md said a non-zero Stop hook blocks the stop. Its skip paths exit 1, which is non-blocking, so a host with no JSON parser can end a session with the gate unrun. - The subagent table presented the reviewer's read-only bash as enforced. The `permission:` map that would enforce it is OpenCode-only — Claude Code honours `tools:` alone, which grants unrestricted Bash. - Register IDs used `2026-07-p12`, while the legend says `<feature-slug>` and the slug is `2026-07-p12-conformance-docs-release`. Gate: make verify green — 900 passed, 75 skipped, unchanged. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- .agents/subagents/developer.md | 11 +++-------- .agents/subagents/reviewer.md | 6 +++++- CHANGELOG.md | 11 ++++++----- development/adr/README.md | 19 +++++++++++++------ development/harness-usage.md | 15 ++++++++++++--- 5 files changed, 39 insertions(+), 23 deletions(-) diff --git a/.agents/subagents/developer.md b/.agents/subagents/developer.md index 82c611d..765d1b0 100644 --- a/.agents/subagents/developer.md +++ b/.agents/subagents/developer.md @@ -80,14 +80,9 @@ phase boundary. - When a decision exceeds your authority — loosening a test tolerance or assertion, adding a dependency, changing behaviour the spec froze — write a `DECISION-PENDING:` line in `report.md` and stop. Do not - resolve it locally. **You also own its register row**: append the - matching `pending` row to the table in `development/adr/README.md` in - the same change as the marker, then stop. You are the only role that - can — the Reviewer is read-only and the Architect and Product Owner may - not write outside the feature directory — and the Reviewer counts a - marker with no row as a defect. Appending that one row is the sole - sanctioned exception to leaving `development/` alone; flip its Status - once the human answers. + resolve it locally, and do not add the register row yourself — the + caller does that when it services the hand-back (`/build` step 5), so + the marker and its row land in the same change. - Work **one phase at a time**. Do not begin phase N+1 until phase N's tests pass and its `tasks.md` boxes are ticked. - Write the test **first** when the plan calls for behaviour change — diff --git a/.agents/subagents/reviewer.md b/.agents/subagents/reviewer.md index 9fd1a11..f62a9b3 100644 --- a/.agents/subagents/reviewer.md +++ b/.agents/subagents/reviewer.md @@ -114,7 +114,11 @@ fix before the next push. `development/work/<slug>/` directory — is reviewed on the axes that apply (quality, gate, honesty of the PR description). The spec- and plan-conformance axes and both register contracts are `n/a`, not - failures; do not manufacture defects from their absence. + failures; do not manufacture defects from their absence. The scope + check still applies, but judge it against the PR's stated purpose + rather than a plan: a repo-layout or harness change may legitimately + touch every `development/work/` directory at once, and that is only a + defect if the description does not account for it. - **Never trust the Developer's narrative.** Re-run the gate yourself; re-derive every claim you can check (line numbers, test names, measured values) instead of quoting the hand-off summary. diff --git a/CHANGELOG.md b/CHANGELOG.md index 4d53562..bee9496 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -37,11 +37,12 @@ and this project adheres to - Claude Code hooks no longer depend on `jq` alone: they read their payloads via `.agents/hooks/hook-input.sh` (`jq`, then `python3`) and branch on its exit - code. Previously a host without `jq` had every Bash call denied with no - explanation and no in-session way to recover, the `Stop` hook's - `stop_hook_active` loop guard silently defeated, and format-on-edit silently - disabled. With neither parser present the Bash guard still fails closed, now - naming the remedy, and `SessionStart` warns before the first denial. + code. Previously, on a host without `jq`, the destructive-command guard failed + **open** — the pipeline's exit status was the matcher's, so every Bash call was + allowed through unchecked — the `Stop` hook's `stop_hook_active` loop guard was + silently defeated, and format-on-edit silently no-opped. The Bash guard now + fails closed and names the remedy, and `SessionStart` warns when no parser is + present. - `integrations.numba`: `DevmmEMMPlugin` passes the device pointer as a `ctypes.c_void_p`, the only type numba-cuda ≥ 0.30 converts to a driver pointer. diff --git a/development/adr/README.md b/development/adr/README.md index 796e6b4..22a5c1e 100644 --- a/development/adr/README.md +++ b/development/adr/README.md @@ -69,17 +69,24 @@ while burying a real escalation among the quotes. Scan report files, not the whole tree: ```sh -git diff --unified=0 -- 'development/work/*/report.md' | grep -E '^\+.*DECISION-PENDING:' +git diff --unified=0 "$(git merge-base origin/main HEAD)"...HEAD \ + -- 'development/work/*/report.md' | grep -E '^\+.*DECISION-PENDING:' ``` -The register row is the **Developer's** to add, in the same change as the -marker — see `.agents/subagents/developer.md`. Rows whose Source is an ADR or a -direct human grant have no marker by construction, and are equally valid. +The revision range is load-bearing: a bare `git diff` compares the worktree to +the index, so on a committed branch it inspects nothing and the check passes +whatever the diff contains. + +The register row is added by the **caller servicing the hand-back** — `/build` +step 5, which puts the question to the user and appends the row in the same +change as the marker. The Developer writes the marker and stops; it does not +append the row itself. Rows whose Source is an ADR or a direct human grant have +no marker by construction, and are equally valid. | ID | Date | Decision | Status | Source | Evidence | |---|---|---|---|---|---| -| 2026-07-p12.1 | 2026-07-17 | Waive the T2 (CUDA) and T3 (ROCm) hardware gates for the 0.1.0 release; CPU CI plus the fake-api suites stand in. Maintainer sign-off extended the p12 spec's T3-only clause to T2. | accepted | [ADR 0003](0003-gpu-suite-waiver-for-0.1.0.md) | `0900acf` | -| 2026-07-p12.2 | 2026-07-17 | Coverage thresholds for the release gate: ≥ 90% overall, ≥ 95% on `devmm._core` + `devmm._dlpack`, enforced by the `gate-coverage` CI leg. | accepted | [`development/testing.md`](../testing.md#coverage-targets) | `97cc5c9` | +| 2026-07-p12-conformance-docs-release.1 | 2026-07-17 | Waive the T2 (CUDA) and T3 (ROCm) hardware gates for the 0.1.0 release; CPU CI plus the fake-api suites stand in. Maintainer sign-off extended the p12 spec's T3-only clause to T2. | accepted | [ADR 0003](0003-gpu-suite-waiver-for-0.1.0.md) | `0900acf` | +| 2026-07-p12-conformance-docs-release.2 | 2026-07-17 | Coverage thresholds for the release gate: ≥ 90% overall, ≥ 95% on `devmm._core` + `devmm._dlpack`, enforced by the `gate-coverage` CI leg. | accepted | [`development/testing.md`](../testing.md#coverage-targets) | `97cc5c9` | (ID = `<feature-slug>.<k>`, e.g. `2026-07-user-auth.1`. Source = the report or ADR that raised it. Evidence = the PR/commit that settled it.) diff --git a/development/harness-usage.md b/development/harness-usage.md index 18bfbba..c8a36c1 100644 --- a/development/harness-usage.md +++ b/development/harness-usage.md @@ -46,9 +46,15 @@ The five subagents and their access: | `product-owner` | yes | no | author `spec.md`; stop before planning | | `architect` | yes | no | author `plan.md` + `tasks.md`; stop before code | | `developer` | yes | yes | implement phase-by-phase, verify, tick tasks | -| `reviewer` | no | read-only + verify/test/lint | GO / NEEDS-WORK verdict, file:line defects | +| `reviewer` | no | read-only + verify/test/lint\* | GO / NEEDS-WORK verdict, file:line defects | | `explorer` | no | read-only search/git | "where is X / how does Y work" summaries | +\* The Writes?/Bash? columns are the roles' *contracts*. They are enforced by +the `permission:` map only under OpenCode; Claude Code ignores that key and +honours the `tools:` list alone (see [`.agents/README.md`](../.agents/README.md)), +so under Claude Code the read-only roles are bounded by instruction, not by the +harness. The `PreToolUse` deny-list still applies to every role. + Subagents cannot spawn subagents, so a role that needs something run outside itself stops mid-phase and **hands back**: the Product Owner replies with one clarifying question at a time (you answer, the caller @@ -75,8 +81,11 @@ Claude Code only). You cannot prompt around the hooks: edited file after every write. - **PreToolUse (`Bash`)** hard-blocks `rm -rf`, `push --force`, `reset --hard`, `DROP TABLE` (exit 2) via [`.agents/hooks/block-destructive.sh`](../.agents/hooks/block-destructive.sh). -- **Stop** runs `make verify` before the agent is allowed to stop; - non-zero blocks the stop. This is why "done" means "the gate is green". +- **Stop** runs `make verify` before the agent is allowed to stop; a failing + gate exits 2 and blocks the stop. This is why "done" means "the gate is + green". Its skip paths (no JSON parser, `uv` unavailable) exit 1 instead — + visible to you, but **non-blocking**, so on a host missing those the session + can end without the gate having run. CI is the backstop. Permissions allowlist the build tool, read-only git (`status/diff/log/show`), and `rg/ls/cat/head/tail`; destructive operations are denied. From 06940b4ac6ae56bd0f59243174eb794900a37cd1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Enrique=20Gonz=C3=A1lez=20Paredes?= <enriqueg@cscs.ch> Date: Wed, 29 Jul 2026 13:12:39 +0200 Subject: [PATCH 09/10] chore(harness): pin the template to v0.7.0-3-g0fa3c56 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Picks up the three upstream fixes this adoption prompted, all now closed: - _skip_if_exists is root-anchored (#33). A bare "README.md" matched at every depth under gitignore semantics, silently freezing the template's own .agents/README.md, .claude/rules/README.md and development/README.md downstream with no conflict reported in either direction. - The payload reader probes each backend by running it, and the Stop gate reports instead of skipping (#34, #35). A jq that resolves but cannot run now falls through to python3; a gate that cannot be blocked still runs and says so, rather than ending the session unverified. - The deny-list matches operations outside quotes (#36), so a read-only command that merely mentions one is allowed. The SQL pattern still matches anywhere: it has no unquoted form, so its mention and its use are indistinguishable. .agents/README.md takes the template's version wholesale — upstream's rewrite generalises the local caveat this branch carried (hooks are template-owned) to all of .agents/, and adds the safe-to-edit list. development/{harness-usage,tool-bootstrap}.md are _skip_if_exists, so copier leaves them alone and the new behaviour is ported by hand. The Stop-hook paragraph is rewritten rather than replaced: upstream covers the reader path but not the uv-unavailable path, which still skips the gate on exit 1. Every claim added to those two documents was checked against the scripts themselves — 13 deny-list cases, and the reader's exit codes for a broken jq, python3-only, no working parser (3) and an empty payload (4). Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> --- .agents/README.md | 36 +++++++--- .agents/hooks/block-destructive.sh | 111 +++++++++++++++++++++++++++-- .agents/hooks/hook-input.sh | 29 +++++--- .claude/settings.json | 5 +- .copier-answers.yml | 2 +- .opencode/opencode.jsonc | 8 ++- CHANGELOG.md | 15 +++- development/harness-usage.md | 26 +++++-- development/tool-bootstrap.md | 20 +++--- 9 files changed, 201 insertions(+), 51 deletions(-) diff --git a/.agents/README.md b/.agents/README.md index a90326f..4527988 100644 --- a/.agents/README.md +++ b/.agents/README.md @@ -105,22 +105,36 @@ and include synonyms. ## Caveats - **Copier-managed.** This harness is generated from a Copier template - (`gh:grAItools/harness-copier-template`; see `.copier-answers.yml`). Edits to - template-owned files (`.claude/settings.json`, `.opencode/opencode.jsonc`, - `AGENTS.md`, `.agents/hooks/*`, the managed `.gitignore` block) can be reverted - by `copier update`; port durable changes upstream behind a per-agent toggle. - Only files the template never ships (`.gemini/settings.json`) are safe. - The hook scripts are template-owned: `copier update` has rewritten - `ensure-toolchain.sh` and `block-destructive.sh` in place, and `hook-input.sh` - arrives from the template. + (`gh:grAItools/harness-copier-template`; see `.copier-answers.yml`). + Everything under `.agents/` is template-owned — this README, `hooks/*`, + `subagents/*`, `commands/*`, `skills/*` — as are `.claude/settings.json`, + `.opencode/opencode.jsonc`, `AGENTS.md`, and the managed `.gitignore` block. + Edits to any of them can be reverted by `copier update`; port durable changes + upstream behind a per-agent toggle. Safe to edit: files you add yourself + (e.g. `.gemini/settings.json` from the recipe above) and the files the + template hands over on first generation and never rewrites — the root + `README.md`, the task-runner file, `scripts/*.sh`, + `.github/PULL_REQUEST_TEMPLATE.md`, and the `development/` documents + (`architecture.md`, `style.md`, `testing.md`, `glossary.md`, + `tool-bootstrap.md`, `harness-usage.md`, `adr/README.md`). Note that + `development/README.md` is *not* in that set — it is template-owned. - **Symlinks need `core.symlinks=true`.** On Windows checkouts without it, Git materializes a symlink as a text file containing the target path; prefer a stub there. The repo already relies on symlinks for `.claude/` / `.opencode/`. - **Destructive-command deny-list** is canonical in [`hooks/block-destructive.sh`](hooks/block-destructive.sh); OpenCode's deny globs - are a hand-kept mirror (it cannot call a script). + are a hand-kept mirror (it cannot call a script). The script denies an + *operation* only where it is not inside quotes, so a read-only mention of one + (a search pattern, a fixture, a commit message) is allowed; SQL patterns still + match anywhere, quoted or not, since they have no unquoted form. OpenCode's + globs cannot express either distinction and deny every mention. The patterns + themselves are kept in sync by hand in both places. - **Hook payload parsing** is canonical in [`hooks/hook-input.sh`](hooks/hook-input.sh): the Claude Code hooks in `.claude/settings.json` read their JSON input through it (`jq`, with a - `python3` fallback). With neither parser on PATH, the PreToolUse guard fails - closed with an explanatory message and SessionStart prints a warning. + `python3` fallback; each backend is probed by *running* it, so a `jq` that + resolves but cannot run falls through instead of failing the read). With + neither parser working, the PreToolUse guard fails closed with an + explanatory message, SessionStart warns, and the Stop gate still runs but + can only report a red result. Read scalar string or boolean fields: the + backends agree on those, not on number spelling (see the script header). diff --git a/.agents/hooks/block-destructive.sh b/.agents/hooks/block-destructive.sh index 1aba264..5e4d4c5 100755 --- a/.agents/hooks/block-destructive.sh +++ b/.agents/hooks/block-destructive.sh @@ -1,10 +1,37 @@ #!/usr/bin/env sh # block-destructive.sh — canonical destructive-command matcher. # -# Reads candidate command text on stdin; exits 2 (with a message naming the -# deny-list, so a deny is distinguishable from an infrastructure failure) if it -# matches a forbidden pattern, 0 otherwise. This is the single source of truth -# for the deny-list. +# Reads candidate command text on stdin; exits 2 (with a message naming what +# matched, so a deny is distinguishable from an infrastructure failure) if the +# command *runs* a forbidden operation, 0 otherwise. This is the single source +# of truth for the deny-list. +# +# It matches text, not intent — but it does separate a pattern that is run from +# one merely mentioned inside quotes, so read-only commands that quote a +# deny-listed string (a grep pattern, `git log --grep=…`, a printf'd fixture, a +# commit message) are allowed. Three rules: +# +# 1. an operation reached without passing through a quote -> deny +# 2. an operation inside the string another shell will run — +# past the quote a runner opens (sh -c / ssh / eval / su) +# or ahead of a pipe into a shell — where being quoted +# does not make it a mention -> deny +# 3. a text pattern anywhere, quoted or not — SQL only ever +# appears as a quoted argument, so its mention and its +# use are indistinguishable -> deny +# +# Surviving false positives, each explained by the deny message it triggers: a +# quoted mention of a rule-3 pattern, a heredoc body, a mention inside a +# nested-shell string, and one inside the '…'\''…' idiom (the empty '' span the +# grammar has to allow makes the pattern look reachable). Recovery is a +# rephrase, not a different task. +# +# Blind spots, unchanged in kind from the earlier plain-substring form: `eval` +# of a variable, "$(…)" command substitution, aliases, encoded payloads, a +# heredoc body whose lines read as commands, a shell run from a file it wrote, +# an interpreter whose -c is not adjacent to a shell name (bash -euo … -c), and +# any language (`python -c`) doing the same work. This is a backstop against +# accidents, not a sandbox. # # The deny decision deliberately stays on POSIX `grep -qE`: extracting the # matched pattern with `grep -o` would make the decision depend on a non-POSIX @@ -15,11 +42,81 @@ # - Claude Code: the PreToolUse(Bash) hook in .claude/settings.json pipes the # tool input here. # - OpenCode: cannot call a script, so the deny globs in -# .opencode/opencode.jsonc restate these patterns by hand — keep in sync. +# .opencode/opencode.jsonc restate these patterns by hand — as plain +# substrings, since a glob cannot express rule 1, so OpenCode still denies +# the quoted mentions this script allows. Keep the patterns in sync. # # See .agents/README.md for the single-source-of-truth rationale. -if grep -qE 'rm -rf|push --force|reset --hard|DROP TABLE'; then - echo 'block-destructive: denied - the command matches the destructive deny-list (rm -rf, push --force, reset --hard, DROP TABLE).' >&2 + +# --- deny-list (mirror any change into .opencode/opencode.jsonc) ------------- +# Destructive operations: denied when run, allowed when quoted (rules 1 and 2). +operations='rm -rf|push --force|reset --hard' +# Denied anywhere in the command text, quoted or not (rule 3). +text_patterns='DROP TABLE' +# Shell names are listed, never matched as an `sh` suffix: `push` and `refresh` +# end in `sh` too, and a suffix match would deny `git push -c k=v` the moment +# the line mentioned an operation anywhere. +shells='(sh|bash|dash|zsh|ksh|ash)' +# ----------------------------------------------------------------------------- + +# Commands that run a string argument rather than read one, up to the quote that +# argument opens. `[^;&|]*` keeps the runner and the quote in one simple command, +# so `ssh host uptime && grep '<pattern>' .` is not read as handing the pattern +# to ssh. +runner="(^|[^[:alnum:]_./-])((/[a-z/]*)?${shells}[[:space:]]+-[[:alnum:]]*c|(ssh|eval|su)[[:space:]])[^;&|]*['\"]" +# …and shells that take their script from stdin, which the operation reaches by +# being piped into one. `ssh`/`eval`/`su` are absent here on purpose: they run an +# argument, not stdin, so `grep '<pattern>' . | ssh host tee f` is a mention. +piped="[^|]*\\|[[:space:]]*(sudo[[:space:]]+)?(/[a-z/]*)?${shells}([[:space:]]|\$)" + +# Complete quoted spans, escape-aware: to the shell a backslash-escaped quote is +# a literal character, not a delimiter, so consuming it as one would flip the +# in/out-of-quote classification for the rest of the line. +squoted="'[^']*'" # '…' — POSIX: no escapes inside +dquoted="\"([^\"\\\\]|\\\\.)*\"" # "…" — \" does not end the span +escaped="\\\\[\"']" # \" or \' outside quotes: a literal + +# Characters reachable without crossing a quote: complete spans are consumed +# whole, so anything only reachable *through* a quote is a mention. Anchored per +# line — grep is line-oriented, and each line of a multi-line command starts a +# new command. A bare backslash stays an ordinary character here, so the +# alias-bypass form (\rm -rf) is still read as the operation it is. +outside_quotes="^([^'\"]|$escaped|$squoted|$dquoted)*" + +cmd=$(cat) + +matches() { printf '%s\n' "$cmd" | grep -qE "$1"; } + +deny() { + # printf, not echo: an XSI echo (dash, BusyBox) expands backslash escapes, + # so a deny-list pattern containing one would print mangled. + printf '%s\n' "block-destructive: denied - $1" >&2 exit 2 +} + +if matches "$outside_quotes($operations)"; then + deny "the command runs a deny-listed destructive operation. + Deny-list: $operations. + Only occurrences reachable without crossing a quote are denied, so a mention + inside quotes (a search pattern, a fixture, a commit message) passes. This + one was reachable, so it is blocked by design." +fi + +# The operation has to sit inside what the shell will run — after the runner's +# opening quote, or before the pipe into a shell — not merely somewhere on the +# same line as one. +if matches "${runner}[^'\"]*($operations)" || matches "($operations)$piped"; then + deny "the command passes a deny-listed destructive operation to a nested + shell to run — as an argument (sh -c / ssh / eval / su) or piped in as a + script (… | sh) — where being quoted is not a mention. + Deny-list: $operations. Blocked by design." +fi + +if matches "$text_patterns"; then + deny "the command text contains $text_patterns. This pattern is denied + anywhere, quoted or not, because SQL only ever appears as a quoted argument, + so a mention of it cannot be told apart from a use. To search for it, use a + pattern that avoids the literal (e.g. DROP TABL[E])." fi + exit 0 diff --git a/.agents/hooks/hook-input.sh b/.agents/hooks/hook-input.sh index 47ef12c..33d5f38 100755 --- a/.agents/hooks/hook-input.sh +++ b/.agents/hooks/hook-input.sh @@ -4,18 +4,25 @@ # Usage: hook-input.sh <dot.path> (stdin: the hook's JSON payload) # e.g. hook-input.sh .tool_input.command # -# Prints the field's value on stdout, identically under either backend: -# '' for null/absent (including a path through a non-object), 'true'/'false' -# for booleans, raw text for strings and numbers, JSON for objects/arrays. +# Prints the field's value on stdout. The two backends agree byte-for-byte on +# what hooks are meant to read: '' for null/absent (including a path through a +# non-object), 'true'/'false' for booleans, raw text for strings. Objects and +# arrays print as compact JSON under both, but *numbers are not contracted* — +# jq canonicalises number literals (jq 1.6 prints 3.0 as 3) where python3 +# preserves them, bare or nested inside an object or array. Read scalar string +# or boolean fields; treat number spelling as backend-dependent. # # Exit codes: 0 read OK; 3 no working JSON parser on PATH; 4 empty or # unparseable payload. Callers branch on the distinction to pick their own # failure posture and message. # -# Parses with jq when available, falling back to python3. The fallback is -# probed by *running* python3, not `command -v` alone — stock macOS ships a -# /usr/bin/python3 stub that passes `command -v` but fails until the Xcode -# Command Line Tools are installed. +# Parses with jq, falling back to python3. *Both* backends are probed by +# running them, not by `command -v` alone: an asdf/mise shim with no version +# selected, a half-removed package or a broken wrapper resolves and then +# fails, and stock macOS ships a /usr/bin/python3 stub that passes +# `command -v` but fails until the Xcode Command Line Tools are installed. A +# jq that cannot run falls through to python3; a jq that passed the probe and +# then rejects the payload is a real exit 4, not a reason to retry elsewhere. # # Consumers: # - Claude Code: the PostToolUse, PreToolUse, and Stop hooks in @@ -29,8 +36,8 @@ if [ -z "$payload" ]; then exit 4 fi -if command -v jq >/dev/null 2>&1; then - out=$(printf '%s' "$payload" | jq -r --arg p "$1" ' +if command -v jq >/dev/null 2>&1 && printf '{}' | jq -e . >/dev/null 2>&1; then + out=$(printf '%s' "$payload" | jq -c -r --arg p "$1" ' ($p | split(".") | map(select(length > 0))) as $parts | (try getpath($parts) catch null) | if . == null then "" elif type == "boolean" then tostring else . end @@ -57,12 +64,12 @@ if v is None: elif v is True or v is False: print(str(v).lower()) elif isinstance(v, (dict, list)): - print(json.dumps(v)) + print(json.dumps(v, separators=(",", ":"))) else: print(v) ' "$1" exit $? fi -echo 'hook-input.sh: no working JSON parser (jq or python3) on PATH; cannot read the hook input. Install jq (apt-get install jq / brew install jq).' >&2 +echo 'hook-input.sh: no working JSON parser on PATH - jq and python3 are both missing or unable to run; cannot read the hook input. Install jq (apt-get install jq / brew install jq).' >&2 exit 3 diff --git a/.claude/settings.json b/.claude/settings.json index 58b462d..624f3ec 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -28,7 +28,7 @@ }, { "type": "command", - "command": "command -v jq >/dev/null 2>&1 || python3 -c '' >/dev/null 2>&1 || { echo 'warning: no working JSON parser (jq or python3) on PATH - the PreToolUse guard will deny every Bash call until one is installed (apt-get install jq / brew install jq)' >&2; exit 1; }" + "command": "printf '{}' | sh \"${CLAUDE_PROJECT_DIR:-.}/.agents/hooks/hook-input.sh\" .probe >/dev/null 2>&1 || { echo 'warning: the hook payload reader is not working - jq and python3 are both missing or unable to run, or .agents/hooks/hook-input.sh is damaged. The PreToolUse guard will deny every Bash call until this is fixed (apt-get install jq / brew install jq)' >&2; exit 1; }" } ] } @@ -55,12 +55,13 @@ ] } ], + "Stop": [ { "hooks": [ { "type": "command", - "command": "export PATH=\"$HOME/.local/bin:$PATH\"; flag=$(sh \"${CLAUDE_PROJECT_DIR:-.}/.agents/hooks/hook-input.sh\" .stop_hook_active); rc=$?; [ \"$rc\" -eq 3 ] \u0026\u0026 { echo \u0027verify skipped: no JSON parser on PATH (install jq: apt-get install jq / brew install jq)\u0027 \u003e\u00262; exit 1; }; [ \"$rc\" -ne 0 ] \u0026\u0026 { echo \"verify skipped: could not read the hook input (hook-input.sh exit $rc)\" \u003e\u00262; exit 1; }; [ \"$flag\" = \u0027true\u0027 ] \u0026\u0026 exit 0; command -v uv \u003e/dev/null 2\u003e\u00261 || { echo \u0027verify skipped: uv unavailable (run .agents/hooks/ensure-toolchain.sh; see development/tool-bootstrap.md)\u0027 \u003e\u00262; exit 1; }; make verify || exit 2" + "command": "export PATH=\"$HOME/.local/bin:$PATH\"; fail=2; flag=$(sh \"${CLAUDE_PROJECT_DIR:-.}/.agents/hooks/hook-input.sh\" .stop_hook_active); rc=$?; if [ \"$rc\" -ne 0 ]; then if [ \"$rc\" -eq 3 ]; then why=\u0027no JSON parser on PATH; install jq (apt-get install jq / brew install jq)\u0027; else why=\"hook-input.sh exit $rc\"; fi; echo \"verify: cannot read the stop-loop guard ($why) - running the gate anyway, but a red gate can only be reported, not blocked, until this is fixed.\" \u003e\u00262; fail=1; fi; [ \"$flag\" = \u0027true\u0027 ] \u0026\u0026 exit 0; command -v uv \u003e/dev/null 2\u003e\u00261 || { echo \u0027verify skipped: uv unavailable (run .agents/hooks/ensure-toolchain.sh; see development/tool-bootstrap.md)\u0027 \u003e\u00262; exit 1; }; make verify || { [ \"$fail\" -eq 1 ] \u0026\u0026 echo \u0027verify: the gate FAILED and the stop was not blocked (unreadable stop-loop guard) - this session is unverified; re-run the gate yourself.\u0027 \u003e\u00262; exit \"$fail\"; }" } ] } diff --git a/.copier-answers.yml b/.copier-answers.yml index ebee7bd..532d1af 100644 --- a/.copier-answers.yml +++ b/.copier-answers.yml @@ -1,5 +1,5 @@ # Changes here will be overwritten by Copier; NEVER EDIT MANUALLY. -_commit: v0.7.0 +_commit: v0.7.0-3-g0fa3c56 _src_path: https://github.com/grAItools/harness-copier-template.git commit_convention: conventional copilot_code_review: false diff --git a/.opencode/opencode.jsonc b/.opencode/opencode.jsonc index 26649ed..28ade83 100644 --- a/.opencode/opencode.jsonc +++ b/.opencode/opencode.jsonc @@ -19,9 +19,11 @@ // .agents/hooks/block-destructive.sh by hand (OpenCode can't call a script). // OpenCode compiles each glob to an anchored regex (`*` -> `.*`), so a leading // and trailing `*` is required to match a substring — hence the `*…*` forms - // below, which catch the pattern anywhere (e.g. `cd x && rm -rf y`), matching - // the script. The trade-off (shared with the script) is that a benign command - // merely *mentioning* a pattern is also caught. The allow-list matches + // below, which catch the pattern anywhere (e.g. `cd x && rm -rf y`), as the + // script does. Same patterns, stricter matching: the script denies them only + // outside quotes, so it allows a benign command that merely *mentions* one + // (`grep -rn 'rm -rf' .`) while these globs still deny it — a glob cannot + // express the distinction. Keep the patterns in sync. The allow-list matches // .claude/settings.json: the build tool, read-only git, plus read-only // inspection helpers (note: `find` is intentionally not allowed — // `find -delete`/`-exec rm` would bypass the deny-list). diff --git a/CHANGELOG.md b/CHANGELOG.md index bee9496..4256054 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -21,7 +21,8 @@ and this project adheres to documentation (`docs/api.md`). - `README.md`: links to in-repo files are absolute, so they resolve on the PyPI project page as well as on GitHub. -- Agent harness updated to copier template `v0.7.0`: adds `report.md` +- Agent harness updated to copier template `v0.7.0-3-g0fa3c56` — the `v0.7.0` + release plus the three fixes this adoption prompted upstream. Adds `report.md` per work unit, a decision register in [`development/adr/README.md`](development/adr/README.md), and a [glossary](development/glossary.md). The template's simplification wave drops @@ -43,6 +44,18 @@ and this project adheres to silently defeated, and format-on-edit silently no-opped. The Bash guard now fails closed and names the remedy, and `SessionStart` warns when no parser is present. +- `PreToolUse` no longer blocks read-only commands that merely *mention* a + deny-listed operation: `rm -rf`, `push --force` and `reset --hard` match only + outside quotes, so `grep -rn 'rm -rf' .` passes while `cd x && rm -rf y` is + still denied, with a fallback for the string a nested shell runs (`sh -c`, + `ssh`, `eval`, `su`, or a pipe into a shell). `DROP TABLE` still matches + anywhere — SQL has no unquoted form, so its mention and its use are + indistinguishable. +- The `Stop` gate no longer skips when the hook payload reader fails: it runs + `make verify` and reports a red result it cannot block, rather than ending the + session with no gate having run. `hook-input.sh` now probes `jq` by running + it, so a `jq` that resolves but is broken falls through to `python3` instead + of failing the read. - `integrations.numba`: `DevmmEMMPlugin` passes the device pointer as a `ctypes.c_void_p`, the only type numba-cuda ≥ 0.30 converts to a driver pointer. diff --git a/development/harness-usage.md b/development/harness-usage.md index c8a36c1..d003b3a 100644 --- a/development/harness-usage.md +++ b/development/harness-usage.md @@ -81,11 +81,22 @@ Claude Code only). You cannot prompt around the hooks: edited file after every write. - **PreToolUse (`Bash`)** hard-blocks `rm -rf`, `push --force`, `reset --hard`, `DROP TABLE` (exit 2) via [`.agents/hooks/block-destructive.sh`](../.agents/hooks/block-destructive.sh). + The first three are denied only outside quotes — `grep -rn 'rm -rf' .` is a + mention and passes, `cd x && rm -rf y` is not and does not — plus a fallback + for the string a nested shell *runs*: past the quote a runner opens + (`sh -c`, `ssh`, `eval`, `su`) or ahead of a pipe into a shell. A runner + elsewhere on the line does not count, so `ssh host uptime && grep -rn 'rm -rf' .` + still passes. `DROP TABLE` matches anywhere, quoted or not: SQL has no + unquoted form, so a mention of it is denied too. Each deny says which rule + fired; it is a text matcher and a backstop against accidents, not a sandbox. - **Stop** runs `make verify` before the agent is allowed to stop; a failing gate exits 2 and blocks the stop. This is why "done" means "the gate is - green". Its skip paths (no JSON parser, `uv` unavailable) exit 1 instead — - visible to you, but **non-blocking**, so on a host missing those the session - can end without the gate having run. CI is the backstop. + green". Two paths exit 1 instead — visible to you, but **non-blocking**: + `uv` unavailable skips the gate entirely, and a payload reader that cannot + run (no working `jq` or `python3`) still runs the gate but can only *report* + a red result, because blocking needs the stop-loop guard that same reader + provides. Either way the session can end unverified, so read the stop + message; CI is the backstop. Permissions allowlist the build tool, read-only git (`status/diff/log/show`), and `rg/ls/cat/head/tail`; destructive operations are denied. @@ -119,8 +130,11 @@ OpenCode reads `.opencode/opencode.jsonc`, which sets: - **Destructive-bash blocking** uses `permission.bash` deny globs in `*…*` (substring) form that mirror `block-destructive.sh`'s patterns (`rm -rf`, `push --force`, `reset --hard`, `DROP TABLE`), so `cd x && rm -rf y` is caught. - It's glob-not-regex and can't run a custom script, so for richer logic a - `.opencode/plugin` with a `tool.execute.before` hook is the optional hardening. + It's glob-not-regex and can't run a custom script, so the two **diverge on + mentions**: a glob cannot tell a quoted pattern from a command, so OpenCode + also denies `grep -rn 'rm -rf' .`, which the script allows. It is the stricter + surface; for richer logic a `.opencode/plugin` with a `tool.execute.before` + hook is the optional hardening. | Capability | Claude Code | OpenCode | | ---------------------- | ---------------------------------------- | ------------------------------------------ | @@ -128,7 +142,7 @@ OpenCode reads `.opencode/opencode.jsonc`, which sets: | Subagent invocation | "use the X subagent" | `@mention` / auto-delegation (Task tool) | | Auto-format | `PostToolUse` hook | native `formatter` | | Verify gate | `Stop` hook (blocking) | `/verify` + CI only (no session-end hook) | -| Block destructive bash | `PreToolUse` hook (script) | `permission.bash` deny (`*…*` substring globs) | +| Block destructive bash | `PreToolUse` hook (script; quoted mentions pass) | `permission.bash` deny (`*…*` globs; mentions denied too) | | Path-scoped rules | `.claude/rules/` | _(no equivalent)_ | ## Decision guide — which capability for which task diff --git a/development/tool-bootstrap.md b/development/tool-bootstrap.md index ef0062d..ff96236 100644 --- a/development/tool-bootstrap.md +++ b/development/tool-bootstrap.md @@ -22,15 +22,17 @@ The core library needs **no system-level tools** — it is pure Python (`ctypes` is stdlib). Supporting tooling: `make` (task runner), plus **a JSON parser for the Claude Code hooks** — `jq`, or `python3` as the fallback -([`hook-input.sh`](../.agents/hooks/hook-input.sh) tries `jq` first). Any host -that can run this project already has `python3`, so this is normally satisfied; -with neither on PATH the hooks warn at session start and the -destructive-command guard denies every Bash call until one is installed -(`apt-get install jq` / `brew install jq`). `make` is not installed by `uv` -either — on a bare image, `apt-get install -y make`. **GPU work** requires -the relevant vendor stack on the host, detected at runtime (not installed by -`uv`): an NVIDIA driver + `libcudart` for CUDA, or a ROCm install + -`libamdhip64` for ROCm. +([`hook-input.sh`](../.agents/hooks/hook-input.sh) tries `jq` first, probing +each backend by *running* it, so a `jq` that resolves but cannot run falls +through to `python3` instead of failing the read). Any host that can run this +project already has `python3`, so this is normally satisfied; with neither +working the hooks warn at session start, the destructive-command guard denies +every Bash call until one is installed (`apt-get install jq` / +`brew install jq`), and the Stop gate still runs but can only report a red +result. `make` is not installed by `uv` either — on a bare image, +`apt-get install -y make`. **GPU work** requires the relevant vendor stack on +the host, detected at runtime (not installed by `uv`): an NVIDIA driver + +`libcudart` for CUDA, or a ROCm install + `libamdhip64` for ROCm. ## Install `uv` From 77b11d198cde7076725510de459d109a35510710 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Enrique=20Gonz=C3=A1lez=20Paredes?= <enriqueg@cscs.ch> Date: Wed, 29 Jul 2026 15:16:24 +0200 Subject: [PATCH 10/10] test(harness): lock the deny-list verdicts and repair the register scan MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The template's quote-aware guard rewrite fixed the false positives it aimed at, but eight command shapes the previous matcher denied are now allowed. Confirmed by feeding the same text to both versions; the guard only reads stdin, so nothing was executed. Filed upstream as template#40 rather than patched here, keeping .agents/hooks/* byte-identical to the render. What lands instead is the check that was missing: tests/test_harness_deny_list.py pins the verdict for 24 shapes. The guard is template-owned, so copier update rewrites it wholesale and a regression arrives as a clean, conflict-free update with nothing to review. The nine known-bad shapes are xfail(strict=True) against template#40, so an upstream fix fails the gate as an XPASS instead of passing unnoticed. Three project-owned defects fixed: - The decision-register scan substituted "$(git merge-base origin/main HEAD)", which expands to nothing when origin/main does not resolve — a fresh repo, a shallow CI clone, a fork on another integration branch — degenerating the range to HEAD...HEAD and reporting "no escalations" for any diff. The three-dot form resolves the same merge base (verified byte-identical output) and exits 128 on a ref it cannot find. This is the second defect in this one snippet: it is the register contract's only stated check, and both times it failed by looking clean. - harness-usage.md documented one SessionStart hook; this branch added a second, the payload-reader probe whose warning is what precedes every Bash call being denied. - The wheel-vs-sdist invariant lived only in development/README.md, which is template-owned and unprotected by _skip_if_exists. Moved to architecture.md#distribution-boundary, which is protected; the README now points there and says why. Also filed upstream as template#41: the jq probe proves parsing but not query compatibility, so a jq older than 1.5 denies every Bash call while blaming a damaged reader; and the PreToolUse proxy checks the guard is readable, not that it runs, so a truncated guard exits 0 and allows everything. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> --- CHANGELOG.md | 12 ++++ development/README.md | 10 +-- development/adr/README.md | 17 +++-- development/architecture.md | 9 +++ development/harness-usage.md | 5 +- tests/test_harness_deny_list.py | 120 ++++++++++++++++++++++++++++++++ 6 files changed, 163 insertions(+), 10 deletions(-) create mode 100644 tests/test_harness_deny_list.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 4256054..3fcc3d1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,13 @@ and this project adheres to ### Added +- `tests/test_harness_deny_list.py` pins the verdict of the agent harness's + destructive-command guard for 24 command shapes. The guard is template-owned, + so `copier update` rewrites it wholesale and a regression would otherwise + arrive as a clean, conflict-free update. Nine shapes the current template + allows and an earlier one denied are `xfail(strict=True)` against + [template#40](https://github.com/grAItools/harness-copier-template/issues/40), + so an upstream fix fails the gate as an XPASS rather than passing unnoticed. - `gpu-test-cuda` extra and `make test-gpu-cuda` to run the CUDA (T2) hardware suite, with the setup recipe in [`development/testing.md`](development/testing.md#running-the-cuda-gpu-suite-on-hardware). First hardware run recorded in [ADR 0003](development/adr/0003-gpu-suite-waiver-for-0.1.0.md). @@ -56,6 +63,11 @@ and this project adheres to session with no gate having run. `hook-input.sh` now probes `jq` by running it, so a `jq` that resolves but is broken falls through to `python3` instead of failing the read. +- The decision-register scan in `development/adr/README.md` used + `"$(git merge-base origin/main HEAD)"...HEAD`, which degenerates to + `HEAD...HEAD` — reporting "no escalations" — whenever `origin/main` does not + resolve. It now uses the three-dot form directly, which resolves the same + merge base and exits 128 on a ref it cannot find. - `integrations.numba`: `DevmmEMMPlugin` passes the device pointer as a `ctypes.c_void_p`, the only type numba-cuda ≥ 0.30 converts to a driver pointer. diff --git a/development/README.md b/development/README.md index 381db94..9aa29f4 100644 --- a/development/README.md +++ b/development/README.md @@ -3,11 +3,11 @@ Everything agents and developers need to work on this repo: conventions, decisions, and the per-feature work-document lifecycle. This tree is **repo-internal**: never rendered to a documentation site, and nothing here is -part of the public API. It is not secret, though — the sdist ships the whole -repo, so these files travel with `devmm-<version>.tar.gz` (the *wheel* carries -`src/devmm` alone, enforced by -`tests/test_packaging.py::test_wheel_packages_only_devmm`). Write accordingly: -no credentials, no per-developer paths. +part of the public API. It is not secret, though — these files ship inside the +sdist, so write accordingly: no credentials, no per-developer paths. See the +distribution boundary in [`architecture.md`](architecture.md#distribution-boundary), +which is where that invariant is recorded; this file is template-owned and a +`copier update` can revert it. User-facing documentation lives in `docs/` (kept for that purpose alone, the way `src/` is kept for sources) and is written for readers, not rewritten from diff --git a/development/adr/README.md b/development/adr/README.md index 22a5c1e..eb01b1a 100644 --- a/development/adr/README.md +++ b/development/adr/README.md @@ -69,13 +69,22 @@ while burying a real escalation among the quotes. Scan report files, not the whole tree: ```sh -git diff --unified=0 "$(git merge-base origin/main HEAD)"...HEAD \ +git diff --unified=0 origin/main...HEAD \ -- 'development/work/*/report.md' | grep -E '^\+.*DECISION-PENDING:' ``` -The revision range is load-bearing: a bare `git diff` compares the worktree to -the index, so on a committed branch it inspects nothing and the check passes -whatever the diff contains. +The revision range is load-bearing, in two ways. A bare `git diff` compares the +worktree to the index, so on a committed branch it inspects nothing and the +check passes whatever the diff contains. And the range must be written as the +three-dot form above rather than substituting `"$(git merge-base origin/main +HEAD)"`: three-dot already resolves the merge base, so the two agree whenever +both work, but they fail differently. A ref that does not resolve — a fresh +repo before its first push, a shallow CI clone, a fork whose integration branch +is not `origin/main` — makes the substitution expand to nothing, degenerating +the range to `HEAD...HEAD`; the scan then finds no markers and is +indistinguishable from a clean result, which is the one outcome this check must +never fake. The three-dot form exits 128 with `fatal: bad revision` instead. If +your integration branch is not `origin/main`, name yours. The register row is added by the **caller servicing the hand-back** — `/build` step 5, which puts the question to the user and appends the row in the same diff --git a/development/architecture.md b/development/architecture.md index fe0f6a6..24f970c 100644 --- a/development/architecture.md +++ b/development/architecture.md @@ -66,6 +66,15 @@ external boundaries are: effects**; each `install()` returns an `uninstall()`/context manager that restores the prior state. +## Distribution boundary + +The **wheel** carries `src/devmm` and nothing else — enforced at every gate by +`tests/test_packaging.py::test_wheel_packages_only_devmm`, alongside the +bare-venv install that keeps the required-dependency set empty. The **sdist** +ships the whole repository, so everything under `development/` travels inside +`devmm-<version>.tar.gz`: repo-internal is not the same as private, and nothing +here may carry credentials or per-developer paths. + ## Generated code None — `devmm` has no codegen pipeline. `uv.lock` is the only machine-generated diff --git a/development/harness-usage.md b/development/harness-usage.md index d003b3a..37eed46 100644 --- a/development/harness-usage.md +++ b/development/harness-usage.md @@ -76,7 +76,10 @@ Claude Code only). You cannot prompt around the hooks: to **install** `uv` if it is missing. The installer adds it to your shell profile (so it is on PATH for *new* shells); a hook can't change the agent's already-running shells, so the `Stop` hook also exports it onto - PATH for the verify gate. + PATH for the verify gate. A second `SessionStart` hook probes the payload + reader (`printf '{}' | .agents/hooks/hook-input.sh .probe`) and warns on exit + 1 if it fails — the warning to expect if every Bash call is about to be + denied, since the `PreToolUse` guard depends on that same reader. - **PostToolUse (`Write|Edit|MultiEdit`)** runs `scripts/fmt-file.sh` on the edited file after every write. - **PreToolUse (`Bash`)** hard-blocks `rm -rf`, `push --force`, `reset --hard`, diff --git a/tests/test_harness_deny_list.py b/tests/test_harness_deny_list.py new file mode 100644 index 0000000..e09cd38 --- /dev/null +++ b/tests/test_harness_deny_list.py @@ -0,0 +1,120 @@ +"""Behaviour lock for the agent harness's destructive-command guard. + +`.agents/hooks/block-destructive.sh` is the only thing standing between an +agent's Bash call and a recursive delete, a force-push or a hard reset. It is +template-owned (`gh:grAItools/harness-copier-template`), so `copier update` +rewrites it wholesale and a regression arrives as a clean, conflict-free +update. This file is the downstream check: it pins the verdict for each +command shape so a template bump that weakens the guard fails the gate instead +of landing silently. + +The guard matches text and only ever reads stdin, so every case here is a +string fed to a matcher — nothing in this file executes a destructive command. + +`xfail(strict=True)` marks shapes the current template allows and an earlier +one denied. Strict is the point: when upstream fixes one, the XPASS fails the +gate and says to promote the case back to a plain assertion. +""" + +from __future__ import annotations + +import shutil +import subprocess +from pathlib import Path + +import pytest + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +GUARD = PROJECT_ROOT / ".agents" / "hooks" / "block-destructive.sh" + +# The guard is a POSIX shell script; skip where there is no shell to run it. +pytestmark = pytest.mark.skipif(shutil.which("sh") is None, reason="no POSIX shell available") + +DENY = 2 +ALLOW = 0 + +# Upstream issue tracking the fail-open shapes marked xfail below. +UPSTREAM = "grAItools/harness-copier-template#40" + + +def verdict(command: str) -> int: + """Return the guard's exit code for `command` (2 = denied, 0 = allowed).""" + result = subprocess.run( + ["sh", str(GUARD)], + input=command, + capture_output=True, + text=True, + ) + assert result.returncode in (ALLOW, DENY), ( + f"unexpected exit {result.returncode} for {command!r}: {result.stderr}" + ) + return result.returncode + + +@pytest.mark.parametrize( + "command", + [ + "cd x && rm -rf y", + "git push --force origin main", + "git reset --hard origin/main", + "\\rm -rf /tmp/x", # the alias-bypass form is still the operation + "sh -c 'rm -rf /tmp/x'", # nested shell: quoting is not a mention + "echo 'rm -rf /tmp/x' | sh", # piped into a shell + "psql -c 'DROP TABLE users'", # SQL: denied quoted or not + "grep -rn 'DROP TABLE' .", + ], +) +def test_denies_destructive_commands(command: str) -> None: + assert verdict(command) == DENY + + +@pytest.mark.parametrize( + "command", + [ + "git log --oneline -5", + "make verify", + "grep -rn 'rm -rf' .", # a quoted mention is not a use + "rg 'reset --hard' development/", + "git commit -m 'note: never run rm -rf here'", + # A runner elsewhere on the line does not make a later mention a use. + "ssh host uptime && grep -rn 'rm -rf' .", + # `push` ends in `sh`; it must not read as a pipe into a shell. + "grep -rn 'rm -rf' . | git push", + ], +) +def test_allows_read_only_commands(command: str) -> None: + assert verdict(command) == ALLOW + + +@pytest.mark.xfail(strict=True, reason=f"fail-open in the template guard; {UPSTREAM}") +@pytest.mark.parametrize( + "command", + [ + # Rule 2's `[^'\"]*` halts at a quote nested inside the runner's script. + "bash -c 'git fetch && echo \"resetting\" && git reset --hard origin/main'", + # The `-…c` flag must be adjacent to the shell name. + "bash -euo pipefail -c 'rm -rf /tmp/x'", + "bash --norc -c 'rm -rf /tmp/x'", + # A continuation line starts inside the quoted string, so its closing + # quote is unmatched and nothing after it is reachable. + 'git commit -m "fix: thing\n\nLonger body." && rm -rf build', + # $'…' permits \' inside, desynchronising the quote-parity model. + "git commit -m $'fix don\\'t break' && rm -rf build", + # `piped` cannot cross a second pipe, and allows only a `sudo` prefix. + "echo 'rm -rf /data' | tee /tmp/x | sh", + "echo 'rm -rf /data' | env sh", + # Write-then-run: denied by the earlier plain-substring form. + "printf 'rm -rf /tmp/x' > s.sh; sh s.sh", + ], +) +def test_known_fail_open_shapes(command: str) -> None: + """Shapes an earlier template denied that the current one allows.""" + assert verdict(command) == DENY + + +@pytest.mark.xfail( + strict=True, reason=f"--force-with-lease is prefix-matched by --force; {UPSTREAM}" +) +def test_allows_force_with_lease() -> None: + """`--force-with-lease` is the safe, lease-checked push; it has no rewrite.""" + assert verdict("git push --force-with-lease origin main") == ALLOW