diff --git a/.claude/settings.json b/.claude/settings.json deleted file mode 100644 index b70ac80..0000000 --- a/.claude/settings.json +++ /dev/null @@ -1,78 +0,0 @@ -{ - "permissions": { - "allow": [ - "Bash(bash:*)", - "Bash(cat:*)", - "Bash(chmod:*)", - "Bash(cp:*)", - "Bash(date:*)", - "Bash(diff:*)", - "Bash(echo:*)", - "Bash(find:*)", - "Bash(grep:*)", - "Bash(glab api:*)", - "Bash(glab ci list:*)", - "Bash(glab ci status:*)", - "Bash(glab ci trace:*)", - "Bash(glab mr list:*)", - "Bash(glab mr view:*)", - "Bash(gh issue view:*)", - "Bash(gh pr diff:*)", - "Bash(gh pr list:*)", - "Bash(gh pr view:*)", - "Bash(gh run list:*)", - "Bash(gh run view:*)", - "Bash(gh run watch:*)", - "Bash(git add:*)", - "Bash(git checkout:*)", - "Bash(git clone:*)", - "Bash(git commit -m:*)", - "Bash(git commit --amend:*)", - "Bash(git commit:*)", - "Bash(git diff:*)", - "Bash(git fetch:*)", - "Bash(git log:*)", - "Bash(git merge-base:*)", - "Bash(git mv:*)", - "Bash(git pull:*)", - "Bash(git push --force-with-lease:*)", - "Bash(git push origin:*)", - "Bash(git rebase --continue:*)", - "Bash(git rebase:*)", - "Bash(git rev-parse:*)", - "Bash(git rm:*)", - "Bash(rg:*)", - "Bash(git stash:*)", - "Bash(git status:*)", - "Bash(git switch:*)", - "Bash(git worktree:*)", - "Bash(jq:*)", - "Bash(just:*)", - "Bash(ls:*)", - "Bash(mise run:*)", - "Bash(mkdir:*)", - "Bash(mv:*)", - "Bash(head:*)", - "Bash(notify-send:*)", - "Bash(npm install:*)", - "Bash(npm run:*)", - "Bash(npx:*)", - "Bash(osascript:*)", - "Bash(pnpm run docs:build:*)", - "Bash(printf:*)", - "Bash(python3:*)", - "Bash(rm:*)", - "Bash(sed:*)", - "Bash(sort:*)", - "Bash(tr:*)", - "Bash(xargs:*)", - "Bash(wc:*)", - "Edit", - "Read(~/.claude/plugins/**/skills/**)", - "Read(~/.claude/skills/**)", - "WebFetch", - "WebSearch", - "Write" - ] - } -} diff --git a/.gitignore b/.gitignore index 8b4f248..9a78ec1 100644 --- a/.gitignore +++ b/.gitignore @@ -85,10 +85,6 @@ test-results/ *.old # Instance-specific development files (not committed) -.claude/plan.md -.claude/todos.md -.claude/review.md -.claude/skills/ .agents/ # codevoyant context store — in-repo .codevoyant is a symlink to ~/.codevoyant//; never commit it @@ -103,8 +99,7 @@ test-results/ # codevoyant snippets store (in-project fallback — never commit generated snippets) .codevoyant/snippets/ -# Claude Code agent state -.claude/worktrees/ +# Codevoyant agent state .memsearch/ # Generated / untracked assets diff --git a/.mise-tasks/vendor-assets b/.mise-tasks/vendor-assets index 062cc9f..56390f9 100755 --- a/.mise-tasks/vendor-assets +++ b/.mise-tasks/vendor-assets @@ -25,9 +25,11 @@ # the whole source dir is copied; when present the listed entries are the # minimum set that must be vendored — any other file under the source is also # vendored, so a file added to the shared source propagates automatically. The -# vendored set for each target is recorded in skills/vendor.manifest.json so -# --check can flag (and vendor can clean) a previously-vendored copy whose file -# is no longer in the source or the allowlist. +# vendored set for each (asset, target) pair is recorded in +# skills/vendor.manifest.json (key "::") so --check can flag +# (and vendor can clean) a previously-vendored copy whose file is no longer in +# the source or the allowlist — without one asset's cleanup deleting files +# vendored into the same target dir by another asset. # # The Agent Skills standard scopes every skill to its own directory: file # references are relative and one level deep, and there is no cross-skill @@ -228,10 +230,12 @@ for name, asset in assets.items(): targets = resolve_targets(name, asset) files = asset.get("files") if files: - managed = sorted(set(files) | set(source_relfiles(src))) + # `files` is an exclusive filter per the documented contract ("copy only + # these") — tests and other source-dir residents stay in skills/shared/. + managed = sorted(set(files)) for t in targets: files_targets.add(t) - prev = set(manifest.get(t, [])) + prev = set(manifest.get(f"{name}::{t}", [])) stale = sorted(prev - set(managed)) if mode == "check": for f in managed: @@ -252,7 +256,7 @@ for name, asset in assets.items(): if os.path.lexists(os.path.join(root, d)): remove_path(d) print(f"removed stale: {d}") - manifest[t] = managed + manifest[f"{name}::{t}"] = managed else: for t in targets: if mode == "check": @@ -262,7 +266,7 @@ for name, asset in assets.items(): print(f"vendored: {src} -> {t}") if mode != "check": - manifest = {t: v for t, v in manifest.items() if t in files_targets} + manifest = {k: v for k, v in manifest.items() if "::" in k and k.split("::", 1)[1] in files_targets} save_manifest(manifest) sys.exit(fail) diff --git a/.releaserc.json b/.releaserc.json index 8b41f57..da726ea 100644 --- a/.releaserc.json +++ b/.releaserc.json @@ -32,7 +32,7 @@ [ "@semantic-release/exec", { - "prepareCmd": "echo ${nextRelease.version} > version.txt && node scripts/sanitize-changelog.js" + "prepareCmd": "echo ${nextRelease.version} > version.txt && mise run changelog:sanitize" } ], [ diff --git a/CLAUDE.md b/AGENTS.md similarity index 100% rename from CLAUDE.md rename to AGENTS.md diff --git a/docs/.vitepress/config.mjs b/docs/.vitepress/config.mjs index 296d62a..49af877 100644 --- a/docs/.vitepress/config.mjs +++ b/docs/.vitepress/config.mjs @@ -14,11 +14,10 @@ export default defineConfig({ // Root-level files "README.md", "CHANGELOG.md", - "CLAUDE.md", + "AGENTS.md", // Internal dirs - ".claude/**", - ".memsearch/**", ".codevoyant/**", + ".memsearch/**", // Shared skill assets (source of truth for vendored templates/scripts — not docs pages) "skills/shared/**", // Non-recipe skills (exclude entirely) diff --git a/docs/skills/docs.md b/docs/skills/docs.md index 3f7516b..9c18ed3 100644 --- a/docs/skills/docs.md +++ b/docs/skills/docs.md @@ -62,7 +62,7 @@ Checks for: required sections (per template order), prescribed Mermaid diagram t ### retcon -- author the whole docs/ tree from the codebase -The only `docs` command that writes real content. `retcon` reads the code, scaffolds each doc the same way `new` does, then replaces every `@agent` marker with accurate prose, diagrams, and tables. It first handles existing docs: it moves them to `docs/legacy/`, confirms their facts against the code, and carries those facts forward (asking before carrying machine-generated content). It finishes by validating every doc's `globs` against the real code tree. +The only `docs` command that writes real content. `retcon` reads the code, scaffolds each doc the same way `new` does, then fills the contract surface — tables, mermaid diagrams, runnable code samples, constrained `## Requirements`, and `## References` — replacing each `@agent` marker it consumes. It never writes prose elsewhere (see the prose policy, `references/prose-policy.md`): sections marked `` keep their marker untouched until a person writes them. It first handles existing docs: it moves them to `docs/legacy/`, confirms their facts against the code, and carries those facts forward (asking before carrying machine-generated content). It finishes by validating every doc's `globs` against the real code tree. ```bash /docs retcon # author the whole docs/ tree diff --git a/docs/skills/loop.md b/docs/skills/loop.md new file mode 100644 index 0000000..ae7f203 --- /dev/null +++ b/docs/skills/loop.md @@ -0,0 +1,25 @@ +--- +title: loop +--- + +# loop + +Repeat a task until its objective is met or a max iteration count is reached. A loop is not a saved artifact like a flow — `/loop` creates a tracking doc and runs immediately, with every iteration executed by a single background loop agent that performs the task and judges the objective. + +## Usage + +```bash +/loop fix the failing lint errors --until "mise run lint exits 0" --max 5 +/loop keep triaging the backlog --until "all P0 issues are closed" --check "gh issue list --label P0 --state open | wc -l | grep -qx 0" +/loop continue the earlier pass --resume fix-lint-errors +``` + +- **task** (required) — what to repeat each iteration: a skill command, shell command, or agent instruction. +- **--until** (required) — the objective: the verifiable condition that ends the loop, phrased as an outcome. +- **--max N** (default 3) — the hard upper bound; the loop stops at N iterations even if the objective is not met. +- **--check ** (optional) — a deterministic check that exits 0 when the objective is met; when present it overrides the agent's verdict. +- **--resume ** — continue an existing loop's tracking doc instead of starting a new one. + +## How a run works + +Each invocation writes `.codevoyant/loops/{slug}/loop.md` — the tracking doc holding the task, objective, check, bound, status, and one row per iteration — then runs: for each iteration it spawns one `loop-agent` background agent that performs the task and strictly judges the objective from the actual repo state (never from its own claim). On a MET verdict (or a zero-exit `--check`) the loop stops with status `complete`; at the bound it stops with `max-reached`. If an iteration needs input, the question is escalated to the user and the same iteration re-runs without consuming the bound twice. diff --git a/docs/skills/pr.md b/docs/skills/pr.md index 777f6e0..448002c 100644 --- a/docs/skills/pr.md +++ b/docs/skills/pr.md @@ -34,12 +34,14 @@ Read a PR/MR diff and generate AI-authored inline comments. Comments are terse ( Reviews evaluate the change against its **stated intent** first — does the diff actually deliver the PR/MR's purpose end-to-end (tracing the headline use case), not just whether the code is clean? A well-formed change that fails its intent is flagged `BLOCKING`. -Assesses the change with **four subagents in parallel**, one per dimension, then merges their findings into one review: +Runs deterministic pre-checks (CI status, commit-convention consistency, and a static-analysis floor), then assesses the change with **five subagents in parallel** — one per dimension — plus a **claim-checker**, and merges all findings into one review: - **Intent-match** — does the diff deliver the stated intent (from the description, linked issue, or executed spec plan) end-to-end? A well-formed change that fails its intent is `BLOCKING`. - **Unnecessary changes** — a dedicated **slop-detector**: scope creep, stray edits, dead/commented code, accidental reverts, stochastic churn (random renames, reordering, reformatting), boilerplate, debug leftovers, dependency creep. Findings prefixed `Slop:`. A prevalent problem with agentic coding. - **Code quality** — a **code-quality-auditor** judges the added/edited code against the relevant codevoyant skill (`typescript`, `python`, `react`, `svelte`, `sveltekit`, …) or the language/framework standard. Findings prefixed `Quality:`. - **Docs freshness** — a **docs-freshness-checker** decides whether docs should have been updated. By default review stays read-only: stale docs are reported as a `Docs:` finding recommending `/docs update`. Pass `--update-docs` to opt in to having the pass run `/docs update` and refresh docs during the review. Findings prefixed `Docs:`. +- **Adversarial hunt** — a **red-team-adversary** tries to break the change: failure modes, edge cases, negative paths, mutation-mindset test review, STRIDE on security surfaces. Findings prefixed `Adversarial:` — `BLOCKING` only when they carry a concrete input/expected/observed scenario. +- **Claim check** — a **claim-checker** verifies the PR/MR body's claims (Changes bullets, Validation checklist, stated behavior) against the diff. Findings prefixed `Claim:`. ```bash /pr review # draft the review directly on the PR/MR diff --git a/docs/skills/spec.md b/docs/skills/spec.md index 2b6877e..09624ef 100644 --- a/docs/skills/spec.md +++ b/docs/skills/spec.md @@ -8,6 +8,8 @@ Specification-driven development — create structured plans from requirements, Explore requirements and produce a multi-phase implementation plan with objectives, design decisions, and per-phase specs. Every task carries the **complete, ready-to-write code** it will produce. Before `new` reports a plan ready, a mandatory code-completeness gate scans every task and rejects stubs, placeholder markers, omitted code, and prose-only descriptions; it reruns after repairs and fails closed if literal code cannot be resolved. `--validate` still adds the broader multi-agent validation pass. +Enumerable sets in the objective or intent are **tabulated**: rote replacements, target sets to search or touch, and enumerated requirement lists each get a row-per-item table under `.codevoyant/spec/{plan}/tables/`, enumerated from the codebase (never from memory), with every requirement row carrying an Intent ref back to `intent.md`. A completeness gate re-runs each table's enumeration, checks every intent item appears in a row and every row is owned by exactly one task, and fails the plan on any dropped item, orphan row, or drift. + Two ways to give the objective: - **Inline objective** — a description; planning starts immediately. @@ -32,7 +34,7 @@ Two ways to give the objective: `--branch` and `--worktree` are independent — each does one thing, and neither implies the other. `--branch` creates or switches to a branch (bare: derived from the plan slug; with a name: that name). `--worktree` creates a worktree (bare: `.codevoyant/worktrees/`; with a path: that path). Both delegate to the shared `/git worktree` routine. -`--persistent` is an **experimental** doc-aware mode: docs are written first, every phase is scoped to the doc globs it may write, and cross-module interaction happens only through documented public interfaces. It requires valid docs in the repo (`docs/` with `globs:` frontmatter plus an architecture index or a component doc with a public API/interface section); `new` refuses to plan blind otherwise. See the skill's `references/doc-aware.md` for the full model. +`--persistent` is an **experimental** doc-aware mode: docs are written first, every phase is scoped to the doc globs it may write, and cross-module interaction happens only through documented public interfaces. Cross-module changes are discouraged by default (Rule 7): a phase that must write across module boundaries has to call the crossing out with a reason and a rejected restructure, and the executor refuses uncalled-out crossings. It requires valid docs in the repo (`docs/` with `globs:` frontmatter plus an architecture index or a component doc with a public API/interface section); `new` refuses to plan blind otherwise. See the skill's `references/doc-aware.md` for the full model. Pass a Linear, GitHub, or GitLab issue URL as the first argument to pre-fill requirements from the issue title, description, and comments. diff --git a/mise.toml b/mise.toml index 961678e..66fb244 100644 --- a/mise.toml +++ b/mise.toml @@ -12,6 +12,7 @@ shfmt = "latest" gh = "latest" glab = "latest" "npm:skills-ref" = "0.1.5" +"npm:@mermaid-js/mermaid-cli" = "11.16.0" # ============================================================================== # DOCS @@ -49,6 +50,10 @@ run = "cat version.txt" description = "Run semantic-release to version and update changelog" run = ".mise-tasks/upversion" +[tasks."changelog:sanitize"] +description = "Escape bare HTML tags in CHANGELOG.md after generation" +run = "node scripts/sanitize-changelog.js" + # ============================================================================== # SKILLS # ============================================================================== diff --git a/skills/docs/references/docs-review-template.md b/skills/docs/references/docs-review-template.md index 9fe7719..4e1bc22 100644 --- a/skills/docs/references/docs-review-template.md +++ b/skills/docs/references/docs-review-template.md @@ -55,7 +55,7 @@ To re-review after manual edits: /docs review {path} -- regenerates this report ``` -**Severity types:** `STRUCTURE` (missing/malformed section), `DIAGRAM` (missing/wrong diagram type), `LANGUAGE` (language-guide or STE violation), `REFERENCE` (missing References section or entries), `COVERAGE` (missing/duplicate `globs` coverage or API-boundary violation — see `references/coverage-and-api.md`), `GLOB` (a doc's `globs` matches no real paths, or a discovered component has no owning doc — from `validate`). +**Severity types:** `STRUCTURE` (missing/malformed section), `DIAGRAM` (missing/wrong diagram type), `LANGUAGE` (language-guide or STE violation), `REQUIREMENTS` (a requirement in `## Requirements` violates R1–R7 of `references/requirements-guidance.md`), `REFERENCE` (missing References section or entries), `PROSE` (LLM prose outside the prose-policy allowance — not in a `` comment, not in `## Requirements`, not in `## References`, not a minimal artifact label; see `references/prose-policy.md`), `COVERAGE` (missing/duplicate `globs` coverage or API-boundary violation — see `references/coverage-and-api.md`), `GLOB` (a doc's `globs` matches no real paths, or a discovered component has no owning doc — from `validate`). **Principles:** - Each replacement preserves all surrounding human-authored text. The replacement block contains ONLY the text that changes, not the entire file. diff --git a/skills/docs/references/language-guide.md b/skills/docs/references/language-guide.md index 1bf9d42..80a8ade 100644 --- a/skills/docs/references/language-guide.md +++ b/skills/docs/references/language-guide.md @@ -71,5 +71,12 @@ When updating existing docs, preserve human-authored text. Change only text that 8. **Slop vocabulary** (STE slop table): leverage, utilize, ensure, in order to, functionality, enables you to, allows you to, is designed to, aims to, dive into, delve into, robust, powerful, comprehensive, seamlessly, facilitate, streamline, and/or, etc. → the plain replacement from the ruleset. 9. **Condition-first** (STE 5.4): a sentence where `if`/`when` stands after the command ("Increase the timeout if the network is slow" → "If the network is slow, increase the timeout"). 10. **Procedural imperative** (STE 5.3): in a procedural section, an instruction written as a statement instead of an imperative. - -For each violation, record `type: LANGUAGE`, `current_text` = the exact offending sentence, `replacement_text` = the minimal rewrite that fixes only that violation, `rationale` = the rule number/name. Never rephrase working prose for style. +11. **R1 no-impl-terms** (requirements-guidance): a `## Requirements` bullet naming an endpoint, route, class, function, table, SQL, UI widget, or file path as if it were the requirement. +12. **R2 survive-change** (requirements-guidance): a requirement whose wording would need to change if the implementation changed. +13. **R3 fit-criterion** (requirements-guidance): a Functional requirement with no observable outcome or measurable success condition. +14. **R4 smells** (requirements-guidance): subjective language, ambiguous adverbs/adjectives, superlatives, totality terms, baseline-less comparatives in a requirement. +15. **R5 invariant** (requirements-guidance): an implementation invariant stated as a Functional requirement. +16. **R6 source** (requirements-guidance): a domain claim with neither `Source:` nor `[ASSUMPTION — unvalidated]`. +17. **R7 verbs** (requirements-guidance): a requirement using should/would/can instead of the template's prescribed verbs. + +For each violation, record `type: LANGUAGE` (checks 1–10) or `type: REQUIREMENTS` (checks 11–17, applied only inside `## Requirements` sections), `current_text` = the exact offending sentence, `replacement_text` = the minimal rewrite that fixes only that violation, `rationale` = the rule number/name. Never rephrase working prose for style. diff --git a/skills/docs/references/mermaid-guide.md b/skills/docs/references/mermaid-guide.md index a403e59..2c5b23a 100644 --- a/skills/docs/references/mermaid-guide.md +++ b/skills/docs/references/mermaid-guide.md @@ -171,6 +171,17 @@ stateDiagram-v2 - Use `[*]` for entry/exit states - Avoid showing error states unless they have transitions back to valid states +## Artifact quality gate + +`scripts/validate_artifacts.py` enforces these rules on every generated doc (retcon blocks on failures; review and validate report them). One pinned renderer decides what "valid" means: `mmdc` at mermaid-cli 11.16.0 (PATH, else `npx -y @mermaid-js/mermaid-cli@11.16.0`). + +- Every mermaid fence must render (parse failures are blocking). +- Sequence diagrams: ≤8 participants. Graphs/flowcharts: ≤12 nodes. Split larger diagrams. +- Node labels break lines with `
`, never a literal `\n`. +- Tables carry a separator row and no unfilled `{placeholder}`-only rows. + +The semantic caps come from the C4 review checklist: large diagrams carry too much cognitive load to be read; split by focus instead. + ## When NOT to Use a Diagram - Simple 2-step flows (just use prose or a bullet list) diff --git a/skills/docs/references/ml-examples.md b/skills/docs/references/ml-examples.md index 723d861..4294265 100644 --- a/skills/docs/references/ml-examples.md +++ b/skills/docs/references/ml-examples.md @@ -1,5 +1,7 @@ # ML examples — boundaries and requirements +> The functional-vs-non-functional-vs-invariant rule taught here is generalized to all templates in `references/requirements-guidance.md` (R5). The examples below remain the ML-flavored worked subset. + Worked examples for the ml-model, data-pipeline, and experiment templates. The rule that matters most: how a class operates is an implementation invariant, never a functional requirement. ## Module boundaries diff --git a/skills/docs/references/prose-policy.md b/skills/docs/references/prose-policy.md new file mode 100644 index 0000000..6cd77d5 --- /dev/null +++ b/skills/docs/references/prose-policy.md @@ -0,0 +1,23 @@ +# prose-policy — what an LLM may write in a generated doc + +Generated docs are contract-forward. The artifacts (tables, mermaid diagrams, runnable code samples) are the content; prose is what a human writes against them. The LLM's text budget is closed below — anything not listed is a `PROSE` finding in review. + +## Allowed LLM text + +1. **HTML comments.** `` guidance consumed during generation, and `` fill-in prompts that stay in the doc until a human replaces them with prose. +2. **`## Requirements`.** Constrained requirement bullets per `references/requirements-guidance.md` (domain outcomes, never implementation restatements). +3. **`## References`.** Real verified technical/external sources. +4. **Artifact internals.** Table cells, mermaid node/edge labels, and code-sample comments — minimal and barebones: identifiers, types, one-phrase labels. Never narrative sentences inside artifacts. + +## Everything else + +Sections not listed above carry a `` marker and nothing more until a person writes them. The generator never writes prose there, never deletes the marker, and never flags the section as incomplete. + +## Marker semantics + +| Marker | Generator behavior | +| --- | --- | +| `` | Replace with artifacts only (tables/diagrams/code/constrained requirements/references). If the section's content would be prose, emit nothing and leave the marker. | +| `` | Preserve verbatim. Never fill, never delete. It is the hand-off to the human author. | + +Docs written before this policy keep working: human prose already in place is preserved (`update`'s preserve-human-text rule), and review flags only new LLM prose that violates the policy, never text a human wrote. diff --git a/skills/docs/references/requirements-guidance.md b/skills/docs/references/requirements-guidance.md new file mode 100644 index 0000000..9cff23b --- /dev/null +++ b/skills/docs/references/requirements-guidance.md @@ -0,0 +1,46 @@ +# requirements-guidance — how requirements must read + +The single rule set for every requirement the skills write: docs templates' `## Requirements` sections, spec plan.md `## Requirements`, and plan/pm templates that reference this file. Requirements state domain/business purpose (what/why), never design or implementation (how). + +> **Markdown output: soft-wrap prose, never hard-wrap** — when a skill writes a `.md` artifact per this guidance, write each paragraph as one continuous line; do not insert manual newlines to wrap prose at a fixed column width. Newlines still separate paragraphs, list items, headings, and code fences. + +## The rules + +| Rule | Name | Check | +| --- | --- | --- | +| R1 | no-impl-terms | Functional requirements contain no endpoint/route names, class/function names, table names, SQL, UI widgets, or file paths. Identifiers belong in Design/Implementation. Untouchables (code blocks, quoted identifiers) stay exact. | +| R2 | survive-change | "Would this wording need to change if the implementation did?" If yes, rewrite. "The pipeline reads from S3, transforms with Spark, writes to BigQuery" fails; "orders are available to consumers within 25 hours of placement" passes. | +| R3 | fit-criterion | Every Functional requirement names an observable outcome or measurable success condition. "The feature works correctly" fails by format. | +| R4 | smells | No subjective language, ambiguous adverbs/adjectives, superlatives, totality terms ("always", "never"), comparative phrases ("faster", "better") without a baseline. | +| R5 | invariant | Functional requirements do not state implementation invariants. How a class operates internally ("the encoder caches tokens") is an invariant — it belongs in Implementation, never in Functional. | +| R6 | source | Domain claims carry `Source: {citation or evidence}` or are marked `[ASSUMPTION — unvalidated]`. A requirement with neither is flagged, not silently accepted. | +| R7 | verbs | Requirements use the template's prescribed verbs (must / returns / rejects / produces / reports / accepts). "Should", "would", "can" as requirement verbs fail (STE banned modals). | + +## Per-template verb table + +| Template | Functional verbs | Non-Functional emphasis | +| --- | --- | --- | +| api | must, returns, rejects | rate limits, latency, idempotency, security | +| auth | must, rejects | token lifetime, storage, transport, revocation | +| library | must, returns | API surface, error handling, performance | +| data-pipeline | must, produces | throughput, freshness SLO, retry/backoff — numbers | +| ml-model | must, returns, accepts | latency, throughput, accuracy targets, hardware — numbers, not vibes | +| experiment | must, reports | runtime, resource budget, reproducibility — numbers | +| frontend | must (render/handle) | responsiveness, accessibility, performance | +| generic | must, returns, rejects | performance, security, reliability, operability | + +## Good vs slop, per case + +**Services / APIs.** Slop: "Given I visit /login and press the login button…" (procedural UI restatement). Good: "When a returning user signs in, they reach their dashboard without re-entering credentials. Source: {evidence}". Slop: "Must create a SessionService class". Good: "Must issue a session on valid credentials and reject expired or tampered tokens". + +**Libraries / SDKs.** Slop: "The SDK exposes paginate() using a cursor token" (API design, belongs in an ADR/Design). Good: "Callers can retrieve a stable page of results so multi-page reads never miss or duplicate records — Source: {evidence}". + +**Data pipelines.** Slop: "Reads from S3, transforms with Spark, writes to BigQuery". Good: "Orders are available to consumers within 25 hours of placement; late arrivals are included in the next run. Source: {evidence}". + +**ML models.** Slop: "Uses a 24-layer transformer trained with Adam" (architecture trivia). Good: "Returns accurate toxicity labels for English text; used for content moderation, not user scoring. Source: {evidence}" — intended use + out-of-scope, model-card style. + +**Infra / CI.** Slop: "Deploys Terraform ECS Fargate behind an ALB" (mechanism). Good: "99% of Get calls complete in under 100 ms, measured across backends over 1 minute. Source: {evidence}" — SLO language, few and defensible, never absolutes. + +## How gates apply + +Docs review Step 3c runs R1–R7 as executable checks over `## Requirements` sections (LANGUAGE-style finding contract, severity REQUIREMENTS). Spec planning runs the SCOPE=requirements agent over plan.md Requirements (BLOCK when the objective is entirely a deliverable list — ask "What changes for users or the business if this ships successfully?"). Judgments are by intent, not blind substring matching — the same rule as the code-completeness blocklist. diff --git a/skills/docs/references/template-contract.md b/skills/docs/references/template-contract.md index fdb5070..a864041 100644 --- a/skills/docs/references/template-contract.md +++ b/skills/docs/references/template-contract.md @@ -18,6 +18,7 @@ Workflows read these markers from the template. Do not invent other marker token | Marker | Meaning | Used by | | ----------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------- | | `` | The base annotation form — authoring guidance with no mechanical effect (never a line-level edit). Carries no token. The shared contract is `skills/shared/annotations.md`. | retcon, update | +| `` | A permanent fill-in prompt: the generator never writes prose here and never deletes the marker (see `references/prose-policy.md`). The section is still required — its presence, not its content, is checked. | retcon, update, review 3a | | `` | Marks the **block it precedes** as optional: a section whose FIRST marker under the heading starts with `(optional)`, OR a diagram/table block inside a required section. Absence of an optional block is never flagged. | review 3a, update --scaffold, retcon | | `` | This heading is Design's **Components** subsection — where a parent names+links its child docs (Rule 3). It may contain the `### Components` system diagram. Exactly one per component and architecture template. Always required. | review 3f, coverage-and-api Step C, validate | | `` | This heading is the doc's **public API surface** (Rule 4). Exactly one per component template. Its heading is the API section other docs may reference. Always required. | review 3e, coverage-and-api Rule 4, validate | @@ -35,7 +36,7 @@ Collect every `##` and `###` heading in template order. A heading is **required* - the first `` marker directly under it starts with `(optional)`, AND - it does NOT carry a `[components]` or `[public-api]` marker. -A `[components]` or `[public-api]` heading is always required, even if an `(optional)` diagram block sits inside it. The doc must contain every required heading (in template order). That is the entire structure check — edit the template, and what review requires changes with it. +A `[components]` or `[public-api]` heading is always required, even if an `(optional)` diagram block sits inside it. The doc must contain every required heading (in template order). That is the entire structure check — edit the template, and what review requires changes with it. A required heading whose template marker is `` is satisfied by the heading plus the marker (or by human-written prose replacing it) — an unfilled `@human` section is never a gap. ### 2. Required diagrams diff --git a/skills/docs/references/templates/api.md b/skills/docs/references/templates/api.md index 2957635..0ac37a0 100644 --- a/skills/docs/references/templates/api.md +++ b/skills/docs/references/templates/api.md @@ -8,20 +8,20 @@ globs: ## Overview - + ## Requirements ### Functional - + - {auth/access requirement} - {response-shape requirement} ### Non-Functional - + - {rate-limit / latency requirement} @@ -115,7 +115,7 @@ sequenceDiagram ### Query Patterns - + ### Configuration / Environment Variables diff --git a/skills/docs/references/templates/architecture.md b/skills/docs/references/templates/architecture.md index 171abf2..8bd7f23 100644 --- a/skills/docs/references/templates/architecture.md +++ b/skills/docs/references/templates/architecture.md @@ -9,7 +9,7 @@ globs: ## Overview - + ### Technology Stack @@ -60,7 +60,7 @@ graph LR ### {Section Heading, Repeats} - + ## References diff --git a/skills/docs/references/templates/auth.md b/skills/docs/references/templates/auth.md index 9818082..e6ab144 100644 --- a/skills/docs/references/templates/auth.md +++ b/skills/docs/references/templates/auth.md @@ -7,20 +7,20 @@ globs: ## Overview - + ## Requirements ### Functional - + - {e.g. Must issue a session on valid credentials} - {e.g. Must reject expired or tampered tokens} ### Non-Functional - + - {e.g. Tokens must be signed and short-lived} - {e.g. Must not expose refresh tokens to client-side JavaScript} @@ -113,7 +113,7 @@ sequenceDiagram ### Modules/Objects - + ### Environment Variables diff --git a/skills/docs/references/templates/ci.md b/skills/docs/references/templates/ci.md index 61d6fe8..fae081a 100644 --- a/skills/docs/references/templates/ci.md +++ b/skills/docs/references/templates/ci.md @@ -34,7 +34,7 @@ globs: ## Overview - + ```mermaid flowchart LR @@ -77,7 +77,7 @@ flowchart LR #### {module-name} - + ### Environments @@ -89,7 +89,7 @@ flowchart LR ### Provisioning - + ### Resources diff --git a/skills/docs/references/templates/data-pipeline.md b/skills/docs/references/templates/data-pipeline.md index bbfdeb4..f4974f2 100644 --- a/skills/docs/references/templates/data-pipeline.md +++ b/skills/docs/references/templates/data-pipeline.md @@ -8,19 +8,19 @@ globs: ## Overview - + ## Requirements ### Functional - + - {requirement} ### Non-Functional - + - {requirement} @@ -54,7 +54,7 @@ graph LR ### {Section Heading, Repeats} - + ## References diff --git a/skills/docs/references/templates/experiment.md b/skills/docs/references/templates/experiment.md index fea5dcf..b9575cd 100644 --- a/skills/docs/references/templates/experiment.md +++ b/skills/docs/references/templates/experiment.md @@ -8,19 +8,19 @@ globs: ## Overview - + ## Requirements ### Functional - + - {requirement} ### Non-Functional - + - {requirement} @@ -53,7 +53,7 @@ graph TD ### {Section Heading, Repeats} - + ## References diff --git a/skills/docs/references/templates/frontend.md b/skills/docs/references/templates/frontend.md index 91f693e..8a460c1 100644 --- a/skills/docs/references/templates/frontend.md +++ b/skills/docs/references/templates/frontend.md @@ -7,19 +7,19 @@ globs: ## Overview - + ## Requirements ### Functional - + - {state requirements: empty, loading, error} ### Non-Functional - + - {responsive/a11y requirement} - {performance requirement} @@ -57,7 +57,7 @@ flowchart TD ### State Management - + ### API @@ -79,11 +79,11 @@ flowchart TD ### Data Loading - + ### Accessibility - + - {keyboard navigation} - {screen reader support} diff --git a/skills/docs/references/templates/generic.md b/skills/docs/references/templates/generic.md index c35e11c..597bc13 100644 --- a/skills/docs/references/templates/generic.md +++ b/skills/docs/references/templates/generic.md @@ -8,19 +8,19 @@ globs: ## Overview - + ## Requirements ### Functional - + - {requirement} ### Non-Functional - + - {requirement} @@ -48,7 +48,7 @@ graph TD ### {Section Heading, Repeats} - + ## References diff --git a/skills/docs/references/templates/library.md b/skills/docs/references/templates/library.md index 49026dc..a969413 100644 --- a/skills/docs/references/templates/library.md +++ b/skills/docs/references/templates/library.md @@ -8,19 +8,19 @@ globs: ## Overview - + ## Requirements ### Functional - + - {e.g. Must work in both browser and server environments} ### Non-Functional - + - {e.g. Must not expose internal Firestore types in the public API} @@ -74,7 +74,7 @@ const result = await {functionName}({example args}) ### Modules/Objects - + ## References diff --git a/skills/docs/references/templates/ml-model.md b/skills/docs/references/templates/ml-model.md index 3256a9f..ecf4230 100644 --- a/skills/docs/references/templates/ml-model.md +++ b/skills/docs/references/templates/ml-model.md @@ -8,19 +8,19 @@ globs: ## Overview - + ## Requirements ### Functional - + - {requirement} ### Non-Functional - + - {requirement} @@ -47,7 +47,7 @@ graph TD ### Training & Evaluation - + ### API @@ -64,7 +64,7 @@ pred = model.predict({example}) # {Type} ### {Section Heading, Repeats} - + ## References diff --git a/skills/docs/references/templates/project-readme.md b/skills/docs/references/templates/project-readme.md index 2f64b1c..5d9b3e2 100644 --- a/skills/docs/references/templates/project-readme.md +++ b/skills/docs/references/templates/project-readme.md @@ -8,7 +8,7 @@ globs: ## Overview - + ## Quick Start diff --git a/skills/docs/references/templates/user-guide.md b/skills/docs/references/templates/user-guide.md index df4a128..da5734e 100644 --- a/skills/docs/references/templates/user-guide.md +++ b/skills/docs/references/templates/user-guide.md @@ -7,7 +7,7 @@ globs: ## Overview - + ## Install diff --git a/skills/docs/references/workflows/retcon.md b/skills/docs/references/workflows/retcon.md index d1314f3..02ff8c5 100644 --- a/skills/docs/references/workflows/retcon.md +++ b/skills/docs/references/workflows/retcon.md @@ -4,7 +4,7 @@ retcon reads the code and writes the full mandated documentation. It fills every - **Markdown output: soft-wrap prose, never hard-wrap** — when this workflow writes a `.md` artifact, write each paragraph as one continuous line; do not insert manual newlines to wrap prose at a fixed column width. Newlines still separate paragraphs, list items, headings, and code fences. -retcon scaffolds the same way `new` does. It runs `scripts/scaffold.py` to lay each skeleton. Then it reads each component's code and replaces every `` marker with real content. retcon reads code. It never parses templates itself. +retcon scaffolds the same way `new` does. It runs `scripts/scaffold.py` to lay each skeleton. Then it reads each component's code and fills the contract surface: tables, mermaid diagrams, runnable code samples, constrained `## Requirements` (per `references/requirements-guidance.md`), and `## References` — replacing each `` marker it consumes. It NEVER writes prose elsewhere: sections marked `` keep their marker untouched (see `references/prose-policy.md`). retcon reads code. It never parses templates itself. **Documentation grain.** retcon documents the system the way people think about it: grouped by kind (apps|services, libs, CI), by platform when the repo has more than one, then by module within apps|services. Correctness at the wrong grain is still a bad doc: a per-directory doc shaped like a service reads as a thin wrapper around a list of outputs and hides how the system works. retcon therefore groups discovered components (Step 2.5) before building the manifest (Step 3) and keeps the architecture index at the same grain. On a flat (non-monorepo) repo, retcon first proposes a lib/module → feature breakdown from `references/module-taxonomy.md` and asks the user to confirm it (Step 2.4) before grouping. @@ -232,9 +232,13 @@ Collect all agent results before authoring the top-level/index docs (5c), becaus This copies the resolved template and fills each `{key}` token from the `--vars` dict (`{name}`/`{path}`, so the frontmatter's `globs` already points at the doc's directory). retcon does not parse the template itself. 2. **Read the real code** so the doc is accurate: package metadata (`package.json`/`Cargo.toml`/etc.), entry points and exports (`index.ts`, public modules), route handlers, config files, env vars, and for infra the Terraform/module definitions. Author from what the code actually does — never invent identifiers, endpoints, or env vars. -### Step 5b: Replace each `@agent` marker with real content +### Step 5b: Fill the contract surface — artifacts only -Open the scaffolded doc and replace every `` marker with real content authored from the code, then delete the marker. The marker text is the authoring guidance for that section; the copied mermaid/table below it is the shape to fill. +Open the scaffolded doc. Per `references/prose-policy.md`, the LLM text budget is: artifact internals (tables, mermaid labels, code samples), `## Requirements`, `## References`, and nothing else. + +- `` marker introducing an artifact (table/diagram/code/requirements/references): replace it with the artifact authored from the code, then delete the marker. Keep artifact text minimal and barebones — identifiers, types, one-phrase labels. +- `` marker whose content would be prose: leave the marker in place; do not write prose. +- `` marker: preserve verbatim. Never fill, never delete. 1. **Frontmatter is already correct.** The `---` block is first and `globs:` already points at the doc's directory. Adjust the glob only if the doc owns a narrower/wider subtree than `{path}`. The doc carries no stored type marker — review re-derives the doc's type from its code path (its `globs`) using the type table in `references/structure.md`. 2. **Public API section** (the template's `[public-api]`-marked heading — see `references/template-contract.md`) must be explicit — the surface other modules reference. @@ -243,7 +247,7 @@ Open the scaffolded doc and replace every `` marker with rea 5. **Type-specific detail** from the source: request-lifecycle `sequenceDiagram` in `api` docs only; a data-model (`erDiagram`/type table) in `api`/`library`/`auth` docs; auth flow in `auth`; user flow in `frontend`; per the mermaid guide. 6. **Delete any `(optional)` section** whose content does not apply (e.g. no env vars → delete the Environment Variables section). Keep required sections. 7. **Carry forward legacy facts.** If a legacy doc for this doc's members exists, incorporate its still-correct details (commands, endpoints, env vars, terminology). Do not repeat facts the code contradicts. -8. Apply all language-guide rules to written prose (STE-terse). Leave a `` only for the rare thing that genuinely needs a human decision. +8. Apply all language-guide rules to the prose you are allowed to write (requirements bullets, references). Leave `` markers for every prose section; leave a `` only for the rare thing that genuinely needs a human decision inside an artifact. ### Step 5c: Author the top-level and index docs @@ -279,6 +283,14 @@ for path in sys.argv[1:]: After the fixup, run the coverage-overlap check from `references/coverage-and-api.md` (Step B) over the tree: skip docs carrying `index: true`; a non-nested overlap → warn and suggest narrowing one doc's `globs`; a strict-subset overlap → note the nested parent/child relationship; disjoint → no action. Surface these in the Step 7 summary; do not block the write. +Then run the artifact gate over every authored doc: + +```bash +python3 "$SKILL/scripts/validate_artifacts.py" {authored doc paths...} +``` + +Blocking findings (exit 1) are repair-before-write: fix the offending fence/table and re-run, at most 2 repair rounds. A finding that still fails becomes a `` comment beside the artifact and is surfaced in the Step 7 report; a doc never ships silently unvalidated. NOTE findings (no renderer available) are reported, not blocked. + ### Step 5e: Reconcile cross-references After every doc is written, verify the cross-links between docs. Each parallel agent sees only its own doc, so it cannot verify links to other docs. The reconciliation pass checks: each doc's Components section and the architecture index's Components section must NAME + LINK every sibling/child it delegates to, using that doc's **actual** `[public-api]` surface (the `public_api` summaries collected in Step 5). A link whose target surface does not exist, or that points at another doc's internals, is a bug — fix it in the doc. This pass runs on the final tree, so no agent authors a link it cannot verify. diff --git a/skills/docs/references/workflows/review.md b/skills/docs/references/workflows/review.md index 3e04e70..90b2079 100644 --- a/skills/docs/references/workflows/review.md +++ b/skills/docs/references/workflows/review.md @@ -24,6 +24,8 @@ Derive `SLUG` from `TARGET_PATH`: Rule: lowercase the path, strip the file extension, replace `/` and non-alphanumeric characters with `-`, collapse runs of `-`, trim leading/trailing `-`. ```bash +# $SKILL is this skill's package root (already exported by SKILL.md). Initialize the shared store first. +python3 "$SKILL/scripts/cv_init_store.py" >/dev/null REVIEW_DIR=".codevoyant/review/${SLUG}" mkdir -p "$REVIEW_DIR" ``` @@ -99,6 +101,8 @@ For each missing required section, record: Derive the required diagram set from the resolved template per `references/template-contract.md` (§2 Required diagrams): for each required heading that contains a ` ```mermaid ` fence in the template, the **diagram type** on the line after the fence is required for that section. Detect the doc's diagrams by grepping for ` ```mermaid ` blocks and reading the type on the next line; check that the doc has a matching-type fence somewhere under the corresponding heading. Optional sections never require a diagram. +For every mermaid fence present, also run the artifact gate: `python3 "$SKILL/scripts/validate_artifacts.py" {doc path}`. Blocking DIAGRAM findings (does not render, node/participant caps exceeded, literal `\n` labels) are recorded as DIAGRAM findings with the gate's message as `rationale`. NOTE findings (no renderer) are surfaced in the terminal report only. + For each missing diagram, record: - `type`: DIAGRAM - `current_text`: the prose that describes the flow (or "(no flow description found)") @@ -109,8 +113,10 @@ For each missing diagram, record: Apply the review check set in `references/language-guide.md` (## Review Checks) and the key STE rules from `references/simple-english/ruleset.md`. +Checks 11–17 (the Requirements Checks, R1–R7 from `references/requirements-guidance.md`) apply inside `## Requirements` sections only, and are recorded as `type: REQUIREMENTS` findings. Judge by intent, not blind substring matching — an identifier quoted as evidence is not an R1 violation; a requirement that names an endpoint as the requirement is. + For each violation, record: -- `type`: LANGUAGE +- `type`: LANGUAGE (checks 1–10) or REQUIREMENTS (checks 11–17) - `current_text`: the exact sentence or phrase containing the violation - `replacement_text`: the minimal rewrite that fixes only the violation - `rationale`: the specific rule number and name @@ -141,6 +147,16 @@ Applies to COMPONENT docs and the architecture index doc (skip `user-guide.md` / - **Sub-component docs named + linked in Components.** When this doc has child docs in the mandated structure (nested children under it — a `/` dir with `index.md`, or leaf children beside a nested `index.md`; for the architecture doc, every component doc is a child), each such child MUST be named and linked in the `[components]` section, referencing the child's public API section — NOT in `## References`. If a known child doc is not linked from Components (or is only linked from References), record a COVERAGE finding: `rationale`: "coverage-and-api Rule 3: parent names+links each child in Components, referencing the child's public API section." (Best-effort — when the child set cannot be determined, do not flag.) - **Inline system diagram is optional.** A `graph TD`/`flowchart TD` whose template marker starts with `(optional)` inside `### Components` is optional — do NOT flag its absence (its requiredness is decided by the template marker per `template-contract.md` §2). +### 3g. Prose-policy check + +Applies the allowance in `references/prose-policy.md`. For each section that is NOT `## Requirements` or `## References` and is NOT an artifact block (table, ` ```mermaid ` fence, code fence, HTML comment): narrative prose is a violation UNLESS the section's template marker is `@human` AND the prose is human-authored (preserved by update — treat existing prose in a `@human` section as human-written; do not flag it). Generator-authored prose outside the allowance is flagged: +- `type`: PROSE +- `current_text`: the offending prose block +- `replacement_text`: the `` marker from the resolved template for that section +- `rationale`: "prose-policy: LLM text is limited to comments, Requirements, References, and minimal artifact labels." + +A required heading whose template marker is `@human` is satisfied by the heading plus the marker — never flag an unfilled `@human` section as missing content. + ### 3e. Coverage & API-boundary check diff --git a/skills/docs/references/workflows/update.md b/skills/docs/references/workflows/update.md index b3fa97f..00b23f2 100644 --- a/skills/docs/references/workflows/update.md +++ b/skills/docs/references/workflows/update.md @@ -9,7 +9,7 @@ Update documentation files. Four modes, selected automatically (or by flag): 3. **Diff-scoped audit mode**: if no report exists, restrict scope to the branch diff and author the affected sections by reading the changed code (like retcon, but only what the diff touches). 4. **Escalation mode**: if the needed changes are too large, stop and run `docs review` first so you can inspect before applying. -**Preserve human text.** In all modes, change only text that is inaccurate or structurally incomplete. Do not rephrase working prose for style. This is the first-class principle of this workflow. +**Preserve human text.** In all modes, change only text that is inaccurate or structurally incomplete. Do not rephrase working prose for style. Never convert a `` marker into generated prose, and never delete one (see `references/prose-policy.md`). This is the first-class principle of this workflow. **Preserve coverage & API boundaries.** In all modes, follow `references/coverage-and-api.md`. Keep the doc's `globs:` frontmatter accurate; never add content that covers paths another doc owns (unless nested — then reference the child doc's interface only); when adding cross-references, use the target module's documented public API only; preserve the parent/child interface relationship (Step 4c). diff --git a/skills/docs/references/workflows/validate.md b/skills/docs/references/workflows/validate.md index 5e39a7b..f5c16f5 100644 --- a/skills/docs/references/workflows/validate.md +++ b/skills/docs/references/workflows/validate.md @@ -66,6 +66,16 @@ Group `COMPONENTS` into the architecture hierarchy first (same taxonomy as `retc If the repo has CI or infra config (Step 1) but no `docs/ci.md`, flag it (present-if-applicable, per `references/structure.md`). +### 3c. Artifact gate — every diagram and table validates + +Run the artifact gate over every managed doc: + +```bash +python3 "$SKILL/scripts/validate_artifacts.py" {managed doc paths...} +``` + +Blocking findings become `type: DIAGRAM` (fence/table issues) with the gate message; NOTEs are reported but do not fail validation. + ## Step 4: Check boundaries (reuse coverage-and-api) Run the detection procedure from `references/coverage-and-api.md` Steps A–C over the docs — the pairwise overlap check (Rules 1–3) and the API-boundary checks (Rules 3–5), where the API section is each doc's template's `[public-api]` heading. Under `--diff`, restrict the pairwise comparison to docs whose globs intersect the changed set (Steps D–F), flagging only boundaries affected by the change. @@ -86,6 +96,6 @@ docs/ci.md -- clean Summary: {N} findings across {M} files ({C} components discovered, {D} docs validated) ``` -With `--json`, emit the same structure as JSON. Optionally write a copy of the findings to `.codevoyant/review/{slug}/validate.md` in the same format as `docs-review-template.md` so they can be applied via `docs update --scaffold` (component gaps) or manually. +With `--json`, emit the same structure as JSON. Optionally write a copy of the findings to `.codevoyant/review/{slug}/validate.md` in the same format as `docs-review-template.md` so they can be applied via `docs update --scaffold` (component gaps) or manually. Initialize the shared store before that write: `python3 "$SKILL/scripts/cv_init_store.py" >/dev/null` (`$SKILL` is exported by SKILL.md). Exit 0 when clean; exit 1 when any finding is reported. diff --git a/skills/docs/scripts/cv_init_store.py b/skills/docs/scripts/cv_init_store.py new file mode 100644 index 0000000..00bb73c --- /dev/null +++ b/skills/docs/scripts/cv_init_store.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Canonical codevoyant store initializer — shared asset, vendored per skills/vendor.json. + +Ensures the in-repo `.codevoyant` is a symlink to the shared per-project store +(~/.codevoyant//) BEFORE any workflow mkdirs under it, so worktrees +and regular clones of the same project share one store. + +Idempotent. Never migrates an existing real dir (that is /migrate's job). +The slug pipeline is byte-identical to the /migrate skill's computation +(worktree-aware via `git rev-parse --git-common-dir`). + +Usage: cv_init_store.py [repo-root] (defaults to the git toplevel, else cwd) +Prints the computed slug on stdout. Exit 0 in all handled cases. +""" +import os +import re +import subprocess +import sys +from pathlib import Path + + +def _git(args, root): + try: + proc = subprocess.run( + ["git", "-C", str(root), *args], + capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return proc.stdout.strip() if proc.returncode == 0 else "" + + +def compute_slug(root): + """Same pipeline as the bash original: lower, [^a-z0-9]+ -> '-', strip '-'.""" + common = _git(["rev-parse", "--git-common-dir"], root) + if common: + cpath = Path(common) + if not cpath.is_absolute(): + cpath = root / cpath + try: + name = cpath.parent.resolve().name + except OSError: + name = root.name + else: + name = root.name + slug = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") + return slug or "unnamed" + + +def init_store(root, home=None): + """Create the ~/.codevoyant/ symlink for `root`; return the slug. + + - Already a symlink -> no-op. + - Existing real dir -> no-op (migration is /migrate's job). + - Otherwise -> mkdir dest, symlink, append .gitignore entry once. + """ + root = Path(root) + home = Path(home) if home else Path(os.environ["HOME"]) + slug = compute_slug(root) + link = root / ".codevoyant" + if link.is_symlink(): + return slug + if link.is_dir(): + return slug + dest = home / ".codevoyant" / slug + dest.mkdir(parents=True, exist_ok=True) + link.symlink_to(dest) + gi = root / ".gitignore" + lines = gi.read_text(encoding="utf-8").splitlines() if gi.exists() else [] + if ".codevoyant" not in lines: + with gi.open("a", encoding="utf-8") as f: + f.write("\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n") + return slug + + +def main(argv): + if len(argv) > 1: + root = Path(argv[1]) + else: + top = _git(["rev-parse", "--show-toplevel"], Path.cwd()) + root = Path(top) if top else Path.cwd() + print(init_store(root)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/docs/scripts/test_validate_artifacts.py b/skills/docs/scripts/test_validate_artifacts.py new file mode 100644 index 0000000..83e27b9 --- /dev/null +++ b/skills/docs/scripts/test_validate_artifacts.py @@ -0,0 +1,93 @@ +#!/usr/bin/env python3 +"""Unit tests for validate_artifacts.py (semantic stage; render stage needs mmdc).""" +import os +import pathlib +import sys +import tempfile +import unittest + +sys.path.insert(0, str(pathlib.Path(__file__).parent)) +import validate_artifacts as va + + +def findings_for(doc): + with tempfile.NamedTemporaryFile("w", suffix=".md", delete=False) as f: + f.write(doc) + path = f.name + try: + return va.validate_doc(path, skip_render=True) + finally: + os.unlink(path) + + +class SemanticChecks(unittest.TestCase): + def test_literal_backslash_n_flagged(self): + f = findings_for("```mermaid\ngraph TD\n A[Line one\\nLine two] --> B\n```\n") + self.assertTrue(any("literal \\n" in m for _p, t, m, b in f if t == "DIAGRAM")) + + def test_br_labels_pass(self): + f = findings_for('```mermaid\ngraph TD\n A["Line one
Line two"] --> B\n```\n') + self.assertFalse(any(t == "DIAGRAM" and b for _p, t, m, b in f)) + + def test_participant_cap(self): + parts = "\n".join(f" participant P{i}" for i in range(9)) + f = findings_for(f"```mermaid\nsequenceDiagram\n{parts}\n P0->>P1: hi\n```\n") + self.assertTrue(any("participants" in m for _p, t, m, b in f)) + + def test_node_cap(self): + nodes = "\n".join(f" N{i} --> N{i+1}" for i in range(13)) + f = findings_for(f"```mermaid\ngraph TD\n{nodes}\n```\n") + self.assertTrue(any("nodes" in m for _p, t, m, b in f)) + + def test_node_cap_counts_plain_link_edges(self): + nodes = "\n".join(f" N{i} --- N{i+1}" for i in range(13)) + f = findings_for(f"```mermaid\nflowchart TD\n{nodes}\n```\n") + self.assertTrue(any("nodes" in m and b for _p, t, m, b in f)) + + def test_node_cap_counts_dotted_and_thick_edges(self): + rows = [] + for i in range(13): + op = ("-.-", "==>", "<-->")[i % 3] + rows.append(f" N{i} {op} N{i+1}") + f = findings_for("```mermaid\nflowchart TD\n" + "\n".join(rows) + "\n```\n") + self.assertTrue(any("nodes" in m and b for _p, t, m, b in f)) + + def test_node_cap_counts_labeled_edges(self): + nodes = "\n".join(f" N{i} -->|step {i}| N{i+1}" for i in range(13)) + f = findings_for(f"```mermaid\nflowchart TD\n{nodes}\n```\n") + self.assertTrue(any("nodes" in m and b for _p, t, m, b in f)) + + def test_small_plain_link_diagram_passes(self): + f = findings_for("```mermaid\nflowchart TD\n A --- B --- C\n```\n") + self.assertFalse(any(t == "DIAGRAM" and "nodes" in m for _p, t, m, b in f)) + + def test_table_without_separator_flagged(self): + f = findings_for("| A | B |\n| 1 | 2 |\n") + self.assertTrue(any(t == "STRUCTURE" for _p, t, m, b in f)) + + def test_placeholder_only_table_flagged(self): + f = findings_for("| A | B |\n| --- | --- |\n| {x} | {y} |\n") + self.assertTrue(any("placeholders" in m for _p, t, m, b in f)) + + def test_table_inside_code_fence_ignored(self): + doc = "Example table syntax:\n\n```\n| a | b |\n| 1 | 2 |\n```\n" + f = findings_for(doc) + self.assertFalse(any(t == "STRUCTURE" for _p, t, m, b in f)) + + def test_table_inside_mermaid_fence_ignored(self): + doc = "```mermaid\ngraph TD\n A --> B\n```\n" + f = findings_for(doc) + self.assertFalse(any(t == "STRUCTURE" for _p, t, m, b in f)) + + def test_table_after_code_fence_still_flagged(self): + doc = "```\nsome code\n```\n\n| A | B |\n| 1 | 2 |\n" + f = findings_for(doc) + self.assertTrue(any(t == "STRUCTURE" for _p, t, m, b in f)) + + def test_clean_doc_passes(self): + f = findings_for("| A | B |\n| --- | --- |\n| 1 | 2 |\n") + self.assertFalse(any(b for _p, t, m, b in f)) + + +if __name__ == "__main__": + unittest.main() diff --git a/skills/docs/scripts/validate_artifacts.py b/skills/docs/scripts/validate_artifacts.py new file mode 100644 index 0000000..770fc8b --- /dev/null +++ b/skills/docs/scripts/validate_artifacts.py @@ -0,0 +1,187 @@ +#!/usr/bin/env python3 +"""Validate non-text artifacts in codevoyant docs (mermaid fences + tables). + +Two-stage gate: + 1. syntax — render every mermaid fence with mmdc (PATH first, else pinned npx). + 2. semantic — node/participant caps, literal-\\n labels, edge labels, table shape. + +Usage: + validate_artifacts.py [ ...] + validate_artifacts.py --skip-render # semantic stage only + +Exit 0 = clean (or only NOTEs); exit 1 = blocking findings. +Findings print as: : +JSON mode: --json emits [{"path","type","message","blocking"}]. +""" +import json +import re +import shutil +import subprocess +import sys +import tempfile +from pathlib import Path + +MMDC_PIN = "11.16.0" # single pinned renderer: "valid" must be deterministic +MAX_SEQUENCE_PARTICIPANTS = 8 +MAX_NODES = 12 + +FENCE_RE = re.compile(r"```mermaid\s*\n(.*?)```", re.DOTALL) +TABLE_ROW_RE = re.compile(r"^\s*\|.*\|\s*$") +CODE_FENCE_RE = re.compile(r"^\s*(```|~~~)") +# flowchart edge operators: plain (---, -->), dotted (-.-, -.->), thick +# (===, ==>), bidirectional (<-->, <==>, <-.->), special tips (--x, --o, +# x--x, o--o), and invisible links (~~~). Labels can sit between segments +# (A -->|text| B), so labels are stripped before splitting on these. +EDGE_OP_RE = re.compile(r"<==>|<-->|<-\.->|<-{2,3}|-\.+-?>?|={2,3}>?|-{2,3}[xo]?>?|[xo]-{2,3}[xo]|~{3}") +EDGE_LABEL_RE = re.compile(r"\|[^|\n]*\|") +# dash/dot label forms (A -- some text --> B, A -. some text .-> B). The +# label must be multi-word: single tokens stay, so node chains (A --- B --- C) +# are never mistaken for labeled edges. +EDGE_TEXT_LABEL_RE = re.compile(r"(-{2,3}|-\.+)\s+([A-Za-z0-9_]+(?:\s+[A-Za-z0-9_]+)+)\s+(-{2,3}>?|\.+-?>?)") +SHAPE_LABEL_RE = re.compile(r"\[[^\]\n]*\]|\([^\)\n]*\)|\{[^\}\n]*\}") +NODE_TOKEN_RE = re.compile(r"^[A-Za-z0-9_]+$") +GRAPH_KEYWORDS = { + "graph", "flowchart", "subgraph", "end", "direction", "style", + "class", "classdef", "click", "linkstyle", "default", +} +GRAPH_DIRECTIVE_RE = re.compile(r"^(graph|flowchart|subgraph|end|direction|style|classdef|class|click|linkstyle)\b", re.IGNORECASE) + + +def mermaid_fences(text): + return [(m.start(), m.group(1)) for m in FENCE_RE.finditer(text)] + + +def diagram_type(body): + for line in body.splitlines(): + line = line.strip() + if line: + return line.split()[0] + return "" + + +def node_ids(body, dtype): + ids = set() + if dtype == "sequenceDiagram": + for line in body.splitlines(): + m = re.match(r"\s*participant\s+(\S+)", line) + if m: + ids.add(m.group(1)) + return ids + for m in re.finditer(r"^\s*([A-Za-z0-9_]+)\s*(?:\[|\(|\{)", body, re.MULTILINE): + ids.add(m.group(1)) + for line in body.splitlines(): + line = line.strip() + if not line or GRAPH_DIRECTIVE_RE.match(line): + continue + line = EDGE_TEXT_LABEL_RE.sub(r"\1 \3", line) + line = EDGE_LABEL_RE.sub(" ", line) + line = SHAPE_LABEL_RE.sub(" ", line) + for token in EDGE_OP_RE.split(line): + token = token.strip().strip('"') + if token and NODE_TOKEN_RE.match(token) and token.lower() not in GRAPH_KEYWORDS: + ids.add(token) + return ids + + +def check_semantics(path, body, findings): + dtype = diagram_type(body) + if not dtype: + findings.append((str(path), "DIAGRAM", "empty mermaid fence", True)) + return + if re.search(r"\[[^\]\n]*\\n[^\]\n]*\]", body) or re.search(r'"[^"\n]*\\n[^"\n]*"', body): + findings.append((str(path), "DIAGRAM", f"literal \\n in a {dtype} node label — use
", True)) + if dtype == "sequenceDiagram": + n = len(node_ids(body, dtype)) + if n > MAX_SEQUENCE_PARTICIPANTS: + findings.append((str(path), "DIAGRAM", f"sequenceDiagram has {n} participants (cap {MAX_SEQUENCE_PARTICIPANTS}) — split it", True)) + elif dtype in ("graph", "flowchart"): + ids = node_ids(body, dtype) + if len(ids) > MAX_NODES: + findings.append((str(path), "DIAGRAM", f"{dtype} has {len(ids)} nodes (cap {MAX_NODES}) — split it", True)) + # unlabeled edges are allowed where the guide says direction implies meaning (dependency graphs) + # erDiagram: entity blocks present; PK/FK guidance is advisory, not gated + + +def check_tables(path, text, findings): + lines = text.splitlines() + in_fence = False + i = 0 + while i < len(lines): + if CODE_FENCE_RE.match(lines[i]): + in_fence = not in_fence + i += 1 + continue + if not in_fence and TABLE_ROW_RE.match(lines[i]): + block = [] + while i < len(lines) and TABLE_ROW_RE.match(lines[i]) and not CODE_FENCE_RE.match(lines[i]): + block.append(lines[i]) + i += 1 + if len(block) >= 2 and not re.match(r"^\s*\|[\s:|-]+\|\s*$", block[1]): + findings.append((str(path), "STRUCTURE", "table without a separator row (| --- |)", True)) + if len(block) >= 3 and all(re.fullmatch(r"\s*\|(\s*\{\w[^}]*\}\s*\|)+\s*", r) for r in block[2:]): + findings.append((str(path), "STRUCTURE", "table rows are all unfilled {placeholders}", True)) + else: + i += 1 + + +def mmdc_cmd(): + if shutil.which("mmdc"): + return ["mmdc"] + if shutil.which("npx"): + return ["npx", "-y", f"@mermaid-js/mermaid-cli@{MMDC_PIN}"] + return None + + +def check_render(path, body, cmd, findings): + with tempfile.TemporaryDirectory() as td: + src = Path(td) / "d.mmd" + out = Path(td) / "d.svg" + src.write_text(body) + try: + proc = subprocess.run(cmd + ["-i", str(src), "-o", str(out)], capture_output=True, text=True, timeout=120) + except (subprocess.TimeoutExpired, OSError) as e: + findings.append((str(path), "NOTE", f"mermaid render skipped ({e})", False)) + return + if proc.returncode != 0: + err = (proc.stderr or proc.stdout or "").strip().splitlines() + msg = err[0][:200] if err else "render failed" + findings.append((str(path), "DIAGRAM", f"mermaid does not render: {msg}", True)) + + +def validate_doc(path, skip_render=False): + text = Path(path).read_text(encoding="utf-8") + findings = [] + cmd = None if skip_render else mmdc_cmd() + if not skip_render and cmd is None: + findings.append((str(path), "NOTE", "no mmdc/npx on PATH — render gate skipped", False)) + for _start, body in mermaid_fences(text): + if cmd: + check_render(path, body, cmd, findings) + check_semantics(path, body, findings) + check_tables(path, text, findings) + return findings + + +def main(argv): + args = [a for a in argv[1:]] + as_json = "--json" in args + skip_render = "--skip-render" in args + paths = [a for a in args if not a.startswith("--")] + if not paths: + print(__doc__) + return 2 + all_findings = [] + for p in paths: + all_findings.extend(validate_doc(p, skip_render)) + if as_json: + print(json.dumps([{"path": p, "type": t, "message": m, "blocking": b} for p, t, m, b in all_findings], indent=2)) + else: + for p, t, m, b in all_findings: + print(f"{p}: {t}{' (blocking)' if b else ''} {m}") + if not all_findings: + print("clean") + return 1 if any(b for *_, b in all_findings) else 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/explore/references/workflows/diff.md b/skills/explore/references/workflows/diff.md index e96482b..e8a2cdd 100644 --- a/skills/explore/references/workflows/diff.md +++ b/skills/explore/references/workflows/diff.md @@ -124,6 +124,14 @@ Target repository: read the repository URL from the invocation text (the argumen **Check for changelog:** Look for `CHANGELOG.md`, `CHANGELOG`, or `RELEASES.md` in both repos. If found, scan recent entries to guide characterization of changes — they often name features and breaking changes explicitly. +Initialize the shared store before writing: + +```bash +# {SKILL_ROOT} = the explore skill's package root (substitute the real path) +python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null +mkdir -p .codevoyant/diffs +``` + Determine the output filename: `.codevoyant/diffs/{YYYY-MM-DD}-{target-repo-name}.md` Write the report using `references/report-template.md` as the structure. Keep each section to **5 bullets or fewer**. File trees should show `*` next to modified/added files. Only include sections that have meaningful content. diff --git a/skills/explore/references/workflows/new.md b/skills/explore/references/workflows/new.md index b8b5bc7..98c0e33 100644 --- a/skills/explore/references/workflows/new.md +++ b/skills/explore/references/workflows/new.md @@ -51,6 +51,10 @@ Store as `GENERATE_PROPOSALS`. Continue immediately — do not wait further. Create the exploration directory structure: ```bash +# {SKILL_ROOT} = the explore skill's package root (substitute the real path). +# Initialize the shared store first so a fresh clone gets the ~/.codevoyant/ +# symlink instead of a real .codevoyant dir. +python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null mkdir -p "$EXPLORE_DIR/research" "$EXPLORE_DIR/proposals" ``` diff --git a/skills/explore/scripts/cv_init_store.py b/skills/explore/scripts/cv_init_store.py new file mode 100644 index 0000000..00bb73c --- /dev/null +++ b/skills/explore/scripts/cv_init_store.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Canonical codevoyant store initializer — shared asset, vendored per skills/vendor.json. + +Ensures the in-repo `.codevoyant` is a symlink to the shared per-project store +(~/.codevoyant//) BEFORE any workflow mkdirs under it, so worktrees +and regular clones of the same project share one store. + +Idempotent. Never migrates an existing real dir (that is /migrate's job). +The slug pipeline is byte-identical to the /migrate skill's computation +(worktree-aware via `git rev-parse --git-common-dir`). + +Usage: cv_init_store.py [repo-root] (defaults to the git toplevel, else cwd) +Prints the computed slug on stdout. Exit 0 in all handled cases. +""" +import os +import re +import subprocess +import sys +from pathlib import Path + + +def _git(args, root): + try: + proc = subprocess.run( + ["git", "-C", str(root), *args], + capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return proc.stdout.strip() if proc.returncode == 0 else "" + + +def compute_slug(root): + """Same pipeline as the bash original: lower, [^a-z0-9]+ -> '-', strip '-'.""" + common = _git(["rev-parse", "--git-common-dir"], root) + if common: + cpath = Path(common) + if not cpath.is_absolute(): + cpath = root / cpath + try: + name = cpath.parent.resolve().name + except OSError: + name = root.name + else: + name = root.name + slug = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") + return slug or "unnamed" + + +def init_store(root, home=None): + """Create the ~/.codevoyant/ symlink for `root`; return the slug. + + - Already a symlink -> no-op. + - Existing real dir -> no-op (migration is /migrate's job). + - Otherwise -> mkdir dest, symlink, append .gitignore entry once. + """ + root = Path(root) + home = Path(home) if home else Path(os.environ["HOME"]) + slug = compute_slug(root) + link = root / ".codevoyant" + if link.is_symlink(): + return slug + if link.is_dir(): + return slug + dest = home / ".codevoyant" / slug + dest.mkdir(parents=True, exist_ok=True) + link.symlink_to(dest) + gi = root / ".gitignore" + lines = gi.read_text(encoding="utf-8").splitlines() if gi.exists() else [] + if ".codevoyant" not in lines: + with gi.open("a", encoding="utf-8") as f: + f.write("\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n") + return slug + + +def main(argv): + if len(argv) > 1: + root = Path(argv[1]) + else: + top = _git(["rev-parse", "--show-toplevel"], Path.cwd()) + root = Path(top) if top else Path.cwd() + print(init_store(root)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/flow/references/workflows/go.md b/skills/flow/references/workflows/go.md index 1005246..2a63060 100644 --- a/skills/flow/references/workflows/go.md +++ b/skills/flow/references/workflows/go.md @@ -28,35 +28,11 @@ Build the parameter map `PARAMS`: Resolve `FLOW_DIR` (the **definition** — read-only) per `references/flow-dir.md` (local-first, then global; `--global` forces global only). If not found in any scope, error: "Flow '{FLOW_NAME}' not found (looked in local and global). Run /flow new {FLOW_NAME} first." -Then resolve the **run instance** per `references/flow-dir.md` → *Run instance*. Run instances live **flat under `.codevoyant/flows/`** — beside the flow definitions — each named `{flow-slug}-{plan-slug}` (keyed by the run's resolved spec-plan slug). The slug does not exist yet at this point, so **bootstrap a provisional instance**: mint `RUN_ID="$(date -u +%Y%m%dT%H%M%SZ)"` and set `FLOW_STATE_ROOT=".codevoyant/flows"`. On a **fresh run**, `RUN_DIR="$FLOW_STATE_ROOT/{flow-slug}-_pending-$RUN_ID"` (the provisional is namespaced by `{flow-slug}` — the definition's directory name — so it never collides with another flow's provisional); on a **resume**, reattach `RUN_DIR` to this run's existing directory instead of minting a new one — the adopted `$FLOW_STATE_ROOT/{flow-slug}-{plan-slug}/` if present, else the newest non-`Complete` `$FLOW_STATE_ROOT/{flow-slug}-_pending-*/` whose `run.md` records `slug: {flow-slug}` (also recognize a legacy `.codevoyant/runs/{flow-slug}/progress.md` sitting directly under the legacy flow dir — reuse `.codevoyant/runs/{flow-slug}/` as `RUN_DIR` in that case). The run instance is **always local**, even when the definition is global. Because the run instance is the LOCAL `.codevoyant/flows/...` first-touch, initialize the shared store before creating it: `cv_init_store && mkdir -p "$RUN_DIR"`. - -`cv_init_store` ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`) **before** the `mkdir` — otherwise a fresh clone would create `.codevoyant` as a real directory instead of the shared symlink. It is idempotent, never migrates an existing real dir (that is `/migrate`'s job), and computes the identical `` as the `/migrate` skill. Define it in the same shell as the `mkdir`: - -```bash -cv_init_store() { - local root; root="$(git rev-parse --show-toplevel 2>/dev/null)" || root=""; [ -n "$root" ] || root="$PWD" - local link="$root/.codevoyant" - [ -L "$link" ] && return 0 # already a symlink → initialized - [ -d "$link" ] && return 0 # old real dir → leave it; /migrate copies it in, never here - local common name slug - common="$(git rev-parse --git-common-dir 2>/dev/null)" || common="" - if [ -n "$common" ]; then - case "$common" in /*) : ;; *) common="$root/$common" ;; esac - name="$(basename "$(cd "$(dirname "$common")" >/dev/null 2>&1 && pwd -P)")" - else - name="$(basename "$root")" - fi - slug="$(printf '%s' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | LC_ALL=C sed 's/[^a-z0-9][^a-z0-9]*/-/g; s/^-*//; s/-*$//')" - [ -n "$slug" ] || slug="unnamed" # empty-slug fallback — matches the /migrate skill - local dest="$HOME/.codevoyant/$slug" - mkdir -p "$dest"; ln -s "$dest" "$link" - local gi="$root/.gitignore" - { [ -f "$gi" ] && grep -qxF '.codevoyant' "$gi"; } || \ - printf '\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n' >> "$gi" -} -``` +Then resolve the **run instance** per `references/flow-dir.md` → *Run instance*. Run instances live **flat under `.codevoyant/flows/`** — beside the flow definitions — each named `{flow-slug}-{plan-slug}` (keyed by the run's resolved spec-plan slug). The slug does not exist yet at this point, so **bootstrap a provisional instance**: mint `RUN_ID="$(date -u +%Y%m%dT%H%M%SZ)"` and set `FLOW_STATE_ROOT=".codevoyant/flows"`. On a **fresh run**, `RUN_DIR="$FLOW_STATE_ROOT/{flow-slug}-_pending-$RUN_ID"` (the provisional is namespaced by `{flow-slug}` — the definition's directory name — so it never collides with another flow's provisional); on a **resume**, reattach `RUN_DIR` to this run's existing directory instead of minting a new one — the adopted `$FLOW_STATE_ROOT/{flow-slug}-{plan-slug}/` if present, else the newest non-`Complete` `$FLOW_STATE_ROOT/{flow-slug}-_pending-*/` whose `run.md` records `slug: {flow-slug}` (also recognize a legacy `.codevoyant/runs/{flow-slug}/progress.md` sitting directly under the legacy flow dir — reuse `.codevoyant/runs/{flow-slug}/` as `RUN_DIR` in that case). The run instance is **always local**, even when the definition is global. Because the run instance is the LOCAL `.codevoyant/flows/...` first-touch, initialize the shared store before creating it: `python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null && mkdir -p "$RUN_DIR"` (`{SKILL_ROOT}` = the flow skill's package root, substituted with the real path). + +The vendored store initializer (`scripts/cv_init_store.py`, shared asset) ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`) **before** the `mkdir` — otherwise a fresh clone would create `.codevoyant` as a real directory instead of the shared symlink. It is idempotent, never migrates an existing real dir (that is `/migrate`'s job), and computes the identical `` as the `/migrate` skill. -Note: only the LOCAL `.codevoyant/...` first-touch is wrapped. The `--global` flow paths write to `$HOME/.codevoyant/flows` directly and must NOT be wrapped with `cv_init_store`. +Note: only the LOCAL `.codevoyant/...` first-touch is wrapped. The `--global` flow paths write to `$HOME/.codevoyant/flows` directly and must NOT run the store initializer. **Pre-adoption resume is best-effort by recency.** Once a run has adopted, its `{flow-slug}-{plan-slug}/` is unambiguous. But if **two** runs of this flow were both interrupted *before* adoption, there are multiple non-`Complete` `{flow-slug}-_pending-*/` dirs and no plan slug yet to tell them apart — reattaching to the newest by mtime may pick the wrong one. Disambiguate cheaply: if the current invocation carries identifying params (`--set`/`input`, or an explicit `--branch`), prefer the provisional whose `run.md`/`context.md` **matches** those (e.g. same `branch:` or objective); otherwise fall back to newest-mtime and note it — `ℹ Multiple interrupted runs of '{FLOW_NAME}' found; resuming the most recent ({flow-slug}-_pending-{RUN_ID}). Pass matching --set/--branch to target a specific one, or /flow status to inspect.` Do not silently guess when it's ambiguous. diff --git a/skills/flow/references/workflows/new.md b/skills/flow/references/workflows/new.md index 0e5955d..1e5b084 100644 --- a/skills/flow/references/workflows/new.md +++ b/skills/flow/references/workflows/new.md @@ -39,32 +39,12 @@ If still no steps after prompting, error: "A flow must have at least one step." - `slug` = `FLOW_NAME` lowercased, spaces replaced with hyphens - **Naming-collision note (documented, not guarded).** Run instances live beside definitions in `.codevoyant/flows/`, named `{flow-slug}-{plan-slug}` (see `references/flow-dir.md` → *Run instance*). A flow definition literally named `{flow-slug}-{plan-slug}` could therefore *in theory* share a directory name with another flow's run instance — but definitions hold `flow.md` while instances hold `progress.md` + a `run.md` whose `slug:` is the flow's own slug, so discovery tells them apart by **content**, never by name (see the discovery filters in `status`/`doctor`). No reserved-name guard is needed here. - `FLOW_DIR = {FLOWS_DIR}/{slug}/` (global if `--global`, else local — from Step 0) -- If `--global`, first `mkdir -p "$HOME/.codevoyant/flows"`. (The `--global` path writes to `$HOME/.codevoyant/flows` directly — do **not** wrap it with `cv_init_store`.) -- If **local** (not `--global`), this is the LOCAL `.codevoyant/flows/...` first-touch, so initialize the shared store first: run `cv_init_store` (defined below) before creating `FLOW_DIR`. `cv_init_store` ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`) before the `mkdir` — otherwise a fresh clone would create `.codevoyant` as a real directory instead of the shared symlink. It is idempotent, never migrates an existing real dir (that is `/migrate`'s job), and computes the identical `` as the `/migrate` skill: +- If `--global`, first `mkdir -p "$HOME/.codevoyant/flows"`. (The `--global` path writes to `$HOME/.codevoyant/flows` directly — do **not** run the store initializer there.) +- If **local** (not `--global`), this is the LOCAL `.codevoyant/flows/...` first-touch, so initialize the shared store first: run the vendored initializer before creating `FLOW_DIR`. It ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`) before the `mkdir` — otherwise a fresh clone would create `.codevoyant` as a real directory instead of the shared symlink. It is idempotent, never migrates an existing real dir (that is `/migrate`'s job), and computes the identical `` as the `/migrate` skill (`{SKILL_ROOT}` = this skill's package root — substitute the real path): ```bash - cv_init_store() { - local root; root="$(git rev-parse --show-toplevel 2>/dev/null)" || root=""; [ -n "$root" ] || root="$PWD" - local link="$root/.codevoyant" - [ -L "$link" ] && return 0 # already a symlink → initialized - [ -d "$link" ] && return 0 # old real dir → leave it; /migrate copies it in, never here - local common name slug - common="$(git rev-parse --git-common-dir 2>/dev/null)" || common="" - if [ -n "$common" ]; then - case "$common" in /*) : ;; *) common="$root/$common" ;; esac - name="$(basename "$(cd "$(dirname "$common")" >/dev/null 2>&1 && pwd -P)")" - else - name="$(basename "$root")" - fi - slug="$(printf '%s' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | LC_ALL=C sed 's/[^a-z0-9][^a-z0-9]*/-/g; s/^-*//; s/-*$//')" - [ -n "$slug" ] || slug="unnamed" # empty-slug fallback — matches the /migrate skill - local dest="$HOME/.codevoyant/$slug" - mkdir -p "$dest"; ln -s "$dest" "$link" - local gi="$root/.gitignore" - { [ -f "$gi" ] && grep -qxF '.codevoyant' "$gi"; } || \ - printf '\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n' >> "$gi" - } - # local scope only: - cv_init_store + # {SKILL_ROOT} = the flow skill's package root (substitute the real path). + # Local scope only (the --global path writes to $HOME/.codevoyant/flows directly and never runs the initializer): + python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null ``` - If `FLOW_DIR/flow.md` already exists: ask "Flow '{slug}' already exists in {scope}. Replace it or cancel? (replace/cancel)" - If cancel: exit without changes. diff --git a/skills/flow/scripts/cv_init_store.py b/skills/flow/scripts/cv_init_store.py new file mode 100644 index 0000000..00bb73c --- /dev/null +++ b/skills/flow/scripts/cv_init_store.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Canonical codevoyant store initializer — shared asset, vendored per skills/vendor.json. + +Ensures the in-repo `.codevoyant` is a symlink to the shared per-project store +(~/.codevoyant//) BEFORE any workflow mkdirs under it, so worktrees +and regular clones of the same project share one store. + +Idempotent. Never migrates an existing real dir (that is /migrate's job). +The slug pipeline is byte-identical to the /migrate skill's computation +(worktree-aware via `git rev-parse --git-common-dir`). + +Usage: cv_init_store.py [repo-root] (defaults to the git toplevel, else cwd) +Prints the computed slug on stdout. Exit 0 in all handled cases. +""" +import os +import re +import subprocess +import sys +from pathlib import Path + + +def _git(args, root): + try: + proc = subprocess.run( + ["git", "-C", str(root), *args], + capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return proc.stdout.strip() if proc.returncode == 0 else "" + + +def compute_slug(root): + """Same pipeline as the bash original: lower, [^a-z0-9]+ -> '-', strip '-'.""" + common = _git(["rev-parse", "--git-common-dir"], root) + if common: + cpath = Path(common) + if not cpath.is_absolute(): + cpath = root / cpath + try: + name = cpath.parent.resolve().name + except OSError: + name = root.name + else: + name = root.name + slug = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") + return slug or "unnamed" + + +def init_store(root, home=None): + """Create the ~/.codevoyant/ symlink for `root`; return the slug. + + - Already a symlink -> no-op. + - Existing real dir -> no-op (migration is /migrate's job). + - Otherwise -> mkdir dest, symlink, append .gitignore entry once. + """ + root = Path(root) + home = Path(home) if home else Path(os.environ["HOME"]) + slug = compute_slug(root) + link = root / ".codevoyant" + if link.is_symlink(): + return slug + if link.is_dir(): + return slug + dest = home / ".codevoyant" / slug + dest.mkdir(parents=True, exist_ok=True) + link.symlink_to(dest) + gi = root / ".gitignore" + lines = gi.read_text(encoding="utf-8").splitlines() if gi.exists() else [] + if ".codevoyant" not in lines: + with gi.open("a", encoding="utf-8") as f: + f.write("\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n") + return slug + + +def main(argv): + if len(argv) > 1: + root = Path(argv[1]) + else: + top = _git(["rev-parse", "--show-toplevel"], Path.cwd()) + root = Path(top) if top else Path.cwd() + print(init_store(root)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/loop/SKILL.md b/skills/loop/SKILL.md new file mode 100644 index 0000000..0a4c2e3 --- /dev/null +++ b/skills/loop/SKILL.md @@ -0,0 +1,72 @@ +--- +name: loop +description: "Repeat a task until its objective is met or a max iteration count is reached. Creates a tracking doc and runs immediately — every iteration executes in one background loop agent that performs the task and judges the objective. Triggers on: 'loop', 'run loop', 'repeat until', 'keep going until'." +license: MIT +compatibility: Works on Claude Code and OpenCode. Uses background agents. +--- + +# loop + +A loop is not a saved artifact like a flow. It is a **tracking doc plus a run**: `/loop` writes `.codevoyant/loops/{slug}/loop.md` (task, objective, bound, and a per-iteration log) and immediately executes iterations, appending each result to the doc, until the objective is met or the bound is reached. There is nothing to define ahead of time and no separate run command. + +**Markdown output: soft-wrap prose, never hard-wrap** — when this skill writes a `.md` artifact (the tracking doc or any generated document), write each paragraph as one continuous line; do not insert manual newlines to wrap prose at a fixed column width. Newlines still separate paragraphs, list items, headings, and code fences. + +## Usage + +``` +/loop --until [--max N] [--check ] [--resume ] +``` + +- **task** (required positional) — what to repeat each iteration: a skill command, shell command, or agent instruction. Runs in a background agent. +- **--until** (required) — the objective: the verifiable condition that ends the loop, phrased as an outcome, not an activity. +- **--max N** (default 3, must be ≥ 1) — the hard upper bound. The loop stops after N iterations even if the objective is not met. +- **--check ** (optional) — a deterministic check that exits 0 when the objective is met. When present it is the authoritative signal and overrides the agent's verdict. +- **--resume ** (optional) — continue an existing tracking doc (append iterations under its existing task/objective/max) instead of starting a new one. + +## Procedure + +All of it is here — there are no workflow files. + +1. **Parse args.** Error out if the task or `--until` is missing, or `--max` is not a positive integer. +2. **Initialize the shared store** before any mkdir (a fresh clone must get the symlink, not a real dir): + ```bash + # {SKILL_ROOT} = this skill's package root (the directory containing this SKILL.md) — substitute the real path. + python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null + ``` +3. **Slug the task** (lowercase, spaces → hyphens, `[a-z0-9-]`, ≤ 50 chars); suffix `-2`, `-3`, … if `.codevoyant/loops/{slug}/` already exists, unless `--resume` names it. `mkdir -p .codevoyant/loops/{slug}`. +4. **Write the tracking doc** `.codevoyant/loops/{slug}/loop.md` (or reuse it under `--resume`): + ```markdown + # Loop: {slug} + + - **Task:** {task} + - **Objective:** {objective} + - **Check:** {command | (none — the loop agent judges)} + - **Max iterations:** {N} + - **Status:** running + + | # | Result | Verdict | Reason | + | --- | --- | --- | --- | + {one row appended per iteration} + ``` +5. **Run iterations** for `i` in `1..N`: + - Spawn ONE `loop-agent` background agent (`agents/loop-agent.md`, `run_in_background: true`) with the task, the objective, the iteration number, and the previous iteration's result line. It performs the task AND judges whether the objective is now met — strictly, from the actual repo state, never from its own claim. Collect with a blocking wait. + - If it returns `NEEDS_INPUT: {question}`: ask the user on the main thread, fold the answer in, and re-run the same iteration (it still counts once toward the bound). + - If `--check` is set, run the check command after the agent returns: exit 0 → MET (the check overrides the agent's verdict); non-zero → NOT_MET. + - Append the iteration row to the tracking doc (result summary, verdict, reason). On MET: set `Status: complete` and stop. Otherwise continue. +6. **Report.** If the loop exits at the bound without MET, set `Status: max-reached`. Print: + ``` + ✓ Loop '{slug}' {complete | max-reached} — {i}/{N} iterations + Final: {last verdict reason} + Tracking doc: .codevoyant/loops/{slug}/loop.md + ``` + On `max-reached`, add: re-run with a higher `--max`, tighten the task, or `--resume {slug}` to continue the same loop later. + +## Guarantees + +- The task never runs inline on the main thread — every iteration is a `loop-agent` background run. +- The loop always terminates: at the bound at the latest. +- A `NEEDS_INPUT` re-run does not consume extra iterations beyond the bound. + +## Agent + +- **loop-agent** (`agents/loop-agent.md`) — performs one iteration of the task and judges the objective from the actual repo state; returns STATUS/RESULT/EVIDENCE/VERDICT. diff --git a/skills/loop/agents/loop-agent.md b/skills/loop/agents/loop-agent.md new file mode 100644 index 0000000..0dc5e2e --- /dev/null +++ b/skills/loop/agents/loop-agent.md @@ -0,0 +1,48 @@ +--- +name: loop-agent +description: Executes one iteration of a loop — performs the task and judges whether the loop's objective is met from the actual repo state. Spawned as a background agent by the loop skill; one agent per iteration, no separate judge. +tools: Read, Grep, Glob, Bash, Edit, Write +metadata: + model-tier: standard +--- + +Your job is one iteration of a repeating loop, in two halves: **do**, then **judge**. You perform the loop's Task once, then you decide — strictly, from the actual repo state and evidence — whether the loop's Objective is now met. You never decide whether to continue; the caller owns the bound. + +## Inputs + +You receive: the Task, the Objective, the iteration number, and the previous iteration's result line (empty on iteration 1). + +## Half one — do + +1. If the Task is a skill command, invoke that skill's workflow. If it is a shell command, run it. If it is an agent instruction, carry it out directly. +2. Make the smallest change the task requires — no drive-by edits. +3. Capture what changed and any output or error that matters for judging the objective. + +## Half two — judge + +1. Read the Objective as a verifiable condition. +2. Check the actual state — run read-only commands, read the relevant files, inspect the evidence you produced. Do NOT take your own "done" claim at face value. +3. Decide MET or NOT_MET. "Met" means the objective is verifiably true, not "probably" or "mostly". If you cannot verify either way, return NOT_MET and say what is missing. + +## Output + +Return exactly this shape: + +``` +ITERATION: {n} +STATUS: [OK | FAILED | NEEDS_INPUT] +RESULT: {one short paragraph — what was done and the observable result} +EVIDENCE: {concrete output the verdict rests on — test results, command output, file state, or "(none)"} +VERDICT: [MET | NOT_MET] +REASON: {one short sentence — the evidence for the verdict} +MISSING: {what is still required if NOT_MET, else "(none)"} +``` + +- `STATUS: NEEDS_INPUT` — the task genuinely needs an answer it cannot derive; state the question in RESULT. The caller asks the user and re-runs this iteration. +- `STATUS: FAILED` — the task errored; the error goes in RESULT and EVIDENCE. A failed iteration still counts toward the bound; judge the objective on whatever state actually resulted. + +Keep RESULT and EVIDENCE terse — the caller reads them every iteration. + +## Markdown output + +**Markdown output: soft-wrap prose, never hard-wrap** — when you emit markdown, write each paragraph as one continuous line. Do not insert manual newlines to wrap prose at a fixed column width; let the renderer wrap. Newlines still separate paragraphs, list items, headings, and code fences. diff --git a/skills/loop/scripts/cv_init_store.py b/skills/loop/scripts/cv_init_store.py new file mode 100644 index 0000000..00bb73c --- /dev/null +++ b/skills/loop/scripts/cv_init_store.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Canonical codevoyant store initializer — shared asset, vendored per skills/vendor.json. + +Ensures the in-repo `.codevoyant` is a symlink to the shared per-project store +(~/.codevoyant//) BEFORE any workflow mkdirs under it, so worktrees +and regular clones of the same project share one store. + +Idempotent. Never migrates an existing real dir (that is /migrate's job). +The slug pipeline is byte-identical to the /migrate skill's computation +(worktree-aware via `git rev-parse --git-common-dir`). + +Usage: cv_init_store.py [repo-root] (defaults to the git toplevel, else cwd) +Prints the computed slug on stdout. Exit 0 in all handled cases. +""" +import os +import re +import subprocess +import sys +from pathlib import Path + + +def _git(args, root): + try: + proc = subprocess.run( + ["git", "-C", str(root), *args], + capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return proc.stdout.strip() if proc.returncode == 0 else "" + + +def compute_slug(root): + """Same pipeline as the bash original: lower, [^a-z0-9]+ -> '-', strip '-'.""" + common = _git(["rev-parse", "--git-common-dir"], root) + if common: + cpath = Path(common) + if not cpath.is_absolute(): + cpath = root / cpath + try: + name = cpath.parent.resolve().name + except OSError: + name = root.name + else: + name = root.name + slug = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") + return slug or "unnamed" + + +def init_store(root, home=None): + """Create the ~/.codevoyant/ symlink for `root`; return the slug. + + - Already a symlink -> no-op. + - Existing real dir -> no-op (migration is /migrate's job). + - Otherwise -> mkdir dest, symlink, append .gitignore entry once. + """ + root = Path(root) + home = Path(home) if home else Path(os.environ["HOME"]) + slug = compute_slug(root) + link = root / ".codevoyant" + if link.is_symlink(): + return slug + if link.is_dir(): + return slug + dest = home / ".codevoyant" / slug + dest.mkdir(parents=True, exist_ok=True) + link.symlink_to(dest) + gi = root / ".gitignore" + lines = gi.read_text(encoding="utf-8").splitlines() if gi.exists() else [] + if ".codevoyant" not in lines: + with gi.open("a", encoding="utf-8") as f: + f.write("\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n") + return slug + + +def main(argv): + if len(argv) > 1: + root = Path(argv[1]) + else: + top = _git(["rev-parse", "--show-toplevel"], Path.cwd()) + root = Path(top) if top else Path.cwd() + print(init_store(root)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/migrate/SKILL.md b/skills/migrate/SKILL.md index a4556b9..1acf5eb 100644 --- a/skills/migrate/SKILL.md +++ b/skills/migrate/SKILL.md @@ -57,7 +57,7 @@ cv_init_store() { name="$(basename "$root")" fi slug="$(printf '%s' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | LC_ALL=C sed 's/[^a-z0-9][^a-z0-9]*/-/g; s/^-*//; s/-*$//')" - [ -n "$slug" ] || slug="unnamed" # empty-slug fallback — matches cv_init_store + [ -n "$slug" ] || slug="unnamed" # empty-slug fallback — matches skills/shared/store-init/cv_init_store.py local dest="$HOME/.codevoyant/$slug" mkdir -p "$dest" export CV_STORE="$dest" diff --git a/skills/plan/references/workflows/plan.md b/skills/plan/references/workflows/plan.md index 7625345..7ad339d 100644 --- a/skills/plan/references/workflows/plan.md +++ b/skills/plan/references/workflows/plan.md @@ -244,32 +244,11 @@ After running the checkpoint, write the Quality Brief (3–5 bullets) and use it ## Step 5: Build Milestone Task Plan -Create plan directory. `cv_init_store` ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`) **before** the first `mkdir` — otherwise a fresh clone would create `.codevoyant` as a real directory instead of the shared symlink. It is idempotent, never migrates an existing real dir (that is `/migrate`'s job), and computes the identical `` as the `/migrate` skill. +Create plan directory. Before the first `mkdir`, run the vendored store initializer — it ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`), so worktrees and regular clones share one store. Idempotent; never migrates an existing real dir (that is `/migrate`'s job). Substitute `{SKILL_ROOT}` with this skill's package root (the directory containing the plan skill's SKILL.md, as reported when the skill loaded). ```bash -cv_init_store() { - local root; root="$(git rev-parse --show-toplevel 2>/dev/null)" || root=""; [ -n "$root" ] || root="$PWD" - local link="$root/.codevoyant" - [ -L "$link" ] && return 0 # already a symlink → initialized - [ -d "$link" ] && return 0 # old real dir → leave it; /migrate copies it in, never here - local common name slug - common="$(git rev-parse --git-common-dir 2>/dev/null)" || common="" - if [ -n "$common" ]; then - case "$common" in /*) : ;; *) common="$root/$common" ;; esac - name="$(basename "$(cd "$(dirname "$common")" >/dev/null 2>&1 && pwd -P)")" - else - name="$(basename "$root")" - fi - slug="$(printf '%s' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | LC_ALL=C sed 's/[^a-z0-9][^a-z0-9]*/-/g; s/^-*//; s/-*$//')" - [ -n "$slug" ] || slug="unnamed" # empty-slug fallback — matches the /migrate skill - local dest="$HOME/.codevoyant/$slug" - mkdir -p "$dest"; ln -s "$dest" "$link" - local gi="$root/.gitignore" - { [ -f "$gi" ] && grep -qxF '.codevoyant' "$gi"; } || \ - printf '\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n' >> "$gi" -} - -cv_init_store +# {SKILL_ROOT} = the plan skill's package root (substitute the real path) +python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null mkdir -p .codevoyant/plan/{slug}/tasks mkdir -p .codevoyant/explore/{slug} ``` diff --git a/skills/plan/scripts/cv_init_store.py b/skills/plan/scripts/cv_init_store.py new file mode 100644 index 0000000..00bb73c --- /dev/null +++ b/skills/plan/scripts/cv_init_store.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Canonical codevoyant store initializer — shared asset, vendored per skills/vendor.json. + +Ensures the in-repo `.codevoyant` is a symlink to the shared per-project store +(~/.codevoyant//) BEFORE any workflow mkdirs under it, so worktrees +and regular clones of the same project share one store. + +Idempotent. Never migrates an existing real dir (that is /migrate's job). +The slug pipeline is byte-identical to the /migrate skill's computation +(worktree-aware via `git rev-parse --git-common-dir`). + +Usage: cv_init_store.py [repo-root] (defaults to the git toplevel, else cwd) +Prints the computed slug on stdout. Exit 0 in all handled cases. +""" +import os +import re +import subprocess +import sys +from pathlib import Path + + +def _git(args, root): + try: + proc = subprocess.run( + ["git", "-C", str(root), *args], + capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return proc.stdout.strip() if proc.returncode == 0 else "" + + +def compute_slug(root): + """Same pipeline as the bash original: lower, [^a-z0-9]+ -> '-', strip '-'.""" + common = _git(["rev-parse", "--git-common-dir"], root) + if common: + cpath = Path(common) + if not cpath.is_absolute(): + cpath = root / cpath + try: + name = cpath.parent.resolve().name + except OSError: + name = root.name + else: + name = root.name + slug = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") + return slug or "unnamed" + + +def init_store(root, home=None): + """Create the ~/.codevoyant/ symlink for `root`; return the slug. + + - Already a symlink -> no-op. + - Existing real dir -> no-op (migration is /migrate's job). + - Otherwise -> mkdir dest, symlink, append .gitignore entry once. + """ + root = Path(root) + home = Path(home) if home else Path(os.environ["HOME"]) + slug = compute_slug(root) + link = root / ".codevoyant" + if link.is_symlink(): + return slug + if link.is_dir(): + return slug + dest = home / ".codevoyant" / slug + dest.mkdir(parents=True, exist_ok=True) + link.symlink_to(dest) + gi = root / ".gitignore" + lines = gi.read_text(encoding="utf-8").splitlines() if gi.exists() else [] + if ".codevoyant" not in lines: + with gi.open("a", encoding="utf-8") as f: + f.write("\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n") + return slug + + +def main(argv): + if len(argv) > 1: + root = Path(argv[1]) + else: + top = _git(["rev-parse", "--show-toplevel"], Path.cwd()) + root = Path(top) if top else Path.cwd() + print(init_store(root)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/pm/references/workflows/explore.md b/skills/pm/references/workflows/explore.md index c123064..ee60c77 100644 --- a/skills/pm/references/workflows/explore.md +++ b/skills/pm/references/workflows/explore.md @@ -46,32 +46,11 @@ Do **not** ask the user to confirm mode selection. State the inferred modes brie ## Step 2: Launch parallel research agents -`cv_init_store` ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`) **before** the first `mkdir` — otherwise a fresh clone would create `.codevoyant` as a real directory instead of the shared symlink. It is idempotent, never migrates an existing real dir (that is `/migrate`'s job), and computes the identical `` as the `/migrate` skill. +Before the first `mkdir`, run the vendored store initializer — it ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`), so worktrees and regular clones share one store. Idempotent; never migrates an existing real dir (that is `/migrate`'s job). Substitute `{SKILL_ROOT}` with this skill's package root (the directory containing the pm skill's SKILL.md, as reported when the skill loaded). ```bash -cv_init_store() { - local root; root="$(git rev-parse --show-toplevel 2>/dev/null)" || root=""; [ -n "$root" ] || root="$PWD" - local link="$root/.codevoyant" - [ -L "$link" ] && return 0 # already a symlink → initialized - [ -d "$link" ] && return 0 # old real dir → leave it; /migrate copies it in, never here - local common name slug - common="$(git rev-parse --git-common-dir 2>/dev/null)" || common="" - if [ -n "$common" ]; then - case "$common" in /*) : ;; *) common="$root/$common" ;; esac - name="$(basename "$(cd "$(dirname "$common")" >/dev/null 2>&1 && pwd -P)")" - else - name="$(basename "$root")" - fi - slug="$(printf '%s' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | LC_ALL=C sed 's/[^a-z0-9][^a-z0-9]*/-/g; s/^-*//; s/-*$//')" - [ -n "$slug" ] || slug="unnamed" # empty-slug fallback — matches the /migrate skill - local dest="$HOME/.codevoyant/$slug" - mkdir -p "$dest"; ln -s "$dest" "$link" - local gi="$root/.gitignore" - { [ -f "$gi" ] && grep -qxF '.codevoyant' "$gi"; } || \ - printf '\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n' >> "$gi" -} - -cv_init_store +# {SKILL_ROOT} = the pm skill's package root (substitute the real path) +python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null mkdir -p ".codevoyant/explore/{SLUG}/research" ``` diff --git a/skills/pm/references/workflows/prd.md b/skills/pm/references/workflows/prd.md index c69a2d3..6b6ed86 100644 --- a/skills/pm/references/workflows/prd.md +++ b/skills/pm/references/workflows/prd.md @@ -122,32 +122,11 @@ Set: - OUTPUT_PATH = `.codevoyant/prds/{SLUG}/{SLUG}.md` -`cv_init_store` ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`) **before** the first `mkdir` — otherwise a fresh clone would create `.codevoyant` as a real directory instead of the shared symlink. It is idempotent, never migrates an existing real dir (that is `/migrate`'s job), and computes the identical `` as the `/migrate` skill. +Before the first `mkdir`, run the vendored store initializer — it ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`), so worktrees and regular clones share one store. Idempotent; never migrates an existing real dir (that is `/migrate`'s job). Substitute `{SKILL_ROOT}` with this skill's package root (the directory containing the pm skill's SKILL.md, as reported when the skill loaded). ```bash -cv_init_store() { - local root; root="$(git rev-parse --show-toplevel 2>/dev/null)" || root=""; [ -n "$root" ] || root="$PWD" - local link="$root/.codevoyant" - [ -L "$link" ] && return 0 # already a symlink → initialized - [ -d "$link" ] && return 0 # old real dir → leave it; /migrate copies it in, never here - local common name slug - common="$(git rev-parse --git-common-dir 2>/dev/null)" || common="" - if [ -n "$common" ]; then - case "$common" in /*) : ;; *) common="$root/$common" ;; esac - name="$(basename "$(cd "$(dirname "$common")" >/dev/null 2>&1 && pwd -P)")" - else - name="$(basename "$root")" - fi - slug="$(printf '%s' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | LC_ALL=C sed 's/[^a-z0-9][^a-z0-9]*/-/g; s/^-*//; s/-*$//')" - [ -n "$slug" ] || slug="unnamed" # empty-slug fallback — matches the /migrate skill - local dest="$HOME/.codevoyant/$slug" - mkdir -p "$dest"; ln -s "$dest" "$link" - local gi="$root/.gitignore" - { [ -f "$gi" ] && grep -qxF '.codevoyant' "$gi"; } || \ - printf '\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n' >> "$gi" -} - -cv_init_store +# {SKILL_ROOT} = the pm skill's package root (substitute the real path) +python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null mkdir -p ".codevoyant/prds/{SLUG}" ``` diff --git a/skills/pm/scripts/cv_init_store.py b/skills/pm/scripts/cv_init_store.py new file mode 100644 index 0000000..00bb73c --- /dev/null +++ b/skills/pm/scripts/cv_init_store.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Canonical codevoyant store initializer — shared asset, vendored per skills/vendor.json. + +Ensures the in-repo `.codevoyant` is a symlink to the shared per-project store +(~/.codevoyant//) BEFORE any workflow mkdirs under it, so worktrees +and regular clones of the same project share one store. + +Idempotent. Never migrates an existing real dir (that is /migrate's job). +The slug pipeline is byte-identical to the /migrate skill's computation +(worktree-aware via `git rev-parse --git-common-dir`). + +Usage: cv_init_store.py [repo-root] (defaults to the git toplevel, else cwd) +Prints the computed slug on stdout. Exit 0 in all handled cases. +""" +import os +import re +import subprocess +import sys +from pathlib import Path + + +def _git(args, root): + try: + proc = subprocess.run( + ["git", "-C", str(root), *args], + capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return proc.stdout.strip() if proc.returncode == 0 else "" + + +def compute_slug(root): + """Same pipeline as the bash original: lower, [^a-z0-9]+ -> '-', strip '-'.""" + common = _git(["rev-parse", "--git-common-dir"], root) + if common: + cpath = Path(common) + if not cpath.is_absolute(): + cpath = root / cpath + try: + name = cpath.parent.resolve().name + except OSError: + name = root.name + else: + name = root.name + slug = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") + return slug or "unnamed" + + +def init_store(root, home=None): + """Create the ~/.codevoyant/ symlink for `root`; return the slug. + + - Already a symlink -> no-op. + - Existing real dir -> no-op (migration is /migrate's job). + - Otherwise -> mkdir dest, symlink, append .gitignore entry once. + """ + root = Path(root) + home = Path(home) if home else Path(os.environ["HOME"]) + slug = compute_slug(root) + link = root / ".codevoyant" + if link.is_symlink(): + return slug + if link.is_dir(): + return slug + dest = home / ".codevoyant" / slug + dest.mkdir(parents=True, exist_ok=True) + link.symlink_to(dest) + gi = root / ".gitignore" + lines = gi.read_text(encoding="utf-8").splitlines() if gi.exists() else [] + if ".codevoyant" not in lines: + with gi.open("a", encoding="utf-8") as f: + f.write("\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n") + return slug + + +def main(argv): + if len(argv) > 1: + root = Path(argv[1]) + else: + top = _git(["rev-parse", "--show-toplevel"], Path.cwd()) + root = Path(top) if top else Path.cwd() + print(init_store(root)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/pr/SKILL.md b/skills/pr/SKILL.md index 049b53d..7a60cd9 100644 --- a/skills/pr/SKILL.md +++ b/skills/pr/SKILL.md @@ -82,5 +82,7 @@ If `references/workflows/{VERB}.md` does not exist, fall back to `references/wor - **slop-detector** (`agents/slop-detector.md`) — Dimension 2: unnecessary/out-of-scope edits, stochastic churn, boilerplate, dead/debug leftovers, accidental reverts - **code-quality-auditor** (`agents/code-quality-auditor.md`) — Dimension 3: judges added/edited code against the relevant codevoyant skill or the language/framework standard - **docs-freshness-checker** (`agents/docs-freshness-checker.md`) — Dimension 4: decides whether docs need updating and invokes `/docs update` when they are stale +- **red-team-adversary** (`agents/red-team-adversary.md`) — Dimension 5: adversarial hunt — failure modes, edge cases, mutation-mindset test review, STRIDE on security surfaces; BLOCKING requires a concrete failing scenario +- **claim-checker** (`agents/claim-checker.md`) — verifies the PR/MR body's claims against the diff; unfulfilled claims are BLOCKING -(Dimension 1, intent-match & correctness, runs as an inline reviewer agent defined in `references/workflows/review.md`.) +(Dimension 1, intent-match & correctness, runs as an inline reviewer agent defined in `references/workflows/review.md`. Reviews post file-level comments only; the top-level comment is reserved for un-anchorable structural issues — see `references/new-review-template.md`.) diff --git a/skills/pr/agents/claim-checker.md b/skills/pr/agents/claim-checker.md new file mode 100644 index 0000000..90d025c --- /dev/null +++ b/skills/pr/agents/claim-checker.md @@ -0,0 +1,44 @@ +--- +name: claim-checker +description: Verifies the PR/MR body's claims (Changes bullets, Validation checklist, stated behavior) against the actual diff for /pr review. An unfulfilled claim is reported with the bullet and what the diff actually does. Used by /pr review as a mechanical claim gate. +tools: Read, Grep, Glob, Bash +metadata: + model-tier: light +--- + +Your entire job is one question: **does the diff do what the PR/MR body says it does?** The body's `Changes` bullets, `Validation` checklist, and any stated behavior are claims; the diff is the evidence. You prove or disprove each claim. You do not review code quality, intent framing, or slop — other passes own those. + +## How to work + +1. Parse the body into individual claims: each `Changes` bullet, each `Validation` checkbox, each behavioral statement ("now retries 3 times", "adds the X flag"). +2. For each claim, find its proof in the diff (and, where a claim is about runtime behavior, in the files the diff touches). +3. Classify each claim: + - **proven** — the diff demonstrably does it. + - **unfulfilled** — the diff does not do it (missing, partial, or different). + - **unverifiable** — cannot be determined from the diff (say why). +4. Report unfulfilled and unverifiable claims only. Proven claims produce no finding. + +## Output + +Return a JSON array (empty `[]` when every claim is proven): + +```json +[ + { + "file": "src/foo.ts", + "line": 1, + "candidate_severity": "BLOCKING | CONSIDER", + "body": "Claim: 'adds retry with backoff'. The diff adds the flag but never reads it.", + "reference": "" + } +] +``` + +- Anchor on the file the claim concerns (line 1 if the claim spans the PR). `candidate_severity` is BLOCKING for an unfulfilled claim, CONSIDER for an unverifiable one. +- Never fabricate a claim the body does not make. Never flag a claim that is proven. + +Follow `references/voice.md`: name the bullet, then the gap. One or two short sentences. + +## Markdown output + +**Markdown output: soft-wrap prose, never hard-wrap** — when you emit markdown, write each paragraph as one continuous line. Newlines still separate paragraphs, list items, headings, and code fences. diff --git a/skills/pr/agents/red-team-adversary.md b/skills/pr/agents/red-team-adversary.md new file mode 100644 index 0000000..6a02c21 --- /dev/null +++ b/skills/pr/agents/red-team-adversary.md @@ -0,0 +1,82 @@ +--- +name: red-team-adversary +description: Adversarial bug-hunt pass for /pr review (Dimension 5). Hunts failure modes, edge cases, negative paths, and security weaknesses in a PR/MR diff; reviews tests with a mutation mindset; walks STRIDE on security-sensitive surfaces. Findings carry a concrete failing scenario; severity is assigned by the caller, not by this agent. +tools: Read, Grep, Glob, Bash +metadata: + model-tier: standard +--- + +Your entire job is to try to break this change. The other review passes own intent (Dimension 1), slop (slop-detector), craft (code-quality-auditor), and docs (docs-freshness-checker). You own what they all miss: failure modes, edge cases, negative paths, and security weaknesses. + +Work as a skeptical engineer, not a hostile one. Assume the author is wrong as a **hypothesis generator** — then desk-check every hypothesis against the diff and the surrounding code, and drop what does not survive. You report candidates with evidence; you never decide severity — the review workflow assigns BLOCKING/CONSIDER/NOTE from your scenarios. + +## What to hunt + +**1. Failure modes and negative paths** +- Inputs that are empty, oversized, malformed, duplicated, or attacker-controlled. +- Error branches: what happens when the call fails, times out, or returns null? Is the error swallowed, retried forever, or surfaced? +- Concurrency: races, deadlocks, double-submission, out-of-order delivery — where the diff touches shared state. +- Partial failure: what state is left behind if step 3 of 5 fails? + +**2. Edge cases** +- Boundaries: zero, one, max, max+1, negative, unicode, timezone/DST. +- First-run and empty-state paths; migration of pre-existing data. + +**3. Tests, adversarially (mutation mindset)** +- For each new/changed test: would it change color if the code it covers regressed? Flip a line mentally — if no test notices, say so. +- Does the test assert observable behavior, or a tautology (asserting the implementation against itself)? +- Does the test codify the *intended* contract or merely the *implemented* behavior (which silently blesses bugs)? + +**4. Security (STRIDE) — only where the diff touches a security surface** +If any changed line touches auth/authz, crypto, secrets, external input parsing, (de)serialization, network/file I/O, shell execution, or dependencies, emit one line per STRIDE category — a finding or an explicit "checked, no finding": +- **S**poofing, **T**ampering, **R**epudiation, **I**nformation disclosure, **D**enial of service, **E**levation of privilege. +Tag each security finding with its STRIDE letter and a CWE class (e.g. `T / CWE-22`). Your security findings are hypotheses that complement static tools — never claim you ran a scanner. + +## What is NOT your job (do not duplicate) + +- Intent gaps ("does the diff deliver the stated purpose") — Dimension 1 owns those. +- Unnecessary/out-of-scope changes and slop — slop-detector owns those. +- Idiom, naming, structure, typing style — code-quality-auditor owns those. +- Docs freshness — docs-freshness-checker owns those. + +## How to work + +1. Read the diff hunk by hunk. For each changed function, read its callers and callees (Grep/Read) — bugs live at the seams. +2. Generate hypotheses (assume the author is wrong), then desk-check each: trace a concrete input through the changed code and write down expected vs observed behavior. +3. Keep only hypotheses that survive the desk-check. For each, produce the scenario below. If you cannot produce a concrete scenario, downgrade your own candidate to CONSIDER and say what you could not verify. +4. Large diffs (>150 changed lines) are where review quality collapses — prioritize the seams (callers/callees of changed functions) and list what you could not cover in `what_was_not_verified`. + +## Output + +Return a JSON object: + +```json +{ + "findings": [ + { + "file": "src/upload.ts", + "line": 88, + "candidate_severity": "BLOCKING | CONSIDER | NOTE", + "body": "Terse: the problem and the ask.", + "reference": "", + "scenario": { + "input": "concrete input or trigger", + "expected": "what should happen", + "observed": "what the code actually does", + "stride": "T", + "cwe": "CWE-22" + } + } + ], + "what_was_not_verified": ["areas of the diff you could not desk-check"] +} +``` + +- `scenario` is REQUIRED for every BLOCKING candidate. `stride`/`cwe` appear only on security findings. +- Return `{"findings": [], "what_was_not_verified": []}` for a diff that survives the hunt — never invent findings. + +Follow `references/voice.md`: adversarial findings state `input → expected → observed` and stay falsifiable, not persuasive. One or two short sentences plus the scenario. + +## Markdown output + +**Markdown output: soft-wrap prose, never hard-wrap** — when you emit markdown — a `.md` artifact or a markdown field in your returned output — write each paragraph as one continuous line. Do not insert manual newlines to wrap prose at a fixed column width; let the renderer wrap. Newlines still separate paragraphs, list items, headings, and code fences. diff --git a/skills/pr/references/new-review-template.md b/skills/pr/references/new-review-template.md index 2fc5645..f0f09f9 100644 --- a/skills/pr/references/new-review-template.md +++ b/skills/pr/references/new-review-template.md @@ -12,9 +12,17 @@ Write this structure to `.codevoyant/review/{slug}/new-review.md`. - **Stats**: +{additions} -{deletions} across {changedFiles} files - **Reviewed**: {timestamp} -## Summary — does this deliver its intent? +## Overall issues -{Lead with an intent verdict: does the change deliver its stated purpose end-to-end? Trace the headline use case. Then the main concern. No filler phrases.} +{BLOCKING findings that are structural/PR-wide and cannot be anchored to a file:line. One bullet per issue: the problem and what must change. If there are none, delete this entire section — the review then posts file-level comments only. Never restate what the review did, never summarize the diff, never narrate the verdict.} + +## Verification + +{Which deterministic gates ran and their status — CI, project format/lint/typecheck, semgrep/bandit/trufflehog, commit consistency — one line each, including skips.} + +## What was NOT verified + +{The red-team-adversary's what_was_not_verified list plus anything the reviewer could not confirm from the diff. One bullet each.} ## Inline Comments diff --git a/skills/pr/references/security-gates.md b/skills/pr/references/security-gates.md new file mode 100644 index 0000000..66c29f8 --- /dev/null +++ b/skills/pr/references/security-gates.md @@ -0,0 +1,24 @@ +# security-gates — the deterministic floor for /pr review + +Review's machine-backed layer: the project's own tooling first, then free industry-standard static tools (no subscriptions, local runs only). Detection is opportunistic — a missing tool is reported as a skip in the review's Verification section, never a failure and never silently ignored. Pin one version per tool so "clean" means the same thing on every run. + +## Project tooling (run first) + +Detect the project's task runner (`mise.toml` / `justfile` / `Makefile` / `package.json` scripts) and run its **format check**, **lint**, and **typecheck** recipes when they exist — the same recipes the git-commit workflow runs. Violations become `Quality:` findings anchored to the offending files. No task runner or no recipe → skip (recorded in Verification). + +## Static floor (free tools, pinned) + +| Tool | Purpose | Install (opportunistic) | License | +| --- | --- | --- | --- | +| semgrep `1.99.0` | SAST patterns across languages | `pipx install semgrep==1.99.0` or `pip install semgrep==1.99.0` | LGPL-2.1 (OSS engine) | +| bandit `1.8.3` | Python AST security scan | `pipx install bandit==1.8.3` | Apache-2.0 | +| trufflehog `3.88.2` | secret discovery (800+ types, live validation) | `brew install trufflehog` or `pipx install trufflehog==3.88.2` | AGPL-3.0 | + +Run only what the diff's languages need (bandit only when Python files changed; semgrep with `--config auto` scoped to changed files; trufflehog on the branch range `main..HEAD` — or the PR base — filesystem mode). Raw findings go to the `security-curator` step in review.md: it filters false positives, adds CWE/STRIDE tags, and never invents findings the tools did not produce. Secrets are trufflehog's job — the LLM passes only look for secret-like patterns the detectors miss. + +## Rules + +- Never block review on a missing tool — record the skip in the Verification section. +- Never let an LLM pass claim a scanner's output — tool findings are attributed to the tool. +- Static findings the curator cannot verify downgrade to NOTE (transparent, not dropped). +- Track AGPL-3.0 (trufflehog) for distribution contexts that restrict it. diff --git a/skills/pr/references/voice.md b/skills/pr/references/voice.md index 9b91193..afa76fd 100644 --- a/skills/pr/references/voice.md +++ b/skills/pr/references/voice.md @@ -42,6 +42,8 @@ Prefer (terse — problem + ask): Then, if it helps, a short code suggestion — not a paragraph. +**File-level by default.** Reviews post resolvable file-level comments. A top-level comment appears only for structural/overall issues that must be addressed and cannot be pinned to a file:line — and never to restate what the review did or what the diff contains. If everything is anchored, there is no top-level comment. + ## Bugfix descriptions Bugfix bodies (`pr-bug.md`, used by `/pr open --bug` or a `fix/`/`bug/` branch) have three extra rules on top of the voice above: diff --git a/skills/pr/references/workflows/address.md b/skills/pr/references/workflows/address.md index 7c07432..a840972 100644 --- a/skills/pr/references/workflows/address.md +++ b/skills/pr/references/workflows/address.md @@ -41,7 +41,13 @@ Identical to `new.md` Steps 1–2. Populate `PROVIDER`, `PR_NUMBER`, `PR_TITLE`, ## Step 2: Resolve Review Directory -Derive `SLUG` (or use `--name`); set `REVIEW_DIR=.codevoyant/review/{SLUG}`; create if absent. +Derive `SLUG` (or use `--name`). Initialize the shared store, then set `REVIEW_DIR=.codevoyant/review/{SLUG}`; create if absent: + +```bash +# {SKILL_ROOT} = the pr skill's package root (substitute the real path) +python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null +mkdir -p .codevoyant/review/{SLUG} +``` - If `${REVIEW_DIR}/comments.md` already exists, overwrite it — it is always regenerated from the live PR/MR state. - If `${REVIEW_DIR}/address.md` exists with any `Status: APPLIED` entries, warn: diff --git a/skills/pr/references/workflows/publish.md b/skills/pr/references/workflows/publish.md index b99d380..7dca65f 100644 --- a/skills/pr/references/workflows/publish.md +++ b/skills/pr/references/workflows/publish.md @@ -65,13 +65,13 @@ If none of the three is true: `✓ Nothing to publish — PR/MR #{PR_NUMBER} is ## Step 2.5: Resolve the review body -Before any review submission, resolve a **non-empty markdown** `REVIEW_BODY`: +Before any review submission, resolve a **non-empty markdown** `REVIEW_BODY` — the top-level comment is reserved for structural/overall issues and never restates the review: -1. If a matching local review doc exists, read its `## Summary` section and use that paragraph (trimmed) as the top-level review comment, formatted as markdown. +1. If a matching local review doc exists, read its `## Overall issues` section. If it has content, use those bullets (trimmed) as the top-level review comment. If the section is absent or empty, use the fixed minimal body `Inline comments only.` — never generate prose, never summarize the findings. 2. Else if a pending review already carries a body, reuse it (GitHub: `gh api "repos/:owner/:repo/pulls/{PR_NUMBER}/reviews/{review_id}" --jq '.body'`). -3. Else derive a one-line summary that matches the event: `Submitted via /pr publish.` (COMMENT), `Approved via /pr publish.` (APPROVE), or `Requesting changes via /pr publish.` (REQUEST_CHANGES). +3. Else use the fixed minimal body matching the event: `Inline comments only.` (COMMENT), `Approved — inline comments only.` (APPROVE), or `Requesting changes — see inline comments.` (REQUEST_CHANGES). -Never submit a review with an empty body — GitHub rejects it (`Review cannot be submitted with empty body and comments`), and an empty top-level comment renders as nothing. +Never submit a review with an empty body — GitHub rejects it (`Review cannot be submitted with empty body and comments`). Never write a top-level restatement of what the review did. ## Step 3: Push an unpublished local review doc diff --git a/skills/pr/references/workflows/review.md b/skills/pr/references/workflows/review.md index bc3e0d4..c95e9c4 100644 --- a/skills/pr/references/workflows/review.md +++ b/skills/pr/references/workflows/review.md @@ -76,6 +76,9 @@ Store: `PR_NUMBER`, `PR_TITLE`, `PR_URL`. If `--name` was given, use it. Otherwise derive from `PR_TITLE`: lowercase, replace non-alnum with `-`, collapse runs of `-`, trim to 50 chars. ```bash +# {SKILL_ROOT} = the pr skill's package root (substitute the real path). +# Initialize the shared store before first use. +python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null REVIEW_DIR=".codevoyant/review/${SLUG}" mkdir -p "$REVIEW_DIR" ``` @@ -114,12 +117,23 @@ HEAD_REF=$(echo "$META" | jq -r '.source_branch') ## Step 6: Assess the change with parallel subagents -`/pr review` intentionally assesses the branch/PR across **four dimensions**, each handled by a focused subagent. Launch all four **in the same message** so they run concurrently, then merge their findings (Step 6e) before writing the review. The four dimensions: +### Step 5.5: Deterministic pre-checks (non-blocking) + +Before the agent fan-out, run the deterministic checks from `references/security-gates.md` and record each result for the review's Verification section: + +1. **CI status (best-effort warning).** GitHub: `gh pr checks "$PR_NUMBER"`; GitLab: `glab ci status`. Not green → record `⚠ CI is {failing|pending}` — review proceeds (merge/publish gate CI later). No CLI or no CI configured → skip, recorded. +2. **Commit consistency.** Fetch the commits (GitHub: `gh pr view "$PR_NUMBER" --json commits --jq '.commits[] | .messageHeadline'`; GitLab: `glab mr view "$PR_NUMBER" --output json | jq -r '.commits[].title'`). A non-conventional subject or a message that contradicts its diff becomes a NOTE finding (prefix `Commits: `). +3. **Static floor + project tooling.** Per `references/security-gates.md`: the project's format/lint/typecheck recipes first, then semgrep/bandit/trufflehog where applicable. Save raw findings as `STATIC_RAW`; every skip is recorded. + +`/pr review` intentionally assesses the branch/PR across **five dimensions**, each handled by a focused subagent. Launch all five **in the same message** so they run concurrently, then merge their findings (Step 6e) before writing the review. The five dimensions: 1. **Intent-match** (Dimension 1, below) — does the diff deliver the stated intent end-to-end? 2. **Unnecessary changes** (Dimension 2 / Step 6b) — scope creep, stray edits, dead/commented code, accidental reverts, unrelated churn from a poorly-harnessed agentic run. 3. **Code quality** (Dimension 3 / Step 6c) — is the added/edited code high quality per the relevant codevoyant skill or the language/framework standard? 4. **Docs freshness** (Dimension 4 / Step 6d) — were docs updated? If not, report a `Docs:` finding recommending `/docs update` (default, read-only), or — only when `--update-docs` is set — invoke `/docs` to update them. +5. **Adversarial hunt** (Dimension 5 / Step 6e-launch) — the `red-team-adversary` agent tries to break the change: failure modes, edge cases, negative paths, mutation-mindset test review, STRIDE on security surfaces. + +In the same message, also launch the **claim-checker** agent (`agents/claim-checker.md`) with `{TITLE}`, `{BODY}`, and `{DIFF_CONTENT}` — it verifies the body's claims against the diff (Step 6f). ### Dimension 1 — Intent-match & correctness @@ -197,20 +211,35 @@ Launch the **docs-freshness-checker** agent (`agents/docs-freshness-checker.md`) **By default (`UPDATE_DOCS=false`), review stays read-only:** if docs are stale and not updated in the diff, the agent returns a `Docs:` CONSIDER finding recommending the author run `/docs update` — it does **not** mutate the working tree. Only when `UPDATE_DOCS=true` (the caller passed `--update-docs`) does it invoke `/docs update` to bring docs current and return a NOTE recording what it did. If docs are fine (or the change needs none) it returns `[]`. -### Step 6e — Merge all dimensions +### Dimension 5 — Adversarial hunt (Step 6e-launch) + +Launch the **red-team-adversary** agent (`agents/red-team-adversary.md`) via the Agent tool with `subagent_type: red-team-adversary` — in the **same message** as Dimensions 2–4 and the claim-checker. Give it `{TITLE}`, `{BODY}`, and `{DIFF_CONTENT}`. It returns `{"findings": [...], "what_was_not_verified": [...]}` per its schema. Store the not-verified list for the template's disclosure section. Do NOT feed it prior review comments or anchor metadata — the prompt stays skeptical and unanchored by design. -Merge every dimension's findings into a single comment array before Step 7: -- Concatenate the four arrays: reviewer (Dimension 1), slop-detector (Dimension 2), code-quality-auditor (Dimension 3), docs-freshness-checker (Dimension 4). -- De-duplicate by `file:line` + overlapping intent (keep the more specific/severe of a pair). -- Prefix bodies by source so the author sees which pass raised each: slop findings with `Slop: `, code-quality findings with `Quality: `, docs findings with `Docs: ` (the reviewer's intent/correctness findings are unprefixed). +### Claim check (Step 6f) -Extend the overall summary: after the intent verdict, add one line each for any non-empty dimension — how many unnecessary-change findings, code-quality findings, and whether docs were refreshed (or need refreshing) — so the reader sees the full assessment at a glance. +The **claim-checker** agent (`agents/claim-checker.md`) is launched in the same message as the five dimensions. Give it `{TITLE}`, `{BODY}`, and `{DIFF_CONTENT}`. It parses the body into individual claims — each `Changes` bullet, each `Validation` checkbox, each behavioral statement — and proves or disproves each one against the diff, classifying every claim as proven, unfulfilled, or unverifiable. It returns a JSON array in the same schema (`file`, `line`, `candidate_severity`, `body`, `reference`): unfulfilled claims carry `candidate_severity: BLOCKING`, unverifiable ones `CONSIDER`; proven claims produce no finding. An empty `[]` means every claim in the body checks out. + +### Step 6e — Merge and assign severity (the decision layer) + +Agents detect; this step decides. Merge every source into one comment array and assign each finding's final severity here — severity never comes from an agent's prompt: + +1. **Concatenate:** reviewer (Dimension 1), slop-detector (2), code-quality-auditor (3), docs-freshness-checker (4), red-team-adversary findings (5), claim-checker (6f), and curated `STATIC_RAW` findings (Step 5.5). +2. **Prefix bodies by source:** `Slop: `, `Quality: `, `Docs: `, `Adversarial: `, `Claim: `, `Static: ` — Dimension 1 findings stay unprefixed. Adversarial findings append their scenario (`Input: … expected: … observed: …`), and security findings their STRIDE/CWE tags. +3. **Assign severity:** + - Adversarial findings: **BLOCKING iff** `scenario.input/expected/observed` are all concrete and consistent with the diff; any vague or missing leg → downgrade to CONSIDER. No scenario at all → CONSIDER. + - Claim findings: unfulfilled claim → BLOCKING; unverifiable → CONSIDER. + - Static findings the curator could not verify → NOTE (transparent, not dropped). Docs findings stay capped at CONSIDER/NOTE. + - All others keep their agent-assigned severity. +4. **De-duplicate** by `file:line` + overlapping intent (keep the more specific/severe of a pair). +5. **Classify anchorability:** a finding is *un-anchorable* when it is structural/PR-wide and has no specific file:line (e.g. the change's architecture contradicts its stated approach; a one-way door with no rollback). Un-anchorable BLOCKING findings go to the review doc's `## Overall issues` section — everything else stays a file-level comment. + +Do NOT write an overall summary of what the review did — the review speaks through its findings. ## Step 7: Write Review Document Read `references/new-review-template.md`. Replace all `{placeholder}` tokens via direct string substitution using the resolved values (`$TITLE`, `$AUTHOR`, `$BASE_REF`, `$HEAD_REF`, `$ADDITIONS`, `$DELETIONS`, `$CHANGED_FILES`, `$PR_NUMBER`, `$PR_URL`, current timestamp, and the summary paragraph). -For the inline comments section, iterate over the JSON array and render each entry using the template's `### {file}:{line} — {severity}` block. +For the inline comments section, render every ANCHORED finding (all findings except the un-anchorable BLOCKING ones classified in Step 6e) using the template's `### {file}:{line} — {severity}` block. Render the un-anchorable BLOCKING findings into `## Overall issues`; if there are none, delete that section from the doc. Fill `## Verification` from the Step 5.5 records and `## What was NOT verified` from the adversary's list. The doc carries no summary of what the review did. Write the populated content to `${REVIEW_DIR}/new-review.md`. @@ -234,6 +263,8 @@ Do not push anything. - GitHub: `/gh push-comments {PR_NUMBER} --doc {REVIEW_DIR}/new-review.md` then `/gh draft {PR_NUMBER}` - GitLab: `/glab push-comments {PR_NUMBER} --doc {REVIEW_DIR}/new-review.md` then `/glab draft {PR_NUMBER} --draft` +Only file-level comments are pushed here. The top-level review body is resolved at publish time (publish.md Step 2.5) from `## Overall issues` — when that section is absent, publish uses a fixed minimal body and posts no review prose at all. + Report: ``` diff --git a/skills/pr/scripts/cv_init_store.py b/skills/pr/scripts/cv_init_store.py new file mode 100644 index 0000000..00bb73c --- /dev/null +++ b/skills/pr/scripts/cv_init_store.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Canonical codevoyant store initializer — shared asset, vendored per skills/vendor.json. + +Ensures the in-repo `.codevoyant` is a symlink to the shared per-project store +(~/.codevoyant//) BEFORE any workflow mkdirs under it, so worktrees +and regular clones of the same project share one store. + +Idempotent. Never migrates an existing real dir (that is /migrate's job). +The slug pipeline is byte-identical to the /migrate skill's computation +(worktree-aware via `git rev-parse --git-common-dir`). + +Usage: cv_init_store.py [repo-root] (defaults to the git toplevel, else cwd) +Prints the computed slug on stdout. Exit 0 in all handled cases. +""" +import os +import re +import subprocess +import sys +from pathlib import Path + + +def _git(args, root): + try: + proc = subprocess.run( + ["git", "-C", str(root), *args], + capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return proc.stdout.strip() if proc.returncode == 0 else "" + + +def compute_slug(root): + """Same pipeline as the bash original: lower, [^a-z0-9]+ -> '-', strip '-'.""" + common = _git(["rev-parse", "--git-common-dir"], root) + if common: + cpath = Path(common) + if not cpath.is_absolute(): + cpath = root / cpath + try: + name = cpath.parent.resolve().name + except OSError: + name = root.name + else: + name = root.name + slug = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") + return slug or "unnamed" + + +def init_store(root, home=None): + """Create the ~/.codevoyant/ symlink for `root`; return the slug. + + - Already a symlink -> no-op. + - Existing real dir -> no-op (migration is /migrate's job). + - Otherwise -> mkdir dest, symlink, append .gitignore entry once. + """ + root = Path(root) + home = Path(home) if home else Path(os.environ["HOME"]) + slug = compute_slug(root) + link = root / ".codevoyant" + if link.is_symlink(): + return slug + if link.is_dir(): + return slug + dest = home / ".codevoyant" / slug + dest.mkdir(parents=True, exist_ok=True) + link.symlink_to(dest) + gi = root / ".gitignore" + lines = gi.read_text(encoding="utf-8").splitlines() if gi.exists() else [] + if ".codevoyant" not in lines: + with gi.open("a", encoding="utf-8") as f: + f.write("\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n") + return slug + + +def main(argv): + if len(argv) > 1: + root = Path(argv[1]) + else: + top = _git(["rev-parse", "--show-toplevel"], Path.cwd()) + root = Path(top) if top else Path.cwd() + print(init_store(root)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/qa/references/workflows/debug.md b/skills/qa/references/workflows/debug.md index bffacfe..4fa116f 100644 --- a/skills/qa/references/workflows/debug.md +++ b/skills/qa/references/workflows/debug.md @@ -11,32 +11,11 @@ DESCRIPTION --desc "..." (optional one-line description of the bug) ## Step 1: Create report directory -`cv_init_store` ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`) **before** the first `mkdir` — otherwise a fresh clone would create `.codevoyant` as a real directory instead of the shared symlink. It is idempotent, never migrates an existing real dir (that is `/migrate`'s job), and computes the identical `` as the `/migrate` skill. +Before the first `mkdir`, run the vendored store initializer — it ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`), so worktrees and regular clones share one store. Idempotent; never migrates an existing real dir (that is `/migrate`'s job). Substitute `{SKILL_ROOT}` with this skill's package root (the directory containing the qa skill's SKILL.md, as reported when the skill loaded). ```bash -cv_init_store() { - local root; root="$(git rev-parse --show-toplevel 2>/dev/null)" || root=""; [ -n "$root" ] || root="$PWD" - local link="$root/.codevoyant" - [ -L "$link" ] && return 0 # already a symlink → initialized - [ -d "$link" ] && return 0 # old real dir → leave it; /migrate copies it in, never here - local common name slug - common="$(git rev-parse --git-common-dir 2>/dev/null)" || common="" - if [ -n "$common" ]; then - case "$common" in /*) : ;; *) common="$root/$common" ;; esac - name="$(basename "$(cd "$(dirname "$common")" >/dev/null 2>&1 && pwd -P)")" - else - name="$(basename "$root")" - fi - slug="$(printf '%s' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | LC_ALL=C sed 's/[^a-z0-9][^a-z0-9]*/-/g; s/^-*//; s/-*$//')" - [ -n "$slug" ] || slug="unnamed" # empty-slug fallback — matches the /migrate skill - local dest="$HOME/.codevoyant/$slug" - mkdir -p "$dest"; ln -s "$dest" "$link" - local gi="$root/.gitignore" - { [ -f "$gi" ] && grep -qxF '.codevoyant' "$gi"; } || \ - printf '\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n' >> "$gi" -} - -cv_init_store +# {SKILL_ROOT} = the qa skill's package root (substitute the real path) +python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null mkdir -p .codevoyant/qa/{slug} ``` diff --git a/skills/qa/references/workflows/smoke.md b/skills/qa/references/workflows/smoke.md index c136d02..35efc13 100644 --- a/skills/qa/references/workflows/smoke.md +++ b/skills/qa/references/workflows/smoke.md @@ -162,6 +162,8 @@ agent-browser --session {slug} close ## Step 7: Write smoke report +Before writing the report, initialize the shared store — substitute `{SKILL_ROOT}` with the qa skill's package root: `python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null`. + Write `.codevoyant/qa/{slug}/smoke-report.md` using `references/templates/smoke-report.md`. ## Step 8: Report diff --git a/skills/qa/scripts/cv_init_store.py b/skills/qa/scripts/cv_init_store.py new file mode 100644 index 0000000..00bb73c --- /dev/null +++ b/skills/qa/scripts/cv_init_store.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Canonical codevoyant store initializer — shared asset, vendored per skills/vendor.json. + +Ensures the in-repo `.codevoyant` is a symlink to the shared per-project store +(~/.codevoyant//) BEFORE any workflow mkdirs under it, so worktrees +and regular clones of the same project share one store. + +Idempotent. Never migrates an existing real dir (that is /migrate's job). +The slug pipeline is byte-identical to the /migrate skill's computation +(worktree-aware via `git rev-parse --git-common-dir`). + +Usage: cv_init_store.py [repo-root] (defaults to the git toplevel, else cwd) +Prints the computed slug on stdout. Exit 0 in all handled cases. +""" +import os +import re +import subprocess +import sys +from pathlib import Path + + +def _git(args, root): + try: + proc = subprocess.run( + ["git", "-C", str(root), *args], + capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return proc.stdout.strip() if proc.returncode == 0 else "" + + +def compute_slug(root): + """Same pipeline as the bash original: lower, [^a-z0-9]+ -> '-', strip '-'.""" + common = _git(["rev-parse", "--git-common-dir"], root) + if common: + cpath = Path(common) + if not cpath.is_absolute(): + cpath = root / cpath + try: + name = cpath.parent.resolve().name + except OSError: + name = root.name + else: + name = root.name + slug = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") + return slug or "unnamed" + + +def init_store(root, home=None): + """Create the ~/.codevoyant/ symlink for `root`; return the slug. + + - Already a symlink -> no-op. + - Existing real dir -> no-op (migration is /migrate's job). + - Otherwise -> mkdir dest, symlink, append .gitignore entry once. + """ + root = Path(root) + home = Path(home) if home else Path(os.environ["HOME"]) + slug = compute_slug(root) + link = root / ".codevoyant" + if link.is_symlink(): + return slug + if link.is_dir(): + return slug + dest = home / ".codevoyant" / slug + dest.mkdir(parents=True, exist_ok=True) + link.symlink_to(dest) + gi = root / ".gitignore" + lines = gi.read_text(encoding="utf-8").splitlines() if gi.exists() else [] + if ".codevoyant" not in lines: + with gi.open("a", encoding="utf-8") as f: + f.write("\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n") + return slug + + +def main(argv): + if len(argv) > 1: + root = Path(argv[1]) + else: + top = _git(["rev-parse", "--show-toplevel"], Path.cwd()) + root = Path(top) if top else Path.cwd() + print(init_store(root)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/shared/store-init/cv_init_store.py b/skills/shared/store-init/cv_init_store.py new file mode 100644 index 0000000..00bb73c --- /dev/null +++ b/skills/shared/store-init/cv_init_store.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Canonical codevoyant store initializer — shared asset, vendored per skills/vendor.json. + +Ensures the in-repo `.codevoyant` is a symlink to the shared per-project store +(~/.codevoyant//) BEFORE any workflow mkdirs under it, so worktrees +and regular clones of the same project share one store. + +Idempotent. Never migrates an existing real dir (that is /migrate's job). +The slug pipeline is byte-identical to the /migrate skill's computation +(worktree-aware via `git rev-parse --git-common-dir`). + +Usage: cv_init_store.py [repo-root] (defaults to the git toplevel, else cwd) +Prints the computed slug on stdout. Exit 0 in all handled cases. +""" +import os +import re +import subprocess +import sys +from pathlib import Path + + +def _git(args, root): + try: + proc = subprocess.run( + ["git", "-C", str(root), *args], + capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return proc.stdout.strip() if proc.returncode == 0 else "" + + +def compute_slug(root): + """Same pipeline as the bash original: lower, [^a-z0-9]+ -> '-', strip '-'.""" + common = _git(["rev-parse", "--git-common-dir"], root) + if common: + cpath = Path(common) + if not cpath.is_absolute(): + cpath = root / cpath + try: + name = cpath.parent.resolve().name + except OSError: + name = root.name + else: + name = root.name + slug = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") + return slug or "unnamed" + + +def init_store(root, home=None): + """Create the ~/.codevoyant/ symlink for `root`; return the slug. + + - Already a symlink -> no-op. + - Existing real dir -> no-op (migration is /migrate's job). + - Otherwise -> mkdir dest, symlink, append .gitignore entry once. + """ + root = Path(root) + home = Path(home) if home else Path(os.environ["HOME"]) + slug = compute_slug(root) + link = root / ".codevoyant" + if link.is_symlink(): + return slug + if link.is_dir(): + return slug + dest = home / ".codevoyant" / slug + dest.mkdir(parents=True, exist_ok=True) + link.symlink_to(dest) + gi = root / ".gitignore" + lines = gi.read_text(encoding="utf-8").splitlines() if gi.exists() else [] + if ".codevoyant" not in lines: + with gi.open("a", encoding="utf-8") as f: + f.write("\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n") + return slug + + +def main(argv): + if len(argv) > 1: + root = Path(argv[1]) + else: + top = _git(["rev-parse", "--show-toplevel"], Path.cwd()) + root = Path(top) if top else Path.cwd() + print(init_store(root)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/shared/store-init/test_cv_init_store.py b/skills/shared/store-init/test_cv_init_store.py new file mode 100644 index 0000000..4603573 --- /dev/null +++ b/skills/shared/store-init/test_cv_init_store.py @@ -0,0 +1,116 @@ +#!/usr/bin/env python3 +"""Unit tests for cv_init_store.py (shared store initializer). + +Run by `mise run test` (discovery pattern skills/**/test_*.py). +Builds throwaway git repos under a temp dir; never touches the real HOME. +""" +import os +import pathlib +import subprocess +import sys +import tempfile +import unittest + +sys.path.insert(0, str(pathlib.Path(__file__).parent)) +import cv_init_store as cvis + + +def git(repo, *args): + env = dict(os.environ) + env.update({ + "GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@example.com", + "GIT_COMMITTER_NAME": "t", "GIT_COMMITTER_EMAIL": "t@example.com", + }) + subprocess.run( + ["git", "-C", str(repo), *args], + check=True, capture_output=True, text=True, env=env, + ) + + +def make_repo(base, name): + repo = base / name + repo.mkdir(parents=True) + git(repo, "init", "-q") + (repo / "file.txt").write_text("x") + git(repo, "add", "file.txt") + git(repo, "commit", "-qm", "init") + return repo + + +class StoreInit(unittest.TestCase): + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self.addCleanup(self._tmp.cleanup) + self.base = pathlib.Path(self._tmp.name) + self.home = self.base / "home" + self.home.mkdir() + + def test_regular_repo_creates_symlink_named_by_slug(self): + repo = make_repo(self.base, "My Proj") + slug = cvis.init_store(repo, home=self.home) + self.assertEqual(slug, "my-proj") + link = repo / ".codevoyant" + self.assertTrue(link.is_symlink()) + self.assertEqual(os.readlink(link), str(self.home / ".codevoyant" / "my-proj")) + self.assertIn(".codevoyant", (repo / ".gitignore").read_text().splitlines()) + + def test_worktree_and_main_repo_share_one_store(self): + repo = make_repo(self.base, "origin-repo") + wt = self.base / "wt-checkout" + git(repo, "worktree", "add", "-q", "-b", "wt", str(wt)) + slug_main = cvis.init_store(repo, home=self.home) + slug_wt = cvis.init_store(wt, home=self.home) + self.assertEqual(slug_wt, slug_main) + self.assertEqual( + os.readlink(wt / ".codevoyant"), + os.readlink(repo / ".codevoyant"), + ) + + def test_idempotent_and_single_gitignore_entry(self): + repo = make_repo(self.base, "idem") + cvis.init_store(repo, home=self.home) + cvis.init_store(repo, home=self.home) + lines = (repo / ".gitignore").read_text().splitlines() + self.assertEqual(lines.count(".codevoyant"), 1) + self.assertTrue((repo / ".codevoyant").is_symlink()) + + def test_existing_real_dir_left_untouched(self): + repo = make_repo(self.base, "realdir") + (repo / ".codevoyant").mkdir() + cvis.init_store(repo, home=self.home) + self.assertTrue((repo / ".codevoyant").is_dir()) + self.assertFalse((repo / ".codevoyant").is_symlink()) + + def test_non_git_dir_falls_back_to_basename_slug(self): + plain = self.base / "Plain Dir" + plain.mkdir() + slug = cvis.init_store(plain, home=self.home) + self.assertEqual(slug, "plain-dir") + self.assertTrue((plain / ".codevoyant").is_symlink()) + + def test_slug_parity_with_bash_pipeline(self): + # Names that exercise lowercasing, runs of specials, and leading/trailing hyphens. + cases = { + "Codevoyant": "codevoyant", + "My--Cool Repo!!": "my-cool-repo", + "__edge__": "edge", + "already-lower": "already-lower", + } + for name, want in cases.items(): + d = self.base / name + d.mkdir() + self.assertEqual(cvis.compute_slug(d), want, name) + + def test_cli_prints_slug(self): + repo = make_repo(self.base, "cli-repo") + import io + import contextlib + buf = io.StringIO() + with contextlib.redirect_stdout(buf): + rc = cvis.main(["cv_init_store.py", str(repo)]) + self.assertEqual(rc, 0) + self.assertEqual(buf.getvalue().strip(), "cli-repo") + + +if __name__ == "__main__": + unittest.main() diff --git a/skills/shared/test_vendor_assets.py b/skills/shared/test_vendor_assets.py index d9ac5c9..01f358a 100644 --- a/skills/shared/test_vendor_assets.py +++ b/skills/shared/test_vendor_assets.py @@ -60,19 +60,36 @@ def target_files(self): (self.root / "skills" / "spec" / "scripts").iterdir() ) - def test_propagates_new_source_file(self): - """A file added to the shared source propagates and --check passes.""" + def test_unlisted_source_file_stays_in_shared(self): + """`files` is an exclusive filter: a source-dir file not listed is NOT vendored.""" src = self.root / "skills" / "shared" / "scope-scripts" (src / "helper.py").write_text("HELPER\n") proc = self.run_tool() self.assertEqual(proc.returncode, 0, proc.stderr) - self.assertTrue((self.root / "skills" / "spec" / "scripts" / "helper.py").exists()) - self.assertTrue((self.root / "skills" / "docs" / "scripts" / "helper.py").exists()) + self.assertFalse((self.root / "skills" / "spec" / "scripts" / "helper.py").exists()) + self.assertFalse((self.root / "skills" / "docs" / "scripts" / "helper.py").exists()) proc = self.run_tool("--check") self.assertEqual(proc.returncode, 0, proc.stdout + proc.stderr) + def test_unfiltered_asset_walks_whole_source(self): + """An asset without `files` still walks the whole source dir.""" + cfg = json.loads(self.config.read_text()) + del cfg["assets"]["scope-scripts"]["files"] + cfg["assets"]["scope-scripts"]["source"] = "skills/shared/simple-english" + cfg["assets"]["scope-scripts"]["destination"] = "references/simple-english" + src = self.root / "skills" / "shared" / "simple-english" + src.mkdir(parents=True) + (src / "ruleset.md").write_text("RULES\n") + (src / "NOTICE.md").write_text("NOTICE\n") + self.config.write_text(json.dumps(cfg)) + + proc = self.run_tool() + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertTrue((self.root / "skills" / "spec" / "references" / "simple-english" / "ruleset.md").exists()) + self.assertTrue((self.root / "skills" / "spec" / "references" / "simple-english" / "NOTICE.md").exists()) + def test_stale_copy_flagged_and_cleaned(self): """A removed allowlist entry leaves a stale target copy --check flags and vendor cleans.""" proc = self.run_tool() diff --git a/skills/skill/references/workflows/feedback.md b/skills/skill/references/workflows/feedback.md index 6761bd9..2ff6377 100644 --- a/skills/skill/references/workflows/feedback.md +++ b/skills/skill/references/workflows/feedback.md @@ -99,32 +99,11 @@ Show the draft title and body for review. ### If `--save` is set -Write the report to `.codevoyant/feedback/`. `cv_init_store` ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`) **before** the `mkdir` — otherwise a fresh clone would create `.codevoyant` as a real directory instead of the shared symlink. It is idempotent, never migrates an existing real dir (that is `/migrate`'s job), and computes the identical `` as the `/migrate` skill. +Write the report to `.codevoyant/feedback/`. Before the `mkdir`, run the vendored store initializer — it ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`), so worktrees and regular clones share one store. Idempotent; never migrates an existing real dir (that is `/migrate`'s job). Substitute `{SKILL_ROOT}` with this skill's package root (the directory containing the skill skill's SKILL.md, as reported when the skill loaded). ```bash -cv_init_store() { - local root; root="$(git rev-parse --show-toplevel 2>/dev/null)" || root=""; [ -n "$root" ] || root="$PWD" - local link="$root/.codevoyant" - [ -L "$link" ] && return 0 # already a symlink → initialized - [ -d "$link" ] && return 0 # old real dir → leave it; /migrate copies it in, never here - local common name slug - common="$(git rev-parse --git-common-dir 2>/dev/null)" || common="" - if [ -n "$common" ]; then - case "$common" in /*) : ;; *) common="$root/$common" ;; esac - name="$(basename "$(cd "$(dirname "$common")" >/dev/null 2>&1 && pwd -P)")" - else - name="$(basename "$root")" - fi - slug="$(printf '%s' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | LC_ALL=C sed 's/[^a-z0-9][^a-z0-9]*/-/g; s/^-*//; s/-*$//')" - [ -n "$slug" ] || slug="unnamed" # empty-slug fallback — matches the /migrate skill - local dest="$HOME/.codevoyant/$slug" - mkdir -p "$dest"; ln -s "$dest" "$link" - local gi="$root/.gitignore" - { [ -f "$gi" ] && grep -qxF '.codevoyant' "$gi"; } || \ - printf '\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n' >> "$gi" -} - -cv_init_store +# {SKILL_ROOT} = the skill skill's package root (substitute the real path) +python3 "{SKILL_ROOT}/scripts/cv_init_store.py" >/dev/null mkdir -p .codevoyant/feedback TIMESTAMP=$(date +%Y%m%d-%H%M%S) FILE=".codevoyant/feedback/${SKILL_NAME}-${TIMESTAMP}.md" diff --git a/skills/skill/scripts/cv_init_store.py b/skills/skill/scripts/cv_init_store.py new file mode 100644 index 0000000..00bb73c --- /dev/null +++ b/skills/skill/scripts/cv_init_store.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Canonical codevoyant store initializer — shared asset, vendored per skills/vendor.json. + +Ensures the in-repo `.codevoyant` is a symlink to the shared per-project store +(~/.codevoyant//) BEFORE any workflow mkdirs under it, so worktrees +and regular clones of the same project share one store. + +Idempotent. Never migrates an existing real dir (that is /migrate's job). +The slug pipeline is byte-identical to the /migrate skill's computation +(worktree-aware via `git rev-parse --git-common-dir`). + +Usage: cv_init_store.py [repo-root] (defaults to the git toplevel, else cwd) +Prints the computed slug on stdout. Exit 0 in all handled cases. +""" +import os +import re +import subprocess +import sys +from pathlib import Path + + +def _git(args, root): + try: + proc = subprocess.run( + ["git", "-C", str(root), *args], + capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return proc.stdout.strip() if proc.returncode == 0 else "" + + +def compute_slug(root): + """Same pipeline as the bash original: lower, [^a-z0-9]+ -> '-', strip '-'.""" + common = _git(["rev-parse", "--git-common-dir"], root) + if common: + cpath = Path(common) + if not cpath.is_absolute(): + cpath = root / cpath + try: + name = cpath.parent.resolve().name + except OSError: + name = root.name + else: + name = root.name + slug = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") + return slug or "unnamed" + + +def init_store(root, home=None): + """Create the ~/.codevoyant/ symlink for `root`; return the slug. + + - Already a symlink -> no-op. + - Existing real dir -> no-op (migration is /migrate's job). + - Otherwise -> mkdir dest, symlink, append .gitignore entry once. + """ + root = Path(root) + home = Path(home) if home else Path(os.environ["HOME"]) + slug = compute_slug(root) + link = root / ".codevoyant" + if link.is_symlink(): + return slug + if link.is_dir(): + return slug + dest = home / ".codevoyant" / slug + dest.mkdir(parents=True, exist_ok=True) + link.symlink_to(dest) + gi = root / ".gitignore" + lines = gi.read_text(encoding="utf-8").splitlines() if gi.exists() else [] + if ".codevoyant" not in lines: + with gi.open("a", encoding="utf-8") as f: + f.write("\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n") + return slug + + +def main(argv): + if len(argv) > 1: + root = Path(argv[1]) + else: + top = _git(["rev-parse", "--show-toplevel"], Path.cwd()) + root = Path(top) if top else Path.cwd() + print(init_store(root)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/spec/agents/spec-executor.md b/skills/spec/agents/spec-executor.md index a1503b9..96253ab 100644 --- a/skills/spec/agents/spec-executor.md +++ b/skills/spec/agents/spec-executor.md @@ -101,6 +101,7 @@ set +f `SPEC_SKILL` (the spec skill package root) and `PHASE_GLOBS` (this phase's own write globs, read from its `## Doc Scope` block by `go.md`) are substituted into your prompt — never guess them. - **Permitted crossings (Rule 6).** If the phase's `## Doc Scope` boundary callouts explicitly permit a crossing, you may perform it, but you MUST append a `[DEVIATION]` entry to `execution-log.md` naming the target, the callout that permits it, and the reason. Never write outside the globs without such a callout. +- **Uncalled-out crossings are refused (Rule 7).** Cross-module changes are discouraged by default. If a task would write outside this phase's globs and no boundary callout permits it, do NOT write it and do NOT log-and-continue: stop the task, leave the out-of-scope target untouched, write a `[BLOCKED]` entry to `execution-log.md` naming the target, the task, and the missing callout, and report the defect so the plan is re-planned or the callout is added via `/spec update`. A silent crossing is exactly what doc-aware mode exists to prevent. - **Public interfaces only (Rule 4).** Cross-module interaction (reading or calling into a module owned by another doc/phase) uses only that module's documented public API/interface section — never its internals. If the needed surface is not documented, do not reach for it; log a deviation and continue with the documented surface. When `DOC_GLOBS` is empty, ignore this section entirely and execute in normal mode. diff --git a/skills/spec/references/doc-aware.md b/skills/spec/references/doc-aware.md index 5c2e84f..c6dda05 100644 --- a/skills/spec/references/doc-aware.md +++ b/skills/spec/references/doc-aware.md @@ -58,7 +58,7 @@ A candidate path that is NOT emitted is out of scope — the executor must not w ### Rule 4: Public-interface-only cross-module interaction -When a phase must interact with a module owned by a different doc/phase: +Cross-module interaction is the exception, not the norm (see Rule 7). When a phase must interact with a module owned by a different doc/phase: - Read that module's doc and use ONLY its documented public API/interface section. - Never call into, import from, or reference another module's internals (functions/types/files absent from its documented API). @@ -93,13 +93,22 @@ Mechanics: - If a boundary crossing is truly required, the planner must say so explicitly and give the reason; it is never silent. - The executor re-checks at execution time. A residual crossing (one the plan permits) is still logged as a `[DEVIATION]` entry in `execution-log.md` with the reason, per the spec skill's existing deviation mechanism — the deviation is the audit trail that the crossing was called out, not hidden. +### Rule 7: Cross-module changes are discouraged by default + +A cross-module change — a task that writes outside its phase's declared globs, edits a module another phase owns, or reaches another module's internals — is a smell before it is a solution. The default answer is to restructure, not to call out: + +1. **Restructure first.** Before permitting any crossing, the planner tries: moving the task into the phase that owns the target glob; splitting the work so each phase writes only its own module; sequencing two phases (A changes its own surface, B adapts to it) instead of one phase touching both. +2. **Permit only with justification.** A crossing survives only when restructuring is genuinely worse (it would duplicate logic, break an atomic change, or create a circular dependency). The `## Doc Scope` boundary callout then carries the reason and the rejected alternative — "required because …; restructure rejected because …". +3. **Uncalled-out crossings are defects.** At planning time, review flags a cross-module task with no callout as CRITICAL (re-plan or add the callout). At execution time, the executor refuses to write it: it stops the task and reports the defect rather than logging-and-continuing. +4. **Every crossing stays visible.** go.md's completion report lists all `[DEVIATION]` entries that are boundary crossings, so a run's cross-module footprint is one grep away. + ## How workflows use this file -- `new.md` — parse `--persistent`, run the Rule 1 gate (via `scripts/validate_docs.py`), run the Rule 2 docs-first write, apply Rule 3 scoping and Rule 6 callouts while drafting phases. +- `new.md` — parse `--persistent`, run the Rule 1 gate (via `scripts/validate_docs.py`), run the Rule 2 docs-first write, apply Rule 3 scoping, Rule 6 callouts, and Rule 7 restructure-first while drafting phases. - `update.md` — parse `--persistent`, run the Rule 1 gate (via `scripts/validate_docs.py`), run the Rule 2 docs-first write, apply Rule 3 scoping to the conversational/annotation edits. - `go.md` — read the plan's `Doc Globs:` metadata as the doc-aware activation signal, read each phase's own write globs from its `## Doc Scope` block, and pass them (as `PHASE_GLOBS`, plus `SPEC_SKILL` for the checker path) to that phase's executor; executors enforce Rules 3–4. -- `agents/spec-executor.md` — enforce Rules 3–4 for every Write/Edit and log Rule 6 deviations. -- `agents/spec-planner.md` — produce the Rule 3 scoping (per-phase `## Doc Scope` globs with at least one write glob each — a globless phase is invalid, plan.md `Doc Globs:` union) and the Rule 6 boundary callouts while drafting phases. +- `agents/spec-executor.md` — enforce Rules 3–4 for every Write/Edit, log Rule 6 deviations, and refuse uncalled-out crossings per Rule 7. +- `agents/spec-planner.md` — produce the Rule 3 scoping (per-phase `## Doc Scope` globs with at least one write glob each — a globless phase is invalid, plan.md `Doc Globs:` union), the Rule 6 boundary callouts, and the Rule 7 restructure-first justification while drafting phases. - `agents/spec-updater.md` — preserve the Rule 3 scoping and Rule 6 callouts while applying plan updates (keep `## Doc Scope` blocks and the `Doc Globs:` union in sync). ## Scripts diff --git a/skills/spec/references/implementation-template.md b/skills/spec/references/implementation-template.md index bac271b..99600b4 100644 --- a/skills/spec/references/implementation-template.md +++ b/skills/spec/references/implementation-template.md @@ -68,6 +68,9 @@ Acceptance: **Files to modify / create:** - `path/to/file.ext` — {specific changes} +**Table rows (only when the task consumes a `tables/*.md` table):** +- `tables/{set-slug}.md` — rows {#–#}: mark each row's Status `[x]` as it is completed; a row is done only when its change is applied and validated + **Validation (run after every task):** - [ ] `{fmt command}` — no formatting changes outstanding - [ ] `{lint command}` — zero warnings/errors diff --git a/skills/spec/references/intent-template.md b/skills/spec/references/intent-template.md index c9636aa..f5c8a3b 100644 --- a/skills/spec/references/intent-template.md +++ b/skills/spec/references/intent-template.md @@ -18,6 +18,10 @@ Delete any section you don't need. {Must-haves, tech choices to keep or avoid, performance/security/compatibility needs, deadlines.} +## Enumerable sets (tabulated into the plan) + +{Rote replacements (X → Y), pages/files to search or touch, and checklist-style sets of things that must be done. List every item — the planner tabulates these into `tables/` and validation checks that every item survives into the plan. An item left out here can be dropped; an item listed here cannot.} + ## Out of scope {What this should explicitly NOT touch or do — helps keep the plan tight.} diff --git a/skills/spec/references/plan-template.md b/skills/spec/references/plan-template.md index 7d7c052..8d19447 100644 --- a/skills/spec/references/plan-template.md +++ b/skills/spec/references/plan-template.md @@ -16,9 +16,12 @@ Write this structure to `.codevoyant/spec/{plan-name}/plan.md` when creating a n {What this plan is solving, for whom, and why now. 2–3 sentences.} ## Requirements -{Measurable outcomes, not deliverables. State the success condition the plan must achieve.} -- {outcome bullet 1} -- {outcome bullet 2} +{Measurable outcomes, not deliverables — per the docs skill's requirements-guidance.md rules R1–R3, R6. State the success condition the plan must achieve. No implementation terms; wording that survives an implementation change; each bullet names an observable outcome.} +- {outcome statement, domain-phrased} — {fit criterion} — {Source: … | [ASSUMPTION — unvalidated]} + +## Tables +{One line per table in `tables/` — required when the objective or intent contains enumerable sets (rote replacements, page/file sets, requirement sets); see references/tabulation.md. Omit the section when there are none.} +- `tables/{set-slug}.md` — {row count} rows, owned by Phase {N} ## Design [High-level solution architecture — major classes/functions/concepts] diff --git a/skills/spec/references/tabulation.md b/skills/spec/references/tabulation.md new file mode 100644 index 0000000..de18c74 --- /dev/null +++ b/skills/spec/references/tabulation.md @@ -0,0 +1,49 @@ +# tabulation — exhaustive enumeration of user-specified sets + +LLMs silently drop enumerated items. Tabulation makes the enumerable sets in a plan explicit, file-backed, and validation-checked, so nothing the user wrote — in intent.md or the objective — can vanish between intent and execution. + +## When a plan MUST tabulate + +1. **Rote replacements** — "replace X with Y", "rename A to B", "update every occurrence of Z", vendoring/repathing changes. One row per occurrence or per file, enumerated from the codebase. +2. **Target sets to search or touch** — "these N pages", "all files matching P", "each module in this list", migration/rollout sets. One row per target, enumerated from the codebase (glob/grep) when the set is derivable. +3. **Enumerated requirement sets** — checklist-style items the user wrote in intent.md (numbered or bulleted lists of concrete things that must be done). One row per item, each carrying an Intent-ref back to the intent.md line it came from. + +If the objective or intent contains any of these and the plan has no table for it, that is a tabulation defect — validation fails the plan. + +## Where tables live + +`$PLAN_DIR/tables/{set-slug}.md` — one file per enumerable set. plan.md lists every table in its `## Tables` section with the row count and the phase that owns it. + +## Table shape + +Header row, then one row per item. Required columns: `#` (row number), `Item`, `Status` (starts `[ ]`; execution marks `[x]`). Set-specific columns as needed: + +- Rote replacements: `File`, `Old`, `New`. +- Target sets: `Path`, `Action`. +- Requirement sets: `Intent ref` (REQUIRED — the intent.md section/line the item comes from). + +```markdown +# Table: {set name} + +Source: {intent.md section | glob/grep command used to enumerate} + +| # | Item | {set-specific columns} | Status | +| --- | --- | --- | --- | +| 1 | {item} | {...} | [ ] | +``` + +## Enumeration rules + +- **Enumerate from the codebase, never from memory.** For rote replacements and target sets, run the real glob/grep and list the actual paths/occurrences found; record the command in the table's `Source:` line. A table written from memory is a tabulation defect. +- **Every enumerated requirement item traces to intent.md.** A requirement row without an Intent ref fails validation; an intent.md item present in no table fails validation. +- **Every row is owned by exactly one task.** Phase tasks that consume a table name it and mark rows `[x]` as they complete them; a row no task references fails validation. +- **Recount at validation.** For codebase-enumerated tables, validation re-runs the Source command; a row-count mismatch is drift and fails the gate (the set changed — re-enumerate). + +## Validation-time checks (SCOPE=tabulation) + +1. Every enumerable set in the objective/intent has a table. +2. Every requirement-set row carries an Intent ref, and every enumerated intent.md item appears in some table. +3. Codebase-enumerated tables match a fresh recount of their Source command. +4. Every table row is referenced by a phase task, and plan.md's `## Tables` section lists every table file. + +Any failure is blocking — the planner repairs the table (or the tasks) before the plan is ready. diff --git a/skills/spec/references/validation-loop.md b/skills/spec/references/validation-loop.md index 0f47ad2..652fe67 100644 --- a/skills/spec/references/validation-loop.md +++ b/skills/spec/references/validation-loop.md @@ -6,7 +6,7 @@ Run a minimum of 2 validation rounds autonomously (no user prompts). After each round that surfaces issues, apply all fixes before running the next round. -Within each round, launch one validation agent **per phase**, one plan-level agent, and one **code-completeness** agent — all in parallel. Merge results before applying fixes. +Within each round, launch one validation agent **per phase**, one plan-level agent, one **code-completeness** agent, and one **tabulation** agent — all in parallel. Merge results before applying fixes. ## Per-Round Execution @@ -51,7 +51,15 @@ prompt: [contents of references/validation-prompt.md with SCOPE=code-completenes It scans every `implementation/phase-*.md` and fails any task whose code block is missing, empty, elided (`...`), a stub, a TODO/placeholder, or a prose-only description instead of the literal code. This is the gate that stops planners from shipping partial snippets. -Store all Task IDs: `[PLAN_LEVEL_TASK_ID, CODE_COMPLETENESS_TASK_ID, PHASE_1_TASK_ID, PHASE_2_TASK_ID, ...]` +**Tabulation agent** — launch one agent (`subagent_type: general-purpose`, `model-tier: light`, `run_in_background: true`): + +``` +prompt: [contents of references/validation-prompt.md with SCOPE=tabulation] +``` + +It checks every enumerable set in the objective/intent against `tables/` per `references/tabulation.md` — missing tables, intent items in no table, drifted codebase recounts, orphan rows. This is the gate that stops planners from silently dropping items the user enumerated. + +Store all Task IDs: `[PLAN_LEVEL_TASK_ID, CODE_COMPLETENESS_TASK_ID, TABULATION_TASK_ID, PHASE_1_TASK_ID, PHASE_2_TASK_ID, ...]` ### c. Collect results @@ -69,6 +77,7 @@ Work through every issue and recommendation from all agents: - Edit the relevant `implementation/phase-N.md` files directly - Rewrite vague plan.md tasks to be specific and actionable - **For every code-completeness failure, replace the placeholder/stub with the complete literal code** — resolve the unknown now (read the codebase, search the web) and paste the real lines; never carry a `...`/`TODO`/prose stub into the next round +- For every tabulation failure, repair the tables — add the missing table or rows, re-enumerate drifted sets from the codebase, add Intent refs, assign orphan rows to tasks, and update plan.md's `## Tables` - Report: `🔧 Round {round} — fixed {N} issues across {M} files: [brief summary]` ### e. Loop control diff --git a/skills/spec/references/validation-prompt.md b/skills/spec/references/validation-prompt.md index 3d2a757..7ebd4e1 100644 --- a/skills/spec/references/validation-prompt.md +++ b/skills/spec/references/validation-prompt.md @@ -30,6 +30,7 @@ Validate the following quality criteria: - Do phase names and task counts in plan.md match what implementation files cover? - Is there a final validation phase (e.g., "Phase N - Testing" or "Phase N - Validation")? - Are inter-phase dependencies called out? +- For doc-aware plans (plan.md carries a `Doc Globs:` line): does every phase file's `## Doc Scope` block justify each boundary crossing with a reason and a rejected restructure (doc-aware Rule 7)? A crossing without a callout is a consistency failure. **Dependencies & Risks** - Are external package/library dependencies noted? @@ -118,6 +119,80 @@ Respond ONLY in this exact format: --- +## Requirements-Quality Agent Prompt (`SCOPE=requirements`) + +``` +You are validating that a software development plan's requirements read as domain/business outcomes, not as restatements of design or implementation. Planners default to deliverable lists and mechanism restatements — your entire job is to catch that. + +Read these files: +1. {PLAN_DIR}/plan.md — the ## Requirements section and the ## Introduction + +Judge against the rule set in the docs skill's requirements-guidance.md (R1–R7) — read it before validating. In particular: + +- Objective framing (BLOCK): is the Requirements section entirely a deliverable list ("ship X", "build Y", "implement Z")? If yes, Status = NEEDS_IMPROVEMENT and the first issue must ask: "What changes for users or the business if this ships successfully?" +- R1: any requirement naming endpoints, classes, files, or other implementation tokens as the requirement itself. +- R2: any requirement whose wording would need to change if the implementation changed. +- R3: any requirement without an observable outcome or measurable success condition. +- R6: any domain claim with neither a Source nor [ASSUMPTION — unvalidated]. + +Judge by intent, not blind substring matching — a token quoted as evidence is not a violation. + +Respond ONLY in this exact format: + +## Validation Report + +### Status: [PASS | NEEDS_IMPROVEMENT] + +### Issues +[requirements, plan.md, R-N] Description of the offending requirement and what domain outcome it should state +(write "none" if every requirement passes) + +### Recommendations +- The exact requirement line to rewrite and a domain-phrased replacement +(write "none" if no recommendations) + +### Missing Details +- Any requirement whose real success condition cannot be determined from the plan +(write "none" if nothing is missing) +``` + +## Tabulation Agent Prompt (`SCOPE=tabulation`) + +``` +You are validating that a software development plan exhaustively tabulates every enumerable set the user specified. LLM planners silently drop enumerated items — your entire job is to catch that. + +Read these files: +1. {PLAN_DIR}/plan.md — the objective and the ## Tables section +2. {PLAN_DIR}/intent.md if it exists (the intent the user wrote or the planner recorded — new.md Step 2 writes it inside the plan dir) — the user's enumerable items +3. Every file in {PLAN_DIR}/tables/ +4. All files in {PLAN_DIR}/implementation/ — to confirm table rows are referenced by tasks + +Validate per references/tabulation.md: + +- Every enumerable set in the objective/intent (rote replacements, target/page sets, enumerated requirement lists) has a table. A set with no table is a failure. +- Every requirement-set row carries an Intent ref, and every enumerated item in intent.md appears in some table row. An intent item in no table is a failure. +- For codebase-enumerated tables, re-run the table's Source command (glob/grep) and compare the row count. A mismatch is drift — a failure. +- Every table row is referenced by exactly one phase task, and plan.md's ## Tables lists every table file. An orphan row or unlisted table is a failure. + +Respond ONLY in this exact format: + +## Validation Report + +### Status: [PASS | NEEDS_IMPROVEMENT] + +### Issues +[tabulation, {table or intent item}] Description of the missing/orphaned/drifted row or set +(write "none" if every enumerable set is fully tabulated) + +### Recommendations +- The exact table file and rows to add or reassign, and the task that should own them +(write "none" if no recommendations) + +### Missing Details +- Any enumerable set whose real members cannot be determined (the planner must enumerate them now, from the codebase) +(write "none" if nothing is missing) +``` + ## Code-Completeness Agent Prompt (`SCOPE=code-completeness`) ``` diff --git a/skills/spec/references/workflows/go.md b/skills/spec/references/workflows/go.md index 05ac920..6c3156b 100644 --- a/skills/spec/references/workflows/go.md +++ b/skills/spec/references/workflows/go.md @@ -109,6 +109,13 @@ After all phases complete, scan `{PLAN_DIR}/execution-log.md` for `[DEVIATION]` ```bash grep "^\[DEVIATION\]" .codevoyant/spec/{plan-name}/execution-log.md 2>/dev/null ``` +and for blocked crossings: + +```bash +grep "^\[BLOCKED\]" .codevoyant/spec/{plan-name}/execution-log.md 2>/dev/null +``` + +Report boundary deviations (deviations that name an out-of-glob target or another module) as a distinct list in the completion report — the run's cross-module footprint — and report any `[BLOCKED]` entry as a run failure that needs `/spec update` before continuing. If any deviations found, include in the final report: diff --git a/skills/spec/references/workflows/new.md b/skills/spec/references/workflows/new.md index 9730449..2a5307f 100644 --- a/skills/spec/references/workflows/new.md +++ b/skills/spec/references/workflows/new.md @@ -97,32 +97,10 @@ WAIT FOR USER decision before proceeding. ## Step 2: Initialize .codevoyant Structure -`cv_init_store` ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`) **before** the first `mkdir` — on a fresh clone this is the first touch, so it must run here or `.codevoyant` would be created as a real directory instead of the shared symlink. It is idempotent, never migrates an existing real dir (that is `/migrate`'s job), and computes the identical `` as the `/migrate` skill so its symlink and `/migrate`'s copy target agree. +Before the first `mkdir`, run the vendored store initializer — it ensures the in-repo `.codevoyant` is a symlink to the shared per-project store (`~/.codevoyant//`), so worktrees and regular clones share one store. Idempotent; never migrates an existing real dir (that is `/migrate`'s job). (`$SPEC_SKILL` is already exported by this skill's SKILL.md.) ```bash -cv_init_store() { - local root; root="$(git rev-parse --show-toplevel 2>/dev/null)" || root=""; [ -n "$root" ] || root="$PWD" - local link="$root/.codevoyant" - [ -L "$link" ] && return 0 # already a symlink → initialized - [ -d "$link" ] && return 0 # old real dir → leave it; /migrate copies it in, never here - local common name slug - common="$(git rev-parse --git-common-dir 2>/dev/null)" || common="" - if [ -n "$common" ]; then - case "$common" in /*) : ;; *) common="$root/$common" ;; esac - name="$(basename "$(cd "$(dirname "$common")" >/dev/null 2>&1 && pwd -P)")" - else - name="$(basename "$root")" - fi - slug="$(printf '%s' "$name" | LC_ALL=C tr '[:upper:]' '[:lower:]' | LC_ALL=C sed 's/[^a-z0-9][^a-z0-9]*/-/g; s/^-*//; s/-*$//')" - [ -n "$slug" ] || slug="unnamed" # empty-slug fallback — matches the /migrate skill - local dest="$HOME/.codevoyant/$slug" - mkdir -p "$dest"; ln -s "$dest" "$link" - local gi="$root/.gitignore" - { [ -f "$gi" ] && grep -qxF '.codevoyant' "$gi"; } || \ - printf '\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n' >> "$gi" -} - -cv_init_store +python3 "$SPEC_SKILL/scripts/cv_init_store.py" >/dev/null mkdir -p .codevoyant/spec .codevoyant/explore if [ ! -f .codevoyant/README.md ]; then printf "# Active Plans\n\n| Name | Status | Plugin | Description | Created | Branch |\n|------|--------|--------|-------------|---------|--------|\n" > .codevoyant/README.md @@ -146,7 +124,7 @@ In **bare-name mode**: 2. `INTENT_FILE` = `.codevoyant/spec/{PLAN_NAME}/intent.md`. 3. **If `INTENT_FILE` exists and is filled in** (content beyond the scaffold — the `## Objective` section is non-empty and not a `{…}` placeholder): read it. Set `OBJECTIVE` from `## Objective`; fold Context / Constraints / Out of scope / Open questions into `RESEARCH_CONTEXT`. Continue to **Step 3b** and plan normally (clarify only if something is still unclear). Do not recreate the file. 4. **Otherwise** (missing, or only the empty scaffold) — scaffold it and **stop**: - a. Ensure the store is initialized, then create the plan dir: run `cv_init_store` (defined in Step 2) before the `mkdir` so a bare-name `/spec new` reached without Step 2 still gets the shared symlink rather than a real `.codevoyant/`. Then `cv_init_store && mkdir -p .codevoyant/spec/{PLAN_NAME}` and write `INTENT_FILE` from `references/intent-template.md` (substitute `{PLAN_NAME}`). + a. Ensure the store is initialized, then create the plan dir: run the vendored initializer (`python3 "$SPEC_SKILL/scripts/cv_init_store.py" >/dev/null`) before the `mkdir` so a bare-name `/spec new` reached without Step 2 still gets the shared symlink rather than a real `.codevoyant/`. Then `python3 "$SPEC_SKILL/scripts/cv_init_store.py" >/dev/null && mkdir -p .codevoyant/spec/{PLAN_NAME}` and write `INTENT_FILE` from `references/intent-template.md` (substitute `{PLAN_NAME}`). b. Print the clickable path: ``` 📝 Tell me what you want built — fill in: @@ -321,7 +299,7 @@ Set `PLAN_WORKTREE=$WORKTREE_RESULT` (used in Step 5.2 and plan metadata). If ne `PLAN_DIR` = `$CHECK_DIR/{plan-name}`. -Create: `$PLAN_DIR/`, `$PLAN_DIR/implementation/`, `$PLAN_DIR/research/`. +Create: `$PLAN_DIR/`, `$PLAN_DIR/implementation/`, `$PLAN_DIR/research/`, `$PLAN_DIR/tables/`. If `RESEARCH_CONTEXT` is set, copy or link the explore artifacts into `$PLAN_DIR/research/` for the execution agent. @@ -384,6 +362,19 @@ Use `references/implementation-template.md`. Move ALL detailed specs here: While drafting each phase, compute the union of its write globs and carry it into `METADATA_DOC_GLOBS` (Step 5.3a). Summarize every boundary callout in the Decision Log under `### Agent Decisions` with a `[boundary]` marker so the crossing is explicitly called out during planning, never silent. +**Cross-module changes are discouraged by default (doc-aware.md Rule 7).** Before writing any boundary callout, attempt to restructure: move the task into the phase that owns the target glob, split the work so each phase writes only its own module, or sequence two phases instead of one phase touching both. Only when restructuring is genuinely worse (duplicated logic, broken atomicity, circular dependency) does the crossing survive — and its callout must carry both the reason and the rejected restructure ("required because …; restructure rejected because …"). A cross-module task whose callout lacks a justification is a planning error: fix it before the plan is written. + +### 5.3d: Tabulate enumerable sets + +Per `references/tabulation.md`, scan `OBJECTIVE`, `RESEARCH_CONTEXT`, and (when present) the intent file for the three trigger classes: rote replacements, target sets to search or touch, and enumerated requirement sets. For each set found: + +1. Create `$PLAN_DIR/tables/{set-slug}.md` with the table shape from `references/tabulation.md`. +2. For codebase-derived sets (rote replacements, target sets), enumerate with the real glob/grep and record the command in the table's `Source:` line — never enumerate from memory. +3. For requirement sets, give every row an Intent ref back to the intent.md item it came from. +4. Assign every row to exactly one phase task (the task references the table), and list every table in plan.md's `## Tables` section with its row count and owning phase. + +If no enumerable set exists, skip this step — do not create empty tables. + ### 5.4: Register Plan ```bash @@ -418,6 +409,10 @@ Immediately after all files are verified, run the code-completeness gate before **Code-completeness gate (required):** Launch one validation agent (`subagent_type: general-purpose`, `model-tier: light`, `run_in_background: true`) with the `SCOPE=code-completeness` prompt from `references/validation-prompt.md` and `{PLAN_DIR}` substituted. +**Requirements gate (required):** In the same message, launch one validation agent (`subagent_type: general-purpose`, `model-tier: light`, `run_in_background: true`) with the `SCOPE=requirements` prompt from `references/validation-prompt.md` and `{PLAN_DIR}` substituted. Collect its report alongside the code-completeness report. If its status is `NEEDS_IMPROVEMENT`, repair plan.md's Requirements section — reframe deliverable bullets as outcomes (ask "What changes for users or the business if this ships successfully?" when the objective is a deliverable list), rewrite R1/R2 violations into domain phrasing, add missing fit criteria and Source/`[ASSUMPTION — unvalidated]` markers — then rerun the gate until it returns `PASS`. Never continue to permission analysis while this gate fails. + +**Tabulation gate (required):** In the same message, launch one validation agent (`subagent_type: general-purpose`, `model-tier: light`, `run_in_background: true`) with the `SCOPE=tabulation` prompt from `references/validation-prompt.md` and `{PLAN_DIR}` substituted. If its status is `NEEDS_IMPROVEMENT`, repair the tables — add missing tables, re-enumerate drifted sets from the codebase, add Intent refs, assign orphan rows to tasks, update plan.md's `## Tables` — then rerun the gate until it returns `PASS`. Never continue to permission analysis while this gate fails. + Collect its report with `TaskOutput(id: CODE_COMPLETENESS_TASK_ID, block: true)`. If its status is `NEEDS_IMPROVEMENT`, repair every reported implementation task by replacing the missing, abbreviated, placeholder, or prose-only block with the complete literal code. Rerun this gate after each repair pass until it returns `PASS`. Never continue to permission analysis, optional validation, or the completion report while this gate fails. If the planner cannot determine the literal code required to repair a task, stop and report that the plan is not ready; do not report a successful plan. diff --git a/skills/spec/references/workflows/review.md b/skills/spec/references/workflows/review.md index 3a8e314..aa3e429 100644 --- a/skills/spec/references/workflows/review.md +++ b/skills/spec/references/workflows/review.md @@ -39,9 +39,11 @@ Additional checks: ## Step 3: Code-Completeness Gate (Pass 1 — CRITICAL) -Before launching scope, ordering, or codebase-alignment review agents, launch one code-completeness agent (`model-tier: light`, `run_in_background: true`) with the `SCOPE=code-completeness` prompt from `references/validation-prompt.md` and `{PLAN_DIR}` substituted. The agent must read `references/code-completeness-blocklist.md` and inspect every implementation task for a complete literal `**Code:**` block. +Before launching scope, ordering, or codebase-alignment review agents, launch one code-completeness agent (`model-tier: light`, `run_in_background: true`) with the `SCOPE=code-completeness` prompt from `references/validation-prompt.md` and `{PLAN_DIR}` substituted, and — in the same message — one requirements agent with the `SCOPE=requirements` prompt. The code-completeness agent must read `references/code-completeness-blocklist.md` and inspect every implementation task for a complete literal `**Code:**` block; the requirements agent judges plan.md's Requirements section against R1–R7 (see `references/validation-prompt.md`). -Wait for the report. Add every `NEEDS_IMPROVEMENT` finding to the critical finding set. Classify a finding as `AUTO-FIX` only when the reviewer can determine and paste the complete literal code from the repository and plan context; otherwise classify it as `ASK`. Do not allow later review findings, AUTO-FIX work, or a report verdict to mark the plan ready until this gate returns `PASS` with no unresolved code-completeness findings. +Wait for both reports. Add every `NEEDS_IMPROVEMENT` finding to the critical finding set. Classify a finding as `AUTO-FIX` only when the reviewer can determine and paste the complete literal code (code-completeness) or the domain-phrased requirement rewrite (requirements) from the repository and plan context; otherwise classify it as `ASK`. Do not allow later review findings, AUTO-FIX work, or a report verdict to mark the plan ready until both gates return `PASS` with no unresolved findings. + +Launch the tabulation gate in the same message as the other two (`SCOPE=tabulation` prompt from `references/validation-prompt.md`). Treat its failures like code-completeness failures: `AUTO-FIX` when the reviewer can enumerate the missing rows from the codebase (re-run the table's Source command, add the rows, assign them to tasks, update plan.md `## Tables`), otherwise `ASK`. The plan is not ready while the tabulation gate has unresolved findings. ## Step 4: Parallel Review Agents (Pass 2 — CRITICAL) @@ -53,6 +55,7 @@ Run four review agents in parallel (`model-tier: light`, `run_in_background: tru - Hero systems: flag if plan reaches for new technology when existing utility would do. - Structural issues: objective clarity, phase ordering, phase headers, meta-tasks, design decisions section. - "What Already Exists" callout: codebase mechanisms this plan should leverage. +- Boundary audit (doc-aware plans — the plan.md carries a `Doc Globs:` line): for every task that writes outside its phase's declared globs or touches a module another phase owns, verify a `## Doc Scope` boundary callout exists with a justification and a rejected restructure. A crossing with no callout, or a callout with no justification, is CRITICAL (doc-aware.md Rule 7). Also flag crossings that a phase restructure could have avoided — move the task, split the phases, or sequence them. **Agent B — Implementation completeness after the code gate:** For each phase-N.md, flag as CRITICAL if: diff --git a/skills/spec/scripts/cv_init_store.py b/skills/spec/scripts/cv_init_store.py new file mode 100644 index 0000000..00bb73c --- /dev/null +++ b/skills/spec/scripts/cv_init_store.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Canonical codevoyant store initializer — shared asset, vendored per skills/vendor.json. + +Ensures the in-repo `.codevoyant` is a symlink to the shared per-project store +(~/.codevoyant//) BEFORE any workflow mkdirs under it, so worktrees +and regular clones of the same project share one store. + +Idempotent. Never migrates an existing real dir (that is /migrate's job). +The slug pipeline is byte-identical to the /migrate skill's computation +(worktree-aware via `git rev-parse --git-common-dir`). + +Usage: cv_init_store.py [repo-root] (defaults to the git toplevel, else cwd) +Prints the computed slug on stdout. Exit 0 in all handled cases. +""" +import os +import re +import subprocess +import sys +from pathlib import Path + + +def _git(args, root): + try: + proc = subprocess.run( + ["git", "-C", str(root), *args], + capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return proc.stdout.strip() if proc.returncode == 0 else "" + + +def compute_slug(root): + """Same pipeline as the bash original: lower, [^a-z0-9]+ -> '-', strip '-'.""" + common = _git(["rev-parse", "--git-common-dir"], root) + if common: + cpath = Path(common) + if not cpath.is_absolute(): + cpath = root / cpath + try: + name = cpath.parent.resolve().name + except OSError: + name = root.name + else: + name = root.name + slug = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") + return slug or "unnamed" + + +def init_store(root, home=None): + """Create the ~/.codevoyant/ symlink for `root`; return the slug. + + - Already a symlink -> no-op. + - Existing real dir -> no-op (migration is /migrate's job). + - Otherwise -> mkdir dest, symlink, append .gitignore entry once. + """ + root = Path(root) + home = Path(home) if home else Path(os.environ["HOME"]) + slug = compute_slug(root) + link = root / ".codevoyant" + if link.is_symlink(): + return slug + if link.is_dir(): + return slug + dest = home / ".codevoyant" / slug + dest.mkdir(parents=True, exist_ok=True) + link.symlink_to(dest) + gi = root / ".gitignore" + lines = gi.read_text(encoding="utf-8").splitlines() if gi.exists() else [] + if ".codevoyant" not in lines: + with gi.open("a", encoding="utf-8") as f: + f.write("\n# codevoyant context store (symlink to ~/.codevoyant//)\n.codevoyant\n") + return slug + + +def main(argv): + if len(argv) > 1: + root = Path(argv[1]) + else: + top = _git(["rev-parse", "--show-toplevel"], Path.cwd()) + root = Path(top) if top else Path.cwd() + print(init_store(root)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv)) diff --git a/skills/vendor.json b/skills/vendor.json index a3e6a13..714529f 100644 --- a/skills/vendor.json +++ b/skills/vendor.json @@ -1,6 +1,12 @@ { "version": 2, "assets": { + "store-init": { + "source": "skills/shared/store-init", + "files": ["cv_init_store.py"], + "skills": ["docs", "explore", "flow", "loop", "plan", "pm", "pr", "qa", "skill", "spec"], + "destination": "scripts" + }, "simple-english": { "source": "skills/shared/simple-english", "skills": ["docs", "pr"], diff --git a/skills/vendor.manifest.json b/skills/vendor.manifest.json index b25fb07..99734bb 100644 --- a/skills/vendor.manifest.json +++ b/skills/vendor.manifest.json @@ -1,19 +1,49 @@ { - "skills/docs/scripts": [ - "scope.py", - "test_scope.py" - ], - "skills/gh/references/templates": [ + "bug-report::skills/gh/references/templates": [ "bug-report.md" ], - "skills/glab/references/templates": [ + "bug-report::skills/glab/references/templates": [ "bug-report.md" ], - "skills/linear/references/templates": [ + "bug-report::skills/linear/references/templates": [ "bug-report.md" ], - "skills/spec/scripts": [ + "scope-scripts::skills/docs/scripts": [ + "scope.py", + "test_scope.py" + ], + "scope-scripts::skills/spec/scripts": [ "scope.py", "test_scope.py" + ], + "store-init::skills/docs/scripts": [ + "cv_init_store.py" + ], + "store-init::skills/explore/scripts": [ + "cv_init_store.py" + ], + "store-init::skills/flow/scripts": [ + "cv_init_store.py" + ], + "store-init::skills/loop/scripts": [ + "cv_init_store.py" + ], + "store-init::skills/plan/scripts": [ + "cv_init_store.py" + ], + "store-init::skills/pm/scripts": [ + "cv_init_store.py" + ], + "store-init::skills/pr/scripts": [ + "cv_init_store.py" + ], + "store-init::skills/qa/scripts": [ + "cv_init_store.py" + ], + "store-init::skills/skill/scripts": [ + "cv_init_store.py" + ], + "store-init::skills/spec/scripts": [ + "cv_init_store.py" ] }