diff --git a/.github/workflows/localization-quality-evaluation.yml b/.github/workflows/localization-quality-evaluation.yml new file mode 100644 index 00000000..06858f24 --- /dev/null +++ b/.github/workflows/localization-quality-evaluation.yml @@ -0,0 +1,223 @@ +name: Localization Quality Evaluation + +on: + workflow_dispatch: + inputs: + source_run_id: + description: Cross-trace run ID whose localization artifacts should be evaluated + required: true + default: "35869966400" + type: string + +permissions: + contents: read + actions: read + +env: + SOURCE_RUN_ID: ${{ inputs.source_run_id || '35869966400' }} + +jobs: + evaluate: + name: Blind Claude localization quality review + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + env: + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + ANTHROPIC_BASE_URL: ${{ vars.ANTHROPIC_BASE_URL || 'https://api.anthropic.com' }} + steps: + - name: Checkout evaluator code + uses: actions/checkout@v4 + with: + path: narness + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Install pinned Claude Code + run: | + set -euo pipefail + npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + claude --version + + - name: Download frozen localization artifacts + env: + GH_TOKEN: ${{ github.token }} + run: | + set -euo pipefail + mkdir -p artifacts/system-one artifacts/claude-reference + gh run download "$SOURCE_RUN_ID" \ + --repo "$GITHUB_REPOSITORY" \ + --name "code-locator-cross-trace-system-one-$SOURCE_RUN_ID" \ + --dir artifacts/system-one + gh run download "$SOURCE_RUN_ID" \ + --repo "$GITHUB_REPOSITORY" \ + --name "code-locator-cross-trace-claude-$SOURCE_RUN_ID" \ + --dir artifacts/claude-reference + + - name: Resolve frozen subject revision + id: source + run: | + set -euo pipefail + python3 - <<'PY' >> "$GITHUB_OUTPUT" + import json + from pathlib import Path + manifest=json.loads(Path("artifacts/system-one/manifest.json").read_text()) + print(f"repo={manifest['subject_repository']}") + print(f"sha={manifest['subject_sha']}") + print(f"query={manifest['query']}") + PY + + - name: Checkout exact subject revision + uses: actions/checkout@v4 + with: + repository: ${{ steps.source.outputs.repo }} + ref: ${{ steps.source.outputs.sha }} + path: subject + fetch-depth: 1 + persist-credentials: false + + - name: Require evaluator configuration + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + run: | + set -euo pipefail + test -n "$ANTHROPIC_API_KEY" + test -n "$ANTHROPIC_BASE_URL" + test -n "$CLAUDE_MODEL" + + - name: Prepare anonymous candidates + run: | + set -euo pipefail + mkdir -p artifacts/quality-evaluation/candidate-a + mkdir -p artifacts/quality-evaluation/candidate-b + + python3 narness/research/code-locator/src/localization_quality_evaluation.py prepare \ + --input artifacts/system-one/localization-result.json \ + --candidate-id candidate-a \ + --output-candidate artifacts/quality-evaluation/candidate-a/candidate.json \ + --output-prompt artifacts/quality-evaluation/candidate-a/prompt.txt + + python3 narness/research/code-locator/src/localization_quality_evaluation.py prepare \ + --input artifacts/claude-reference/localization-result.json \ + --candidate-id candidate-b \ + --output-candidate artifacts/quality-evaluation/candidate-b/candidate.json \ + --output-prompt artifacts/quality-evaluation/candidate-b/prompt.txt + + - name: Evaluate anonymous candidate A + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json \ + --verbose \ + --model "$CLAUDE_MODEL" \ + --effort high \ + --max-turns 80 \ + --permission-mode dontAsk \ + --no-session-persistence \ + --setting-sources '' \ + --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' \ + --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality-evaluation/candidate-a/prompt.txt \ + > ../artifacts/quality-evaluation/candidate-a/evaluator.raw.jsonl + + - name: Evaluate anonymous candidate B + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json \ + --verbose \ + --model "$CLAUDE_MODEL" \ + --effort high \ + --max-turns 80 \ + --permission-mode dontAsk \ + --no-session-persistence \ + --setting-sources '' \ + --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' \ + --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality-evaluation/candidate-b/prompt.txt \ + > ../artifacts/quality-evaluation/candidate-b/evaluator.raw.jsonl + + - name: Build scorecards and report + run: | + set -euo pipefail + python3 narness/research/code-locator/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality-evaluation/candidate-a/candidate.json \ + --raw-jsonl artifacts/quality-evaluation/candidate-a/evaluator.raw.jsonl \ + --subject-root subject \ + --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality-evaluation/candidate-a/scorecard.json \ + --output-markdown artifacts/quality-evaluation/candidate-a/report.md + + python3 narness/research/code-locator/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality-evaluation/candidate-b/candidate.json \ + --raw-jsonl artifacts/quality-evaluation/candidate-b/evaluator.raw.jsonl \ + --subject-root subject \ + --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality-evaluation/candidate-b/scorecard.json \ + --output-markdown artifacts/quality-evaluation/candidate-b/report.md + + python3 narness/research/code-locator/src/localization_quality_evaluation.py report \ + --candidate-a artifacts/quality-evaluation/candidate-a/scorecard.json \ + --candidate-b artifacts/quality-evaluation/candidate-b/scorecard.json \ + --candidate-a-label "System One" \ + --candidate-b-label "Claude Code" \ + --output-json artifacts/quality-evaluation/evaluation-report.json \ + --output-markdown artifacts/quality-evaluation/evaluation-report.md + + python3 - <<'PY' + import json + import os + from pathlib import Path + root=Path("artifacts/quality-evaluation") + payload={ + "schema_version": 1, + "kind": "code-localization-quality-evaluation-run", + "source_run_id": os.environ["SOURCE_RUN_ID"], + "subject_repository": "${{ steps.source.outputs.repo }}", + "subject_sha": "${{ steps.source.outputs.sha }}", + "query": "${{ steps.source.outputs.query }}", + "evaluator_model": os.environ["CLAUDE_MODEL"], + "candidate_mapping": { + "candidate-a": "System One", + "candidate-b": "Claude Code", + }, + } + (root/"manifest.json").write_text( + json.dumps(payload, indent=2, ensure_ascii=False)+"\n" + ) + PY + + cat artifacts/quality-evaluation/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload quality evaluation + if: always() + uses: actions/upload-artifact@v4 + with: + name: code-locator-quality-evaluation-${{ github.run_id }} + path: artifacts/quality-evaluation/ + if-no-files-found: warn + retention-days: 90 diff --git a/.github/workflows/system-one-code-locator-accuracy.yml b/.github/workflows/system-one-code-locator-accuracy.yml new file mode 100644 index 00000000..1135908b --- /dev/null +++ b/.github/workflows/system-one-code-locator-accuracy.yml @@ -0,0 +1,663 @@ +name: System One / Claude Code Cross-Trace Record + +on: + workflow_dispatch: + inputs: + subject_repository: + description: Subject repository + required: true + default: BestNathan/nession + type: string + subject_ref: + description: Subject ref resolved once and shared by both agents + required: true + default: staging + type: string + query: + description: Verbatim localization task given to both systems + required: true + default: Help me optimize the websocket connection implementation + type: string + +permissions: + contents: read + +env: + QUERY: ${{ inputs.query || 'Help me optimize the websocket connection implementation' }} + SUBJECT_REPOSITORY: ${{ inputs.subject_repository || 'BestNathan/nession' }} + SUBJECT_REF: ${{ inputs.subject_ref || 'staging' }} + +jobs: + resolve-subject: + name: Resolve shared subject revision + runs-on: ubuntu-24.04 + outputs: + subject_sha: ${{ steps.revision.outputs.sha }} + steps: + - name: Checkout subject + uses: actions/checkout@v4 + with: + repository: ${{ env.SUBJECT_REPOSITORY }} + ref: ${{ env.SUBJECT_REF }} + fetch-depth: 1 + persist-credentials: false + + - name: Resolve exact revision + id: revision + run: echo "sha=$(git rev-parse HEAD)" >> "$GITHUB_OUTPUT" + + system-one: + name: System One progressive reader + needs: resolve-subject + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + env: + TYPESAFE_MODEL: ${{ vars.TYPESAFE_MODEL || 'jev-latest' }} + steps: + - name: Checkout Narness harness + uses: actions/checkout@v4 + with: + path: narness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout exact subject revision + uses: actions/checkout@v4 + with: + repository: ${{ env.SUBJECT_REPOSITORY }} + ref: ${{ needs.resolve-subject.outputs.subject_sha }} + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Run System One locator + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + TYPESAFE_API_URL: ${{ vars.TYPESAFE_API_URL || 'https://api.typesafe.ai/v1/systemone' }} + run: | + set -euo pipefail + test -n "$TYPESAFE_API_KEY" + mkdir -p artifacts/system-one + python3 narness/research/code-locator/src/system_one_range_runtime.py \ + subject \ + "$QUERY" \ + --directory-threshold 0.50 \ + --file-threshold 0.65 \ + --window-lines 140 \ + --parallel-threshold 0.65 \ + --evidence-threshold 0.65 \ + --max-jumps 2 \ + --max-file-epochs 32 \ + --subject-repository "$SUBJECT_REPOSITORY" \ + --subject-revision "${{ needs.resolve-subject.outputs.subject_sha }}" \ + --model "$TYPESAFE_MODEL" \ + --trace-file artifacts/system-one/trace.jsonl \ + --output-json artifacts/system-one/result.json \ + --output-localization-json artifacts/system-one/localization-result.json + python3 - <<'PY' + import json, os + from pathlib import Path + root=Path("artifacts/system-one") + result=json.loads((root/"result.json").read_text()) + manifest={ + "query": os.environ["QUERY"], + "subject_repository": os.environ["SUBJECT_REPOSITORY"], + "subject_sha": "${{ needs.resolve-subject.outputs.subject_sha }}", + "model": os.environ["TYPESAFE_MODEL"], + "architecture": result.get("architecture"), + "thresholds": result.get("thresholds"), + "metrics": result.get("metrics"), + } + (root/"manifest.json").write_text(json.dumps(manifest, indent=2)+"\n") + PY + + - name: Upload System One evidence + uses: actions/upload-artifact@v4 + with: + name: code-locator-cross-trace-system-one-${{ github.run_id }} + path: artifacts/system-one/ + if-no-files-found: error + retention-days: 90 + + claude-reference: + name: Claude Code System 2 trace + needs: resolve-subject + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + env: + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + ANTHROPIC_BASE_URL: ${{ vars.ANTHROPIC_BASE_URL || 'https://api.anthropic.com' }} + steps: + - name: Checkout Narness harness + uses: actions/checkout@v4 + with: + path: narness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout exact subject revision + uses: actions/checkout@v4 + with: + repository: ${{ env.SUBJECT_REPOSITORY }} + ref: ${{ needs.resolve-subject.outputs.subject_sha }} + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Install pinned Claude Code + run: | + set -euo pipefail + npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + claude --version + + - name: Require DS Claude configuration + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + run: | + set -euo pipefail + test -n "$ANTHROPIC_API_KEY" + test -n "$ANTHROPIC_BASE_URL" + test -n "$CLAUDE_MODEL" + case "$CLAUDE_MODEL" in + sonnet|opus|haiku|default) + echo "Use an explicit CLAUDE_MODEL in ds, not a moving alias." >&2 + exit 2 + ;; + esac + + - name: Build verbatim-task localization prompt + run: | + set -euo pipefail + mkdir -p artifacts/claude-reference + python3 - <<'PY' + import os + from pathlib import Path + query=os.environ["QUERY"] + prompt=f"""User task (verbatim): + {query} + + Perform an independent System 2 code-localization run for this task. + Your only job in this session is to decide the final relevant files + and the exact source ranges that matter. + + Investigate this repository using Read, Glob, Grep, and read-only Bash commands. + Do not modify any file. Do not use git history, GitHub issues, the network, or external context. + Search and read enough source code to identify the files and exact code regions that materially matter to the task. + + IMPORTANT: + - Do NOT calculate, estimate, label, or output confidence in this session. + - Do NOT include any JSON key named "confidence". + - This session decides localization only. Confidence will be analyzed later by a completely fresh session. + + Your final answer MUST be only valid JSON, with no Markdown fences, in this schema: + {{ + "schema_version": 1, + "kind": "code-localization-draft", + "task": {query!r}, + "producer": {{ + "system": "claude_code", + "model": "{os.environ['CLAUDE_MODEL']}" + }}, + "summary": "short description of the implementation areas that matter", + "files": [ + {{ + "path": "repository-relative/path", + "role": "primary|supporting|context", + "reason": "why this file is valuable to the task", + "evidence": [ + {{ + "start_line": 1, + "end_line": 20, + "reason": "what valuable implementation is in this range" + }} + ] + }} + ] + }} + + Definitions: + - primary: directly implements the requested behavior or its core lifecycle. + - supporting: a direct dependency/caller/state component needed to understand the primary behavior. + - context: useful surrounding wiring, but not core implementation. + + Requirements: + - Rank files from most to least relevant. + - Include only repository-relative file paths. + - Include files only after inspecting their contents. + - Ground each primary/supporting file with at least one concrete line range. + - Prefer precision over listing every file whose name contains related words. + - Do not propose edits; this run is localization only. + """ + Path("artifacts/claude-reference/localization-prompt.txt").write_text(prompt) + PY + + - name: Run Claude localization session with raw execution trace + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json \ + --verbose \ + --model "$CLAUDE_MODEL" \ + --effort high \ + --max-turns 100 \ + --permission-mode dontAsk \ + --no-session-persistence \ + --setting-sources '' \ + --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' \ + --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/claude-reference/localization-prompt.txt \ + > ../artifacts/claude-reference/localization.raw.jsonl + + - name: Normalize localization-only draft and execution path + run: | + set -euo pipefail + python3 narness/research/code-locator/src/claude_reference_trace.py \ + --raw-jsonl artifacts/claude-reference/localization.raw.jsonl \ + --subject-root subject \ + --query "$QUERY" \ + --model "$CLAUDE_MODEL" \ + --subject-sha "${{ needs.resolve-subject.outputs.subject_sha }}" \ + --output-draft artifacts/claude-reference/localization-draft.json \ + --output-trace artifacts/claude-reference/execution-path.json \ + --output-manifest artifacts/claude-reference/localization-manifest.json \ + --output-summary artifacts/claude-reference/execution-summary.md + + - name: Build fresh-session confidence prompt + run: | + set -euo pipefail + python3 - <<'PY' + import json + import os + from pathlib import Path + + draft=json.loads( + Path("artifacts/claude-reference/localization-draft.json").read_text() + ) + query=os.environ["QUERY"] + immutable=json.dumps(draft, ensure_ascii=False, indent=2) + + prompt=f"""User task (verbatim): + {query} + + You are a NEW, independent confidence-assessment session. + + The previous session has already completed code localization. + You must NOT search the repository and must NOT change its localization result. + The immutable draft below already contains the final files, roles, reasons, + evidence ranges, and exact source text. + + Your only job is to assess confidence in that fixed result. + + You MUST preserve: + - exact file count, order, and paths; + - exact evidence count, order, start_line, and end_line; + - the previous session's localization decisions. + + Return only valid JSON, with no Markdown fences: + {{ + "schema_version": 1, + "kind": "code-localization-confidence-assessment", + "task": {query!r}, + "overall": {{ + "score": 0.0, + "reason": "why this overall confidence is appropriate" + }}, + "files": [ + {{ + "path": "must exactly match the corresponding draft file", + "score": 0.0, + "reason": "confidence rationale for this file", + "evidence": [ + {{ + "start_line": 1, + "end_line": 20, + "score": 0.0, + "reason": "confidence rationale for this evidence range" + }} + ] + }} + ] + }} + + Scores are self-assessments from 0 to 1, not calibrated probabilities. + Do not add or remove files. Do not add or remove evidence ranges. + Do not change paths or line numbers. Do not perform localization again. + The top-level "files" array is REQUIRED even when you are uncertain. + It must contain exactly {len(draft.get("files", []))} entries, one for + every draft file in the same order. Never return only "overall". + + IMMUTABLE LOCALIZATION DRAFT: + {immutable} + """ + Path("artifacts/claude-reference/confidence-prompt.txt").write_text(prompt) + PY + + - name: Run fresh Claude confidence session without tools + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json \ + --verbose \ + --model "$CLAUDE_MODEL" \ + --effort high \ + --max-turns 20 \ + --permission-mode dontAsk \ + --no-session-persistence \ + --setting-sources '' \ + --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' \ + --disable-slash-commands \ + --tools '' \ + --disallowedTools 'Bash' 'Read' 'Glob' 'Grep' 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/claude-reference/confidence-prompt.txt \ + > ../artifacts/claude-reference/confidence.raw.jsonl + + - name: Merge immutable localization with confidence assessment + run: | + set -euo pipefail + python3 narness/research/code-locator/src/claude_confidence_finalize.py \ + --draft artifacts/claude-reference/localization-draft.json \ + --confidence-raw-jsonl artifacts/claude-reference/confidence.raw.jsonl \ + --localization-manifest artifacts/claude-reference/localization-manifest.json \ + --model "$CLAUDE_MODEL" \ + --subject-sha "${{ needs.resolve-subject.outputs.subject_sha }}" \ + --subject-repository "$SUBJECT_REPOSITORY" \ + --output-result artifacts/claude-reference/localization-result.json \ + --output-confidence-manifest artifacts/claude-reference/confidence-manifest.json \ + --output-manifest artifacts/claude-reference/manifest.json \ + --output-summary artifacts/claude-reference/confidence-summary.md + + - name: Upload Claude trace and result + if: always() + uses: actions/upload-artifact@v4 + with: + name: code-locator-cross-trace-claude-${{ github.run_id }} + path: artifacts/claude-reference/ + if-no-files-found: error + retention-days: 90 + + compare: + name: Compare localization records + needs: + - resolve-subject + - system-one + - claude-reference + runs-on: ubuntu-24.04 + steps: + - name: Checkout comparison code + uses: actions/checkout@v4 + with: + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Download System One evidence + uses: actions/download-artifact@v4 + with: + name: code-locator-cross-trace-system-one-${{ github.run_id }} + path: artifacts/system-one + + - name: Download Claude trace and result + uses: actions/download-artifact@v4 + with: + name: code-locator-cross-trace-claude-${{ github.run_id }} + path: artifacts/claude-reference + + - name: Compare file and range localization + run: | + set -euo pipefail + mkdir -p artifacts/comparison + python3 research/code-locator/src/compare_localization_results.py \ + --left artifacts/system-one/localization-result.json \ + --right artifacts/claude-reference/localization-result.json \ + --output-json artifacts/comparison/comparison.json \ + --output-markdown artifacts/comparison/summary.md + cat artifacts/comparison/summary.md >> "$GITHUB_STEP_SUMMARY" + if test -f artifacts/claude-reference/execution-summary.md; then + printf '\n' >> "$GITHUB_STEP_SUMMARY" + cat artifacts/claude-reference/execution-summary.md >> "$GITHUB_STEP_SUMMARY" + fi + if test -f artifacts/claude-reference/confidence-summary.md; then + printf '\n' >> "$GITHUB_STEP_SUMMARY" + cat artifacts/claude-reference/confidence-summary.md >> "$GITHUB_STEP_SUMMARY" + fi + + python3 - <<'PY' + import json + from pathlib import Path + payload={ + "subject_sha": "${{ needs.resolve-subject.outputs.subject_sha }}", + "system_one": json.loads(Path("artifacts/system-one/manifest.json").read_text()), + "claude": json.loads(Path("artifacts/claude-reference/manifest.json").read_text()), + } + Path("artifacts/comparison/run-manifest.json").write_text( + json.dumps(payload, indent=2, ensure_ascii=False)+"\n" + ) + PY + + - name: Upload accuracy comparison + uses: actions/upload-artifact@v4 + with: + name: code-locator-cross-trace-comparison-${{ github.run_id }} + path: artifacts/comparison/ + if-no-files-found: error + retention-days: 90 + + quality-evaluate: + name: Blind Claude localization quality review + needs: + - resolve-subject + - system-one + - claude-reference + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + env: + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + ANTHROPIC_BASE_URL: ${{ vars.ANTHROPIC_BASE_URL || 'https://api.anthropic.com' }} + steps: + - name: Checkout evaluator code + uses: actions/checkout@v4 + with: + path: narness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout exact subject revision + uses: actions/checkout@v4 + with: + repository: ${{ env.SUBJECT_REPOSITORY }} + ref: ${{ needs.resolve-subject.outputs.subject_sha }} + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Install pinned Claude Code + run: | + set -euo pipefail + npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + claude --version + + - name: Require evaluator configuration + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + run: | + set -euo pipefail + test -n "$ANTHROPIC_API_KEY" + test -n "$ANTHROPIC_BASE_URL" + test -n "$CLAUDE_MODEL" + + - name: Download System One localization result + uses: actions/download-artifact@v4 + with: + name: code-locator-cross-trace-system-one-${{ github.run_id }} + path: artifacts/system-one + + - name: Download Claude localization result + uses: actions/download-artifact@v4 + with: + name: code-locator-cross-trace-claude-${{ github.run_id }} + path: artifacts/claude-reference + + - name: Prepare anonymous candidates and identical rubrics + run: | + set -euo pipefail + mkdir -p artifacts/quality-evaluation/candidate-a + mkdir -p artifacts/quality-evaluation/candidate-b + + python3 narness/research/code-locator/src/localization_quality_evaluation.py prepare \ + --input artifacts/system-one/localization-result.json \ + --candidate-id candidate-a \ + --output-candidate artifacts/quality-evaluation/candidate-a/candidate.json \ + --output-prompt artifacts/quality-evaluation/candidate-a/prompt.txt + + python3 narness/research/code-locator/src/localization_quality_evaluation.py prepare \ + --input artifacts/claude-reference/localization-result.json \ + --candidate-id candidate-b \ + --output-candidate artifacts/quality-evaluation/candidate-b/candidate.json \ + --output-prompt artifacts/quality-evaluation/candidate-b/prompt.txt + + - name: Evaluate anonymous candidate A in fresh Claude session + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json \ + --verbose \ + --model "$CLAUDE_MODEL" \ + --effort high \ + --max-turns 80 \ + --permission-mode dontAsk \ + --no-session-persistence \ + --setting-sources '' \ + --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' \ + --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality-evaluation/candidate-a/prompt.txt \ + > ../artifacts/quality-evaluation/candidate-a/evaluator.raw.jsonl + + - name: Evaluate anonymous candidate B in fresh Claude session + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json \ + --verbose \ + --model "$CLAUDE_MODEL" \ + --effort high \ + --max-turns 80 \ + --permission-mode dontAsk \ + --no-session-persistence \ + --setting-sources '' \ + --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' \ + --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality-evaluation/candidate-b/prompt.txt \ + > ../artifacts/quality-evaluation/candidate-b/evaluator.raw.jsonl + + - name: Finalize scorecards and combined report + run: | + set -euo pipefail + + python3 narness/research/code-locator/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality-evaluation/candidate-a/candidate.json \ + --raw-jsonl artifacts/quality-evaluation/candidate-a/evaluator.raw.jsonl \ + --subject-root subject \ + --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality-evaluation/candidate-a/scorecard.json \ + --output-markdown artifacts/quality-evaluation/candidate-a/report.md + + python3 narness/research/code-locator/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality-evaluation/candidate-b/candidate.json \ + --raw-jsonl artifacts/quality-evaluation/candidate-b/evaluator.raw.jsonl \ + --subject-root subject \ + --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality-evaluation/candidate-b/scorecard.json \ + --output-markdown artifacts/quality-evaluation/candidate-b/report.md + + python3 narness/research/code-locator/src/localization_quality_evaluation.py report \ + --candidate-a artifacts/quality-evaluation/candidate-a/scorecard.json \ + --candidate-b artifacts/quality-evaluation/candidate-b/scorecard.json \ + --candidate-a-label "System One" \ + --candidate-b-label "Claude Code" \ + --output-json artifacts/quality-evaluation/evaluation-report.json \ + --output-markdown artifacts/quality-evaluation/evaluation-report.md + + python3 - <<'PY' + import json + from pathlib import Path + root = Path("artifacts/quality-evaluation") + manifest = { + "schema_version": 1, + "kind": "code-localization-quality-evaluation-run", + "subject_sha": "${{ needs.resolve-subject.outputs.subject_sha }}", + "task": "${{ env.QUERY }}", + "evaluator_model": "${{ env.CLAUDE_MODEL }}", + "candidate_mapping": { + "candidate-a": "System One", + "candidate-b": "Claude Code", + }, + } + (root / "manifest.json").write_text( + json.dumps(manifest, indent=2, ensure_ascii=False) + "\n" + ) + PY + + cat artifacts/quality-evaluation/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload blind quality evaluation + if: always() + uses: actions/upload-artifact@v4 + with: + name: code-locator-quality-evaluation-${{ github.run_id }} + path: artifacts/quality-evaluation/ + if-no-files-found: error + retention-days: 90 diff --git a/.github/workflows/system-one-code-locator.yml b/.github/workflows/system-one-code-locator.yml index c6c00577..13a7abe0 100644 --- a/.github/workflows/system-one-code-locator.yml +++ b/.github/workflows/system-one-code-locator.yml @@ -35,24 +35,44 @@ on: default: Help me optimize the websocket connection implementation type: string directory_threshold: - description: Directory Noul threshold + description: Phase-1 directory Noul threshold required: true - default: "0.35" + default: "0.50" type: string file_threshold: - description: File Noul threshold + description: Phase-1 file Noul threshold required: true - default: "0.50" + default: "0.65" + type: string + reader_file_activation_threshold: + description: Global Phase-2 file activation Noul threshold + required: true + default: "0.65" + type: string + reader_window_lines: + description: Lines read by one ReadRange action + required: true + default: "140" + type: string + reader_soft_reads: + description: Soft per-file read budget; high-relevance evidence can extend beyond it + required: true + default: "4" + type: string + reader_hard_reads: + description: Absolute maximum reads per file + required: true + default: "8" type: string - line_threshold: - description: Source-line Noul threshold + reader_action_threshold: + description: Minimum chosen Choice probability required to execute a read required: true - default: "0.70" + default: "0.40" type: string - batch_size: - description: Noul questions per TypeSafe request + observation_threshold: + description: Noul threshold for retaining an observation as evidence required: true - default: "48" + default: "0.65" type: string model: description: TypeSafe model alias @@ -74,10 +94,14 @@ concurrency: env: QUERY: ${{ inputs.query || 'Help me optimize the websocket connection implementation' }} - DIRECTORY_THRESHOLD: ${{ inputs.directory_threshold || '0.35' }} - FILE_THRESHOLD: ${{ inputs.file_threshold || '0.50' }} - LINE_THRESHOLD: ${{ inputs.line_threshold || '0.70' }} - BATCH_SIZE: ${{ inputs.batch_size || '48' }} + DIRECTORY_THRESHOLD: ${{ inputs.directory_threshold || '0.50' }} + FILE_THRESHOLD: ${{ inputs.file_threshold || '0.65' }} + READER_FILE_ACTIVATION_THRESHOLD: ${{ inputs.reader_file_activation_threshold || '0.65' }} + READER_WINDOW_LINES: ${{ inputs.reader_window_lines || '140' }} + READER_SOFT_READS: ${{ inputs.reader_soft_reads || '4' }} + READER_HARD_READS: ${{ inputs.reader_hard_reads || '8' }} + READER_ACTION_THRESHOLD: ${{ inputs.reader_action_threshold || '0.40' }} + OBSERVATION_THRESHOLD: ${{ inputs.observation_threshold || '0.65' }} TYPESAFE_MODEL: ${{ inputs.model || vars.TYPESAFE_MODEL || 'jev-latest' }} jobs: @@ -106,7 +130,21 @@ jobs: shell: bash run: | set -o pipefail - python3 research/code-locator/src/system_one_code_locator.py research/code-locator/fixtures/repository "$QUERY" --offline-decider --directory-threshold "$DIRECTORY_THRESHOLD" --file-threshold "$FILE_THRESHOLD" --line-threshold 0.60 --batch-size "$BATCH_SIZE" --trace-file artifacts/system-one-code-locator/offline/trace.jsonl --output-json artifacts/system-one-code-locator/offline/result.json 2>&1 | tee artifacts/system-one-code-locator/offline/run.log + python3 research/code-locator/src/system_one_code_locator.py \ + research/code-locator/fixtures/repository \ + "$QUERY" \ + --offline-decider \ + --directory-threshold "$DIRECTORY_THRESHOLD" \ + --file-threshold "$FILE_THRESHOLD" \ + --reader-file-activation-threshold "$READER_FILE_ACTIVATION_THRESHOLD" \ + --reader-window-lines "$READER_WINDOW_LINES" \ + --reader-soft-reads "$READER_SOFT_READS" \ + --reader-hard-reads "$READER_HARD_READS" \ + --reader-action-threshold "$READER_ACTION_THRESHOLD" \ + --observation-threshold "$OBSERVATION_THRESHOLD" \ + --trace-file artifacts/system-one-code-locator/offline/trace.jsonl \ + --output-json artifacts/system-one-code-locator/offline/result.json \ + 2>&1 | tee artifacts/system-one-code-locator/offline/run.log - name: Save run metadata and summary env: @@ -128,12 +166,14 @@ jobs: "mode": "offline", "query": os.environ["QUERY"], "model": "offline-lexical-fixture", - "thresholds": { - "directory": float(os.environ["DIRECTORY_THRESHOLD"]), - "file": float(os.environ["FILE_THRESHOLD"]), - "line": 0.60, + "architecture": result.get("architecture"), + "thresholds": result.get("thresholds", {}), + "reader": { + "file_activation_threshold": result.get("metrics", {}).get("reader_file_activation_threshold"), + "window_lines": result.get("metrics", {}).get("reader_window_lines"), + "soft_reads": result.get("metrics", {}).get("reader_soft_reads"), + "hard_reads": result.get("metrics", {}).get("reader_hard_reads"), }, - "batch_size": int(os.environ["BATCH_SIZE"]), "github_run_id": os.environ["RUN_ID"], "github_run_attempt": os.environ["RUN_ATTEMPT"], "commit_sha": os.environ["COMMIT_SHA"], @@ -237,12 +277,19 @@ jobs: "mode": "typesafe", "query": os.environ["QUERY"], "model": os.environ["TYPESAFE_MODEL"], + "architecture": "two_phase_file_locator_plus_progressive_reader", "thresholds": { "directory": float(os.environ["DIRECTORY_THRESHOLD"]), "file": float(os.environ["FILE_THRESHOLD"]), - "line": float(os.environ["LINE_THRESHOLD"]), + "reader_action": float(os.environ["READER_ACTION_THRESHOLD"]), + "observation": float(os.environ["OBSERVATION_THRESHOLD"]), + }, + "reader": { + "file_activation_threshold": float(os.environ["READER_FILE_ACTIVATION_THRESHOLD"]), + "window_lines": int(os.environ["READER_WINDOW_LINES"]), + "soft_reads": int(os.environ["READER_SOFT_READS"]), + "hard_reads": int(os.environ["READER_HARD_READS"]), }, - "batch_size": int(os.environ["BATCH_SIZE"]), "harness": { "repository": os.environ.get("GITHUB_REPOSITORY"), "revision": revision("narness"), @@ -269,7 +316,21 @@ jobs: shell: bash run: | set -o pipefail - python3 narness/research/code-locator/src/system_one_code_locator.py subject "$QUERY" --directory-threshold "$DIRECTORY_THRESHOLD" --file-threshold "$FILE_THRESHOLD" --line-threshold "$LINE_THRESHOLD" --batch-size "$BATCH_SIZE" --model "$TYPESAFE_MODEL" --trace-file artifacts/system-one-code-locator/typesafe/trace.jsonl --output-json artifacts/system-one-code-locator/typesafe/result.json 2>&1 | tee artifacts/system-one-code-locator/typesafe/run.log + python3 narness/research/code-locator/src/system_one_code_locator.py \ + subject \ + "$QUERY" \ + --directory-threshold "$DIRECTORY_THRESHOLD" \ + --file-threshold "$FILE_THRESHOLD" \ + --reader-file-activation-threshold "$READER_FILE_ACTIVATION_THRESHOLD" \ + --reader-window-lines "$READER_WINDOW_LINES" \ + --reader-soft-reads "$READER_SOFT_READS" \ + --reader-hard-reads "$READER_HARD_READS" \ + --reader-action-threshold "$READER_ACTION_THRESHOLD" \ + --observation-threshold "$OBSERVATION_THRESHOLD" \ + --model "$TYPESAFE_MODEL" \ + --trace-file artifacts/system-one-code-locator/typesafe/trace.jsonl \ + --output-json artifacts/system-one-code-locator/typesafe/result.json \ + 2>&1 | tee artifacts/system-one-code-locator/typesafe/run.log - name: Build research summary if: always() @@ -341,7 +402,7 @@ jobs: needs: - offline - typesafe - if: ${{ always() && github.event_name != 'pull_request' }} + if: ${{ always() && (github.event_name == 'workflow_dispatch' || (github.event_name == 'push' && github.ref_name == 'main')) }} runs-on: ubuntu-24.04 timeout-minutes: 10 permissions: diff --git a/.github/workflows/system-one-localization-algorithms.yml b/.github/workflows/system-one-localization-algorithms.yml new file mode 100644 index 00000000..12b8c5e8 --- /dev/null +++ b/.github/workflows/system-one-localization-algorithms.yml @@ -0,0 +1,181 @@ +name: System One Localization Algorithms + +on: + workflow_dispatch: + inputs: + subject_repository: + description: Subject repository + required: true + default: BestNathan/nession + type: string + subject_ref: + description: Subject ref + required: true + default: staging + type: string + query: + description: Localization task + required: true + default: Help me optimize the websocket connection implementation + type: string + +permissions: + contents: read + +env: + SUBJECT_REPOSITORY: ${{ inputs.subject_repository || 'BestNathan/nession' }} + SUBJECT_REF: ${{ inputs.subject_ref || 'staging' }} + QUERY: ${{ inputs.query || 'Help me optimize the websocket connection implementation' }} + +jobs: + experiment: + strategy: + fail-fast: false + matrix: + algorithm: + - evidence-guided + - adaptive-zoom + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + env: + TYPESAFE_MODEL: ${{ vars.TYPESAFE_MODEL || 'jev-latest' }} + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + path: narness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout subject + uses: actions/checkout@v4 + with: + repository: ${{ env.SUBJECT_REPOSITORY }} + ref: ${{ env.SUBJECT_REF }} + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Record revision + run: | + set -euo pipefail + mkdir -p "artifacts/${{ matrix.algorithm }}" + git -C subject rev-parse HEAD > "artifacts/${{ matrix.algorithm }}/subject-sha.txt" + + - name: Run System One algorithm + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + TYPESAFE_API_URL: ${{ vars.TYPESAFE_API_URL || 'https://api.typesafe.ai/v1/systemone' }} + run: | + set -euo pipefail + test -n "$TYPESAFE_API_KEY" + + case "${{ matrix.algorithm }}" in + evidence-guided) + python3 narness/research/code-locator/src/system_one_evidence_guided_runtime.py \ + subject \ + "$QUERY" \ + --directory-threshold 0.50 \ + --file-threshold 0.65 \ + --window-lines 140 \ + --parallel-threshold 0.65 \ + --evidence-threshold 0.65 \ + --max-jumps 2 \ + --max-file-epochs 32 \ + --model "$TYPESAFE_MODEL" \ + --trace-file artifacts/evidence-guided/trace.jsonl \ + --output-json artifacts/evidence-guided/result.json + ;; + adaptive-zoom) + python3 narness/research/code-locator/src/system_one_adaptive_zoom.py \ + subject \ + "$QUERY" \ + --directory-threshold 0.50 \ + --file-threshold 0.65 \ + --evidence-threshold 0.65 \ + --coarse-regions 16 \ + --probe-lines 32 \ + --beam-width 3 \ + --beam-mass 0.80 \ + --exploration-slots 1 \ + --target-region-lines 40 \ + --exploration-floor 0.05 \ + --max-rounds 8 \ + --stable-rounds 2 \ + --model "$TYPESAFE_MODEL" \ + --trace-file artifacts/adaptive-zoom/trace.jsonl \ + --output-json artifacts/adaptive-zoom/result.json + ;; + esac + + - name: Build summary + if: success() + run: | + set -euo pipefail + python3 - <<'PY' + import json + import os + from pathlib import Path + + algorithm = os.environ["ALGORITHM"] + root = Path("artifacts") / algorithm + result = json.loads((root / "result.json").read_text()) + metrics = result["metrics"] + + lines = [ + f"# {algorithm}", + "", + f"- Architecture: {result['architecture']}", + f"- Phase-1 files: {metrics['files_selected']}", + f"- Valuable files: {metrics['valuable_files']}", + f"- Evidence regions: {metrics['evidence_regions']}", + f"- Model calls: {metrics['model_calls']}", + f"- Input tokens: {metrics['input_tokens']}", + f"- Output tokens: {metrics['output_tokens']}", + f"- Elapsed ms: {metrics['elapsed_ms']}", + ] + + if algorithm == "evidence-guided": + lines += [ + f"- Reads: {metrics['reads_executed']}", + f"- Model stops: {metrics['file_model_stops']}", + f"- Space exhausted: {metrics['file_space_exhausted']}", + f"- Budget exhausted: {metrics['file_budget_exhausted']}", + ] + else: + lines += [ + f"- Probe observations: {metrics['observations']}", + f"- Frontier converged: {metrics['probability_frontier_converged']}", + f"- Round budget exhausted: {metrics['round_budget_exhausted']}", + ] + + lines += ["", "## Result files", ""] + for item in result["result_files"]: + lines.append( + f"- `{item['path']}`: " + f"score={item['score']:.3f} " + f"evidence={len(item['evidence'])}" + ) + + (root / "summary.md").write_text( + "\n".join(lines) + "\n", + encoding="utf-8", + ) + PY + cat "artifacts/${{ matrix.algorithm }}/summary.md" >> "$GITHUB_STEP_SUMMARY" + env: + ALGORITHM: ${{ matrix.algorithm }} + + - name: Upload artifact + if: always() + uses: actions/upload-artifact@v4 + with: + name: system-one-${{ matrix.algorithm }}-${{ github.run_id }} + path: artifacts/${{ matrix.algorithm }}/ + if-no-files-found: error + retention-days: 90 diff --git a/.github/workflows/system-one-range-runtime-v0.yml b/.github/workflows/system-one-range-runtime-v0.yml new file mode 100644 index 00000000..38117b84 --- /dev/null +++ b/.github/workflows/system-one-range-runtime-v0.yml @@ -0,0 +1,141 @@ +name: System One Range Runtime v0 + +on: + workflow_dispatch: + inputs: + subject_repository: + description: Subject repository + required: true + default: BestNathan/nession + type: string + subject_ref: + description: Subject ref + required: true + default: staging + type: string + query: + description: Localization task + required: true + default: Help me optimize the websocket connection implementation + type: string + +permissions: + contents: read + +env: + SUBJECT_REPOSITORY: ${{ inputs.subject_repository || 'BestNathan/nession' }} + SUBJECT_REF: ${{ inputs.subject_ref || 'staging' }} + QUERY: ${{ inputs.query || 'Help me optimize the websocket connection implementation' }} + +jobs: + range-runtime: + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + env: + TYPESAFE_MODEL: ${{ vars.TYPESAFE_MODEL || 'jev-latest' }} + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + path: narness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout subject + uses: actions/checkout@v4 + with: + repository: ${{ env.SUBJECT_REPOSITORY }} + ref: ${{ env.SUBJECT_REF }} + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Record subject revision + run: | + set -euo pipefail + mkdir -p artifacts/range-runtime + git -C subject rev-parse HEAD > artifacts/range-runtime/subject-sha.txt + + - name: Run range-only runtime + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + TYPESAFE_API_URL: ${{ vars.TYPESAFE_API_URL || 'https://api.typesafe.ai/v1/systemone' }} + run: | + set -euo pipefail + test -n "$TYPESAFE_API_KEY" + python3 narness/research/code-locator/src/system_one_range_runtime.py \ + subject \ + "$QUERY" \ + --directory-threshold 0.50 \ + --file-threshold 0.65 \ + --window-lines 140 \ + --parallel-threshold 0.65 \ + --max-jumps 2 \ + --max-file-epochs 32 \ + --model "$TYPESAFE_MODEL" \ + --trace-file artifacts/range-runtime/trace.jsonl \ + --output-json artifacts/range-runtime/result.json + + - name: Write summary + if: success() + run: | + set -euo pipefail + python3 - <<'PY' + import json + from pathlib import Path + + root=Path("artifacts/range-runtime") + result=json.loads((root/"result.json").read_text()) + metrics=result["metrics"] + + lines=[ + "# System One Per-File Range Runtime v0", + "", + f"- Phase-1 files: {metrics['files_selected']}", + f"- File runtimes: {metrics['file_runtimes']}", + f"- Model stops: {metrics['file_model_stops']}", + f"- Space exhausted: {metrics['file_space_exhausted']}", + f"- Budget exhausted: {metrics['file_budget_exhausted']}", + f"- Reads: {metrics['reads_executed']}", + f"- Valuable files: {metrics['valuable_files']}", + f"- Evidence regions: {metrics['evidence_regions']}", + f"- Model calls: {metrics['model_calls']}", + f"- Input tokens: {metrics['input_tokens']}", + f"- Output tokens: {metrics['output_tokens']}", + f"- Elapsed ms: {metrics['elapsed_ms']}", + "", + "## Per-file runtime", + "", + ] + for state in result["file_states"]: + lines.append( + f"- \`{state['path']}\`: " + f"stop={state['termination']} " + f"epochs={state['epoch']} " + f"reads={state['read_count']} " + f"coverage={state['coverage']}" + ) + lines += ["", "## Valuable files", ""] + for item in result["result_files"]: + lines.append( + f"- \`{item['path']}\`: " + f"score={item['score']:.3f} " + f"evidence={len(item['evidence'])}" + ) + (root/"summary.md").write_text("\n".join(lines)+"\n") + PY + cat artifacts/range-runtime/summary.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload result + if: always() + uses: actions/upload-artifact@v4 + with: + name: system-one-range-runtime-v0-${{ github.run_id }} + path: artifacts/range-runtime/ + if-no-files-found: error + retention-days: 90 diff --git a/research/code-locator/README.md b/research/code-locator/README.md index 9c534649..14ec517a 100644 --- a/research/code-locator/README.md +++ b/research/code-locator/README.md @@ -1,219 +1,393 @@ # System One Code Locator -> Experimental research project. This is a localization harness, not a code-changing agent and not part of the canonical Narness runtime. +> Experimental research project. This is a localization harness, not a code-changing agent. -This directory is the complete Code Locator research unit: implementation, fixtures, tests, design notes, pilot analyses, and immutable run records live together here. +Code Locator uses a two-phase architecture with progressive state-space disclosure. -## Project layout +## Phase 1 — File Locator -```text -research/code-locator/ -├── README.md -├── src/ -│ └── system_one_code_locator.py -├── tests/ -│ └── test_system_one_code_locator.py -├── fixtures/ -│ └── repository/ -├── docs/ -│ ├── design.md -│ └── pilots/ -│ └── nession-websocket-2026-09-23.md -└── runs/ - └──