diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml new file mode 100644 index 00000000..4d124874 --- /dev/null +++ b/.github/workflows/full-read-claude-baseline.yml @@ -0,0 +1,156 @@ +name: Full-read Claude relevance baseline + +on: + push: + branches: + - experiment/full-read-baseline-20260924 + +permissions: + contents: read + actions: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + FILE_PATH: crates/nession-agent/src/server/websocket.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + HARNESS_SHA: f93b57d81e9e3f530ec1a26f23bb2472aa90d1b + V1C_RUN_ID: "35952482145" + V1C_ARTIFACT: single-file-range-vs-frontier-v1c-35952482145 + +jobs: + baseline: + name: Build full-read Claude reference field + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: f93b57d81e9e3f530ec1a26f23bb2472aa90d1b + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - name: Setup Python + uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Setup Node + uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Prepare reference prompt + run: | + set -euo pipefail + mkdir -p artifacts/baseline + python3 harness/src/full_read_relevance_baseline.py prepare \ + --source "subject/$FILE_PATH" \ + --query "$QUERY" \ + --window-lines 64 \ + --stride-lines 32 \ + --output-prompt artifacts/baseline/prompt.txt \ + --output-manifest artifacts/baseline/manifest.json + + - name: Generate full-read Claude field + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + test -n "$ANTHROPIC_API_KEY" + test -n "$CLAUDE_MODEL" + claude -p \ + --output-format text \ + --model "$CLAUDE_MODEL" \ + --effort high \ + --max-turns 4 \ + --permission-mode dontAsk \ + --no-session-persistence \ + --setting-sources '' \ + --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' \ + --disable-slash-commands \ + --tools '' \ + --settings '{"disableAllHooks":true}' \ + < artifacts/baseline/prompt.txt \ + > artifacts/baseline/claude-output.txt + + - name: Normalize reference field + run: | + set -euo pipefail + python3 harness/src/full_read_relevance_baseline.py normalize \ + --source "subject/$FILE_PATH" \ + --query "$QUERY" \ + --window-lines 64 \ + --stride-lines 32 \ + --raw-output artifacts/baseline/claude-output.txt \ + --output artifacts/baseline/reference.json + + - name: Download v1c candidate + uses: actions/download-artifact@v5 + with: + name: single-file-range-vs-frontier-v1c-35952482145 + path: artifacts/v1c + run-id: 35952482145 + github-token: ${{ secrets.GITHUB_TOKEN }} + + - name: Project v1c results onto reference field + run: | + set -euo pipefail + python3 harness/src/compare_to_full_read_baseline.py \ + --reference artifacts/baseline/reference.json \ + --candidate artifacts/v1c/frontier/localization-result.json \ + --output-json artifacts/baseline/v1c-frontier-reference-metrics.json + python3 harness/src/compare_to_full_read_baseline.py \ + --reference artifacts/baseline/reference.json \ + --candidate artifacts/v1c/range/localization-result.json \ + --output-json artifacts/baseline/v1c-range-reference-metrics.json + + - name: Summarize + run: | + set -euo pipefail + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + for label, path in [ + ("Range v1c", "artifacts/baseline/v1c-range-reference-metrics.json"), + ("Frontier v1c", "artifacts/baseline/v1c-frontier-reference-metrics.json"), + ]: + r = json.load(open(path)) + m = r["metrics"] + print(f"## {label} vs full-read reference") + print("") + print("| Metric | Value |") + print("| --- | ---: |") + print(f"| weighted relevance recall | {m['weighted_relevance_recall']:.3f} |") + print(f"| high-relevance window recall | {m['high_relevance_window_recall']:.3f} |") + print(f"| core-window recall | {m['core_window_recall']:.3f} |") + print(f"| relevance-weighted precision | {m['relevance_weighted_precision']:.3f} |") + print(f"| source coverage | {r['candidate']['source_coverage']:.3f} |") + print("") + PY + + - name: Upload baseline + uses: actions/upload-artifact@v4 + with: + name: full-read-claude-baseline + path: artifacts/baseline + if-no-files-found: error + retention-days: 90 diff --git a/.github/workflows/single-file-range-vs-frontier-v1b.yml b/.github/workflows/single-file-range-vs-frontier-v1b.yml new file mode 100644 index 00000000..48292520 --- /dev/null +++ b/.github/workflows/single-file-range-vs-frontier-v1b.yml @@ -0,0 +1,418 @@ +name: Single-file Range vs Relevance Frontier v1c + +on: + push: + branches: + - experiment/single-file-frontier-v1b-20260924 + +permissions: + contents: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + FILE_PATH: crates/nession-agent/src/server/websocket.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + HARNESS_SHA: cfc3d696df6c9ecc833372cf616805d2b931d0af + TYPESAFE_MODEL: jev-latest + TYPESAFE_API_URL: https://api.typesafe.ai/v1/systemone + +jobs: + experiment: + name: Single-file Range vs Frontier v1c + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: cfc3d696df6c9ecc833372cf616805d2b931d0af + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Run single-file experiment + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + run: | + set -euo pipefail + test -n "$TYPESAFE_API_KEY" + mkdir -p artifacts/range artifacts/frontier + python3 - <<'PY' + import json + import os + import sys + import time + from pathlib import Path + + sys.path.insert(0, str(Path("harness/src").resolve())) + from localization_result import ( + build_system_one_range_result, + build_system_one_relevance_frontier_result, + ) + from system_one_code_locator import Trace + from system_one_range_runtime import ( + SystemOneFileDecider, + canonical_file_results, + run_file_runtime, + ) + from system_one_relevance_frontier import ( + RelevanceFrontierDecider, + canonical_results as canonical_frontier_results, + run_frontier_file, + ) + + root = Path("subject") + query = os.environ["QUERY"] + model = os.environ["TYPESAFE_MODEL"] + endpoint = os.environ["TYPESAFE_API_URL"] + key = os.environ["TYPESAFE_API_KEY"] + file_path = os.environ["FILE_PATH"] + subject = { + "repository": os.environ["SUBJECT_REPOSITORY"], + "revision": os.environ["SUBJECT_SHA"], + } + candidate = { + "score": 0.90, + "payload": {"path": file_path, "extension": ".rs"}, + } + + range_params = { + "window_lines": 140, + "parallel_threshold": 0.65, + "max_jumps": 2, + "max_file_epochs": 32, + } + frontier_params = { + "max_rounds": 10, + "max_actions_per_round": 2, + "probe_lines": 112, + "target_region_lines": 48, + "final_window_lines": 32, + "refine_threshold": 0.72, + "candidate_threshold": 0.55, + "gradient_threshold": 0.15, + "volatility_threshold": 0.10, + "stable_delta": 0.06, + "stable_rounds": 2, + "max_frontier_leaves": 24, + "final_max_candidates": 24, + } + + def coverage(state): + rows = sorted((int(a), int(b)) for a, b in state.get("coverage", [])) + merged = [] + for a, b in rows: + if not merged or a > merged[-1][1] + 1: + merged.append([a, b]) + else: + merged[-1][1] = max(merged[-1][1], b) + return sum(b - a + 1 for a, b in merged) / max(1, int(state["line_count"])) + + def evidence_ranges(result): + return [ + (int(e["start_line"]), int(e["end_line"])) + for item in result.get("files", []) + for e in item.get("evidence", []) + ] + + def overlap(left, right): + total = sum(b - a + 1 for a, b in left) + if not total: + return 0.0 + hit = 0 + for a, b in left: + for c, d in right: + hit += max(0, min(b, d) - max(a, c) + 1) + return min(1.0, hit / total) + + def action_counts(state): + counts = {} + for snapshot in state.get("action_history", []): + for action in snapshot.get("actions", []): + kind = action.get("kind") + counts[kind] = counts.get(kind, 0) + 1 + return counts + + range_trace = Trace("artifacts/range/trace.jsonl") + range_decider = SystemOneFileDecider(key, range_trace, endpoint, model) + started = time.perf_counter() + range_state, range_usage = run_file_runtime( + root, query, candidate, range_decider, range_trace, **range_params + ) + range_ms = (time.perf_counter() - started) * 1000 + range_files = canonical_file_results([range_state], 0.65) + range_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [range_state], "result_files": range_files, + "metrics": { + **range_usage, + "reads_executed": range_state["read_count"], + "evidence_regions": sum(len(x["evidence"]) for x in range_files), + "elapsed_ms": round(range_ms, 3), + }, + } + range_canonical = build_system_one_range_result(range_engine, model) + + frontier_trace = Trace("artifacts/frontier/trace.jsonl") + frontier_decider = RelevanceFrontierDecider( + key, frontier_trace, endpoint, model + ) + started = time.perf_counter() + frontier_state, frontier_usage = run_frontier_file( + root, query, candidate, frontier_decider, frontier_trace, + **frontier_params + ) + frontier_ms = (time.perf_counter() - started) * 1000 + frontier_files = canonical_frontier_results([frontier_state]) + frontier_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [frontier_state], "result_files": frontier_files, + "metrics": { + **frontier_usage, + "reads_executed": len(frontier_state["observations"]), + "evidence_regions": sum(len(x["evidence"]) for x in frontier_files), + "elapsed_ms": round(frontier_ms, 3), + }, + } + frontier_canonical = build_system_one_relevance_frontier_result( + frontier_engine, model + ) + + report = { + "query": query, + "file": file_path, + "subject": subject, + "model": model, + "frontier_params": frontier_params, + "range_params": range_params, + "range": { + "metrics": range_engine["metrics"], + "termination": range_state["termination"], + "epochs": range_state["epoch"], + "source_coverage": coverage(range_state), + "evidence": evidence_ranges(range_canonical), + }, + "frontier": { + "metrics": frontier_engine["metrics"], + "termination": frontier_state["termination"], + "rounds": frontier_state["round"], + "frontier_coverage": frontier_state["frontier_coverage"], + "source_coverage": frontier_state["source_coverage"], + "action_counts": action_counts(frontier_state), + "evidence": evidence_ranges(frontier_canonical), + }, + } + report["comparison"] = { + "elapsed_ratio": frontier_ms / max(1.0, range_ms), + "read_ratio": len(frontier_state["observations"]) / max(1, range_state["read_count"]), + "input_token_ratio": frontier_usage["input_tokens"] / max(1, range_usage["input_tokens"]), + "output_token_ratio": frontier_usage["output_tokens"] / max(1, range_usage["output_tokens"]), + "frontier_evidence_covered_by_range": overlap( + evidence_ranges(frontier_canonical), evidence_ranges(range_canonical) + ), + "range_evidence_covered_by_frontier": overlap( + evidence_ranges(range_canonical), evidence_ranges(frontier_canonical) + ), + } + + Path("artifacts/range/result.json").write_text( + json.dumps(range_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/range/localization-result.json").write_text( + json.dumps(range_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/result.json").write_text( + json.dumps(frontier_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/localization-result.json").write_text( + json.dumps(frontier_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/report.json").write_text( + json.dumps(report, indent=2, ensure_ascii=False) + "\n" + ) + print(json.dumps(report, indent=2, ensure_ascii=False)) + PY + + - name: Compare canonical results + run: | + set -euo pipefail + python3 harness/src/compare_localization_results.py \ + --left artifacts/range/localization-result.json \ + --right artifacts/frontier/localization-result.json \ + --output-json artifacts/localization-comparison.json \ + --output-markdown artifacts/localization-comparison.md + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + r = json.load(open("artifacts/report.json")) + print("## Single-file Range vs Frontier v1c") + print("") + print("| Metric | Range | Frontier v1c |") + print("| --- | ---: | ---: |") + print("| elapsed ms | %s | %s |" % (r["range"]["metrics"]["elapsed_ms"], r["frontier"]["metrics"]["elapsed_ms"])) + print("| model calls | %s | %s |" % (r["range"]["metrics"]["model_calls"], r["frontier"]["metrics"]["model_calls"])) + print("| input tokens | %s | %s |" % (r["range"]["metrics"]["input_tokens"], r["frontier"]["metrics"]["input_tokens"])) + print("| output tokens | %s | %s |" % (r["range"]["metrics"]["output_tokens"], r["frontier"]["metrics"]["output_tokens"])) + print("| reads | %s | %s |" % (r["range"]["metrics"]["reads_executed"], r["frontier"]["metrics"]["reads_executed"])) + print("| source coverage | %.1f%% | %.1f%% |" % (r["range"]["source_coverage"] * 100, r["frontier"]["source_coverage"] * 100)) + print("| frontier coverage | - | %.1f%% |" % (r["frontier"]["frontier_coverage"] * 100)) + print("| evidence regions | %s | %s |" % (r["range"]["metrics"]["evidence_regions"], r["frontier"]["metrics"]["evidence_regions"])) + print("| termination | %s | %s |" % (r["range"]["termination"], r["frontier"]["termination"])) + print("") + print("Frontier action counts:", r["frontier"]["action_counts"]) + PY + cat artifacts/localization-comparison.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload experiment + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-frontier-v1c-${{ github.run_id }} + path: artifacts/ + if-no-files-found: error + retention-days: 90 + + blind-quality: + name: Blind quality review + needs: experiment + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: cfc3d696df6c9ecc833372cf616805d2b931d0af + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Download experiment + uses: actions/download-artifact@v4 + with: + name: single-file-range-vs-frontier-v1c-${{ github.run_id }} + path: artifacts + + - name: Prepare anonymous candidates + run: | + set -euo pipefail + mkdir -p artifacts/quality/range artifacts/quality/frontier + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/range/localization-result.json \ + --candidate-id candidate-a \ + --output-candidate artifacts/quality/range/candidate.json \ + --output-prompt artifacts/quality/range/prompt.txt + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/frontier/localization-result.json \ + --candidate-id candidate-b \ + --output-candidate artifacts/quality/frontier/candidate.json \ + --output-prompt artifacts/quality/frontier/prompt.txt + + - name: Evaluate range anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/range/prompt.txt \ + > ../artifacts/quality/range/evaluator.raw.jsonl + + - name: Evaluate frontier anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/frontier/prompt.txt \ + > ../artifacts/quality/frontier/evaluator.raw.jsonl + + - name: Finalize blind scorecards + run: | + set -euo pipefail + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/range/candidate.json \ + --raw-jsonl artifacts/quality/range/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/range/scorecard.json \ + --output-markdown artifacts/quality/range/report.md + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/frontier/candidate.json \ + --raw-jsonl artifacts/quality/frontier/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/frontier/scorecard.json \ + --output-markdown artifacts/quality/frontier/report.md + python3 harness/src/localization_quality_evaluation.py report \ + --candidate-a artifacts/quality/range/scorecard.json \ + --candidate-b artifacts/quality/frontier/scorecard.json \ + --candidate-a-label "Range runtime" \ + --candidate-b-label "Relevance frontier v1c" \ + --output-json artifacts/quality/evaluation-report.json \ + --output-markdown artifacts/quality/evaluation-report.md + cat artifacts/quality/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload quality + if: always() + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-frontier-v1c-quality-${{ github.run_id }} + path: artifacts/quality/ + if-no-files-found: error + retention-days: 90 diff --git a/.github/workflows/single-file-range-vs-phase0-choice-v2.yml b/.github/workflows/single-file-range-vs-phase0-choice-v2.yml new file mode 100644 index 00000000..c99b815f --- /dev/null +++ b/.github/workflows/single-file-range-vs-phase0-choice-v2.yml @@ -0,0 +1,424 @@ +name: Single-file Range vs Phase0 + Choice Frontier v2 + +on: + push: + branches: + - experiment/phase0-choice-frontier-v2-20260924 + +permissions: + contents: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + FILE_PATH: crates/nession-agent/src/server/websocket.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + HARNESS_SHA: b2264c9f14d63c2617ac76b335f43e4121da1d8b + TYPESAFE_MODEL: jev-latest + TYPESAFE_API_URL: https://api.typesafe.ai/v1/systemone + +jobs: + experiment: + name: Single-file Range vs Frontier phase0-choice-v2 + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: b2264c9f14d63c2617ac76b335f43e4121da1d8b + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Run single-file experiment + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + run: | + set -euo pipefail + test -n "$TYPESAFE_API_KEY" + mkdir -p artifacts/range artifacts/frontier + python3 - <<'PY' + import json + import os + import sys + import time + from pathlib import Path + + sys.path.insert(0, str(Path("harness/src").resolve())) + from localization_result import ( + build_system_one_range_result, + build_system_one_relevance_frontier_result, + ) + from system_one_code_locator import Trace + from system_one_range_runtime import ( + SystemOneFileDecider, + canonical_file_results, + run_file_runtime, + ) + from system_one_relevance_frontier import ( + ChoiceRelevanceFrontierDecider, + canonical_results as canonical_frontier_results, + run_frontier_file_phase0_choice_v2, + ) + + root = Path("subject") + query = os.environ["QUERY"] + model = os.environ["TYPESAFE_MODEL"] + endpoint = os.environ["TYPESAFE_API_URL"] + key = os.environ["TYPESAFE_API_KEY"] + file_path = os.environ["FILE_PATH"] + subject = { + "repository": os.environ["SUBJECT_REPOSITORY"], + "revision": os.environ["SUBJECT_SHA"], + } + candidate = { + "score": 0.90, + "payload": {"path": file_path, "extension": ".rs"}, + } + + range_params = { + "window_lines": 140, + "parallel_threshold": 0.65, + "max_jumps": 2, + "max_file_epochs": 32, + } + frontier_params = { + "max_rounds": 10, + "max_actions_per_choice": 6, + "probe_lines": 112, + "target_region_lines": 48, + "final_window_lines": 32, + "refine_threshold": 0.72, + "candidate_threshold": 0.55, + "gradient_threshold": 0.15, + "volatility_threshold": 0.10, + "stable_delta": 0.06, + "stable_rounds": 2, + "max_frontier_leaves": 24, + "final_max_candidates": 24, + } + + def coverage(state): + rows = sorted((int(a), int(b)) for a, b in state.get("coverage", [])) + merged = [] + for a, b in rows: + if not merged or a > merged[-1][1] + 1: + merged.append([a, b]) + else: + merged[-1][1] = max(merged[-1][1], b) + return sum(b - a + 1 for a, b in merged) / max(1, int(state["line_count"])) + + def evidence_ranges(result): + return [ + (int(e["start_line"]), int(e["end_line"])) + for item in result.get("files", []) + for e in item.get("evidence", []) + ] + + def overlap(left, right): + total = sum(b - a + 1 for a, b in left) + if not total: + return 0.0 + hit = 0 + for a, b in left: + for c, d in right: + hit += max(0, min(b, d) - max(a, c) + 1) + return min(1.0, hit / total) + + def action_counts(state): + counts = {} + for snapshot in state.get("action_history", []): + policy_action = snapshot.get("policy", {}).get("action") + if isinstance(policy_action, dict): + kind = policy_action.get("kind") + if kind: + counts[kind] = counts.get(kind, 0) + 1 + for action in snapshot.get("actions", []): + kind = action.get("kind") + if kind: + counts[kind] = counts.get(kind, 0) + 1 + return counts + + range_trace = Trace("artifacts/range/trace.jsonl") + range_decider = SystemOneFileDecider(key, range_trace, endpoint, model) + started = time.perf_counter() + range_state, range_usage = run_file_runtime( + root, query, candidate, range_decider, range_trace, **range_params + ) + range_ms = (time.perf_counter() - started) * 1000 + range_files = canonical_file_results([range_state], 0.65) + range_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [range_state], "result_files": range_files, + "metrics": { + **range_usage, + "reads_executed": range_state["read_count"], + "evidence_regions": sum(len(x["evidence"]) for x in range_files), + "elapsed_ms": round(range_ms, 3), + }, + } + range_canonical = build_system_one_range_result(range_engine, model) + + frontier_trace = Trace("artifacts/frontier/trace.jsonl") + frontier_decider = ChoiceRelevanceFrontierDecider( + key, frontier_trace, endpoint, model + ) + started = time.perf_counter() + frontier_state, frontier_usage = run_frontier_file_phase0_choice_v2( + root, query, candidate, frontier_decider, frontier_trace, + **frontier_params + ) + frontier_ms = (time.perf_counter() - started) * 1000 + frontier_files = canonical_frontier_results([frontier_state]) + frontier_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [frontier_state], "result_files": frontier_files, + "metrics": { + **frontier_usage, + "reads_executed": len(frontier_state["observations"]), + "evidence_regions": sum(len(x["evidence"]) for x in frontier_files), + "elapsed_ms": round(frontier_ms, 3), + }, + } + frontier_canonical = build_system_one_relevance_frontier_result( + frontier_engine, model + ) + + report = { + "query": query, + "file": file_path, + "subject": subject, + "model": model, + "frontier_params": frontier_params, + "range_params": range_params, + "range": { + "metrics": range_engine["metrics"], + "termination": range_state["termination"], + "epochs": range_state["epoch"], + "source_coverage": coverage(range_state), + "evidence": evidence_ranges(range_canonical), + }, + "frontier": { + "metrics": frontier_engine["metrics"], + "termination": frontier_state["termination"], + "rounds": frontier_state["round"], + "frontier_coverage": frontier_state["frontier_coverage"], + "source_coverage": frontier_state["source_coverage"], + "action_counts": action_counts(frontier_state), + "evidence": evidence_ranges(frontier_canonical), + }, + } + report["comparison"] = { + "elapsed_ratio": frontier_ms / max(1.0, range_ms), + "read_ratio": len(frontier_state["observations"]) / max(1, range_state["read_count"]), + "input_token_ratio": frontier_usage["input_tokens"] / max(1, range_usage["input_tokens"]), + "output_token_ratio": frontier_usage["output_tokens"] / max(1, range_usage["output_tokens"]), + "frontier_evidence_covered_by_range": overlap( + evidence_ranges(frontier_canonical), evidence_ranges(range_canonical) + ), + "range_evidence_covered_by_frontier": overlap( + evidence_ranges(range_canonical), evidence_ranges(frontier_canonical) + ), + } + + Path("artifacts/range/result.json").write_text( + json.dumps(range_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/range/localization-result.json").write_text( + json.dumps(range_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/result.json").write_text( + json.dumps(frontier_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/localization-result.json").write_text( + json.dumps(frontier_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/report.json").write_text( + json.dumps(report, indent=2, ensure_ascii=False) + "\n" + ) + print(json.dumps(report, indent=2, ensure_ascii=False)) + PY + + - name: Compare canonical results + run: | + set -euo pipefail + python3 harness/src/compare_localization_results.py \ + --left artifacts/range/localization-result.json \ + --right artifacts/frontier/localization-result.json \ + --output-json artifacts/localization-comparison.json \ + --output-markdown artifacts/localization-comparison.md + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + r = json.load(open("artifacts/report.json")) + print("## Single-file Range vs Frontier phase0-choice-v2") + print("") + print("| Metric | Range | Frontier phase0-choice-v2 |") + print("| --- | ---: | ---: |") + print("| elapsed ms | %s | %s |" % (r["range"]["metrics"]["elapsed_ms"], r["frontier"]["metrics"]["elapsed_ms"])) + print("| model calls | %s | %s |" % (r["range"]["metrics"]["model_calls"], r["frontier"]["metrics"]["model_calls"])) + print("| input tokens | %s | %s |" % (r["range"]["metrics"]["input_tokens"], r["frontier"]["metrics"]["input_tokens"])) + print("| output tokens | %s | %s |" % (r["range"]["metrics"]["output_tokens"], r["frontier"]["metrics"]["output_tokens"])) + print("| reads | %s | %s |" % (r["range"]["metrics"]["reads_executed"], r["frontier"]["metrics"]["reads_executed"])) + print("| source coverage | %.1f%% | %.1f%% |" % (r["range"]["source_coverage"] * 100, r["frontier"]["source_coverage"] * 100)) + print("| frontier coverage | - | %.1f%% |" % (r["frontier"]["frontier_coverage"] * 100)) + print("| evidence regions | %s | %s |" % (r["range"]["metrics"]["evidence_regions"], r["frontier"]["metrics"]["evidence_regions"])) + print("| termination | %s | %s |" % (r["range"]["termination"], r["frontier"]["termination"])) + print("") + print("Frontier action counts:", r["frontier"]["action_counts"]) + PY + cat artifacts/localization-comparison.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload experiment + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-v2-${{ github.run_id }} + path: artifacts/ + if-no-files-found: error + retention-days: 90 + + blind-quality: + name: Blind quality review + needs: experiment + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: b2264c9f14d63c2617ac76b335f43e4121da1d8b + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Download experiment + uses: actions/download-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-v2-${{ github.run_id }} + path: artifacts + + - name: Prepare anonymous candidates + run: | + set -euo pipefail + mkdir -p artifacts/quality/range artifacts/quality/frontier + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/range/localization-result.json \ + --candidate-id candidate-a \ + --output-candidate artifacts/quality/range/candidate.json \ + --output-prompt artifacts/quality/range/prompt.txt + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/frontier/localization-result.json \ + --candidate-id candidate-b \ + --output-candidate artifacts/quality/frontier/candidate.json \ + --output-prompt artifacts/quality/frontier/prompt.txt + + - name: Evaluate range anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/range/prompt.txt \ + > ../artifacts/quality/range/evaluator.raw.jsonl + + - name: Evaluate frontier anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/frontier/prompt.txt \ + > ../artifacts/quality/frontier/evaluator.raw.jsonl + + - name: Finalize blind scorecards + run: | + set -euo pipefail + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/range/candidate.json \ + --raw-jsonl artifacts/quality/range/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/range/scorecard.json \ + --output-markdown artifacts/quality/range/report.md + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/frontier/candidate.json \ + --raw-jsonl artifacts/quality/frontier/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/frontier/scorecard.json \ + --output-markdown artifacts/quality/frontier/report.md + python3 harness/src/localization_quality_evaluation.py report \ + --candidate-a artifacts/quality/range/scorecard.json \ + --candidate-b artifacts/quality/frontier/scorecard.json \ + --candidate-a-label "Range runtime" \ + --candidate-b-label "Relevance frontier phase0-choice-v2" \ + --output-json artifacts/quality/evaluation-report.json \ + --output-markdown artifacts/quality/evaluation-report.md + cat artifacts/quality/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload quality + if: always() + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-v2-quality-${{ github.run_id }} + path: artifacts/quality/ + if-no-files-found: error + retention-days: 90 diff --git a/.github/workflows/single-file-range-vs-phase0-choice.yml b/.github/workflows/single-file-range-vs-phase0-choice.yml new file mode 100644 index 00000000..7352ef5e --- /dev/null +++ b/.github/workflows/single-file-range-vs-phase0-choice.yml @@ -0,0 +1,423 @@ +name: Single-file Range vs Phase0 + Choice Frontier + +on: + push: + branches: + - experiment/phase0-choice-frontier-20260924 + +permissions: + contents: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + FILE_PATH: crates/nession-agent/src/server/websocket.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + HARNESS_SHA: fa517f9d5916754030f4d9b0068b075066a7c220 + TYPESAFE_MODEL: jev-latest + TYPESAFE_API_URL: https://api.typesafe.ai/v1/systemone + +jobs: + experiment: + name: Single-file Range vs Frontier phase0-choice + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: fa517f9d5916754030f4d9b0068b075066a7c220 + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Run single-file experiment + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + run: | + set -euo pipefail + test -n "$TYPESAFE_API_KEY" + mkdir -p artifacts/range artifacts/frontier + python3 - <<'PY' + import json + import os + import sys + import time + from pathlib import Path + + sys.path.insert(0, str(Path("harness/src").resolve())) + from localization_result import ( + build_system_one_range_result, + build_system_one_relevance_frontier_result, + ) + from system_one_code_locator import Trace + from system_one_range_runtime import ( + SystemOneFileDecider, + canonical_file_results, + run_file_runtime, + ) + from system_one_relevance_frontier import ( + ChoiceRelevanceFrontierDecider, + canonical_results as canonical_frontier_results, + run_frontier_file_phase0_choice, + ) + + root = Path("subject") + query = os.environ["QUERY"] + model = os.environ["TYPESAFE_MODEL"] + endpoint = os.environ["TYPESAFE_API_URL"] + key = os.environ["TYPESAFE_API_KEY"] + file_path = os.environ["FILE_PATH"] + subject = { + "repository": os.environ["SUBJECT_REPOSITORY"], + "revision": os.environ["SUBJECT_SHA"], + } + candidate = { + "score": 0.90, + "payload": {"path": file_path, "extension": ".rs"}, + } + + range_params = { + "window_lines": 140, + "parallel_threshold": 0.65, + "max_jumps": 2, + "max_file_epochs": 32, + } + frontier_params = { + "max_rounds": 10, + "probe_lines": 112, + "target_region_lines": 48, + "final_window_lines": 32, + "refine_threshold": 0.72, + "candidate_threshold": 0.55, + "gradient_threshold": 0.15, + "volatility_threshold": 0.10, + "stable_delta": 0.06, + "stable_rounds": 2, + "max_frontier_leaves": 24, + "final_max_candidates": 24, + } + + def coverage(state): + rows = sorted((int(a), int(b)) for a, b in state.get("coverage", [])) + merged = [] + for a, b in rows: + if not merged or a > merged[-1][1] + 1: + merged.append([a, b]) + else: + merged[-1][1] = max(merged[-1][1], b) + return sum(b - a + 1 for a, b in merged) / max(1, int(state["line_count"])) + + def evidence_ranges(result): + return [ + (int(e["start_line"]), int(e["end_line"])) + for item in result.get("files", []) + for e in item.get("evidence", []) + ] + + def overlap(left, right): + total = sum(b - a + 1 for a, b in left) + if not total: + return 0.0 + hit = 0 + for a, b in left: + for c, d in right: + hit += max(0, min(b, d) - max(a, c) + 1) + return min(1.0, hit / total) + + def action_counts(state): + counts = {} + for snapshot in state.get("action_history", []): + policy_action = snapshot.get("policy", {}).get("action") + if isinstance(policy_action, dict): + kind = policy_action.get("kind") + if kind: + counts[kind] = counts.get(kind, 0) + 1 + for action in snapshot.get("actions", []): + kind = action.get("kind") + if kind: + counts[kind] = counts.get(kind, 0) + 1 + return counts + + range_trace = Trace("artifacts/range/trace.jsonl") + range_decider = SystemOneFileDecider(key, range_trace, endpoint, model) + started = time.perf_counter() + range_state, range_usage = run_file_runtime( + root, query, candidate, range_decider, range_trace, **range_params + ) + range_ms = (time.perf_counter() - started) * 1000 + range_files = canonical_file_results([range_state], 0.65) + range_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [range_state], "result_files": range_files, + "metrics": { + **range_usage, + "reads_executed": range_state["read_count"], + "evidence_regions": sum(len(x["evidence"]) for x in range_files), + "elapsed_ms": round(range_ms, 3), + }, + } + range_canonical = build_system_one_range_result(range_engine, model) + + frontier_trace = Trace("artifacts/frontier/trace.jsonl") + frontier_decider = ChoiceRelevanceFrontierDecider( + key, frontier_trace, endpoint, model + ) + started = time.perf_counter() + frontier_state, frontier_usage = run_frontier_file_phase0_choice( + root, query, candidate, frontier_decider, frontier_trace, + **frontier_params + ) + frontier_ms = (time.perf_counter() - started) * 1000 + frontier_files = canonical_frontier_results([frontier_state]) + frontier_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [frontier_state], "result_files": frontier_files, + "metrics": { + **frontier_usage, + "reads_executed": len(frontier_state["observations"]), + "evidence_regions": sum(len(x["evidence"]) for x in frontier_files), + "elapsed_ms": round(frontier_ms, 3), + }, + } + frontier_canonical = build_system_one_relevance_frontier_result( + frontier_engine, model + ) + + report = { + "query": query, + "file": file_path, + "subject": subject, + "model": model, + "frontier_params": frontier_params, + "range_params": range_params, + "range": { + "metrics": range_engine["metrics"], + "termination": range_state["termination"], + "epochs": range_state["epoch"], + "source_coverage": coverage(range_state), + "evidence": evidence_ranges(range_canonical), + }, + "frontier": { + "metrics": frontier_engine["metrics"], + "termination": frontier_state["termination"], + "rounds": frontier_state["round"], + "frontier_coverage": frontier_state["frontier_coverage"], + "source_coverage": frontier_state["source_coverage"], + "action_counts": action_counts(frontier_state), + "evidence": evidence_ranges(frontier_canonical), + }, + } + report["comparison"] = { + "elapsed_ratio": frontier_ms / max(1.0, range_ms), + "read_ratio": len(frontier_state["observations"]) / max(1, range_state["read_count"]), + "input_token_ratio": frontier_usage["input_tokens"] / max(1, range_usage["input_tokens"]), + "output_token_ratio": frontier_usage["output_tokens"] / max(1, range_usage["output_tokens"]), + "frontier_evidence_covered_by_range": overlap( + evidence_ranges(frontier_canonical), evidence_ranges(range_canonical) + ), + "range_evidence_covered_by_frontier": overlap( + evidence_ranges(range_canonical), evidence_ranges(frontier_canonical) + ), + } + + Path("artifacts/range/result.json").write_text( + json.dumps(range_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/range/localization-result.json").write_text( + json.dumps(range_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/result.json").write_text( + json.dumps(frontier_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/localization-result.json").write_text( + json.dumps(frontier_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/report.json").write_text( + json.dumps(report, indent=2, ensure_ascii=False) + "\n" + ) + print(json.dumps(report, indent=2, ensure_ascii=False)) + PY + + - name: Compare canonical results + run: | + set -euo pipefail + python3 harness/src/compare_localization_results.py \ + --left artifacts/range/localization-result.json \ + --right artifacts/frontier/localization-result.json \ + --output-json artifacts/localization-comparison.json \ + --output-markdown artifacts/localization-comparison.md + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + r = json.load(open("artifacts/report.json")) + print("## Single-file Range vs Frontier phase0-choice") + print("") + print("| Metric | Range | Frontier phase0-choice |") + print("| --- | ---: | ---: |") + print("| elapsed ms | %s | %s |" % (r["range"]["metrics"]["elapsed_ms"], r["frontier"]["metrics"]["elapsed_ms"])) + print("| model calls | %s | %s |" % (r["range"]["metrics"]["model_calls"], r["frontier"]["metrics"]["model_calls"])) + print("| input tokens | %s | %s |" % (r["range"]["metrics"]["input_tokens"], r["frontier"]["metrics"]["input_tokens"])) + print("| output tokens | %s | %s |" % (r["range"]["metrics"]["output_tokens"], r["frontier"]["metrics"]["output_tokens"])) + print("| reads | %s | %s |" % (r["range"]["metrics"]["reads_executed"], r["frontier"]["metrics"]["reads_executed"])) + print("| source coverage | %.1f%% | %.1f%% |" % (r["range"]["source_coverage"] * 100, r["frontier"]["source_coverage"] * 100)) + print("| frontier coverage | - | %.1f%% |" % (r["frontier"]["frontier_coverage"] * 100)) + print("| evidence regions | %s | %s |" % (r["range"]["metrics"]["evidence_regions"], r["frontier"]["metrics"]["evidence_regions"])) + print("| termination | %s | %s |" % (r["range"]["termination"], r["frontier"]["termination"])) + print("") + print("Frontier action counts:", r["frontier"]["action_counts"]) + PY + cat artifacts/localization-comparison.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload experiment + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-${{ github.run_id }} + path: artifacts/ + if-no-files-found: error + retention-days: 90 + + blind-quality: + name: Blind quality review + needs: experiment + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: fa517f9d5916754030f4d9b0068b075066a7c220 + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Download experiment + uses: actions/download-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-${{ github.run_id }} + path: artifacts + + - name: Prepare anonymous candidates + run: | + set -euo pipefail + mkdir -p artifacts/quality/range artifacts/quality/frontier + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/range/localization-result.json \ + --candidate-id candidate-a \ + --output-candidate artifacts/quality/range/candidate.json \ + --output-prompt artifacts/quality/range/prompt.txt + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/frontier/localization-result.json \ + --candidate-id candidate-b \ + --output-candidate artifacts/quality/frontier/candidate.json \ + --output-prompt artifacts/quality/frontier/prompt.txt + + - name: Evaluate range anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/range/prompt.txt \ + > ../artifacts/quality/range/evaluator.raw.jsonl + + - name: Evaluate frontier anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/frontier/prompt.txt \ + > ../artifacts/quality/frontier/evaluator.raw.jsonl + + - name: Finalize blind scorecards + run: | + set -euo pipefail + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/range/candidate.json \ + --raw-jsonl artifacts/quality/range/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/range/scorecard.json \ + --output-markdown artifacts/quality/range/report.md + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/frontier/candidate.json \ + --raw-jsonl artifacts/quality/frontier/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/frontier/scorecard.json \ + --output-markdown artifacts/quality/frontier/report.md + python3 harness/src/localization_quality_evaluation.py report \ + --candidate-a artifacts/quality/range/scorecard.json \ + --candidate-b artifacts/quality/frontier/scorecard.json \ + --candidate-a-label "Range runtime" \ + --candidate-b-label "Relevance frontier phase0-choice" \ + --output-json artifacts/quality/evaluation-report.json \ + --output-markdown artifacts/quality/evaluation-report.md + cat artifacts/quality/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload quality + if: always() + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-quality-${{ github.run_id }} + path: artifacts/quality/ + if-no-files-found: error + retention-days: 90 diff --git a/experiments/FULL_READ_BASELINE_TRIGGER.md b/experiments/FULL_READ_BASELINE_TRIGGER.md new file mode 100644 index 00000000..9cff40dd --- /dev/null +++ b/experiments/FULL_READ_BASELINE_TRIGGER.md @@ -0,0 +1 @@ +trigger: full-read Claude relevance baseline 2026-09-24 diff --git a/experiments/PHASE0_CHOICE_FRONTIER_TRIGGER.md b/experiments/PHASE0_CHOICE_FRONTIER_TRIGGER.md new file mode 100644 index 00000000..239f4320 --- /dev/null +++ b/experiments/PHASE0_CHOICE_FRONTIER_TRIGGER.md @@ -0,0 +1 @@ +trigger: phase0 coarse scan + Choice policy frontier 2026-09-24 diff --git a/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY.md b/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY.md new file mode 100644 index 00000000..c72e3042 --- /dev/null +++ b/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY.md @@ -0,0 +1 @@ +retry after harness test fixes diff --git a/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY2.md b/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY2.md new file mode 100644 index 00000000..dabbe41f --- /dev/null +++ b/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY2.md @@ -0,0 +1 @@ +retry v2 after multi-action test fixture fix diff --git a/experiments/PHASE0_CHOICE_FRONTIER_V2_TRIGGER.md b/experiments/PHASE0_CHOICE_FRONTIER_V2_TRIGGER.md new file mode 100644 index 00000000..aef25395 --- /dev/null +++ b/experiments/PHASE0_CHOICE_FRONTIER_V2_TRIGGER.md @@ -0,0 +1 @@ +trigger: phase0 coarse scan + multi-action Choice frontier v2 2026-09-24 diff --git a/experiments/single-file-frontier-v1b-trigger.md b/experiments/single-file-frontier-v1b-trigger.md new file mode 100644 index 00000000..8893a4d2 --- /dev/null +++ b/experiments/single-file-frontier-v1b-trigger.md @@ -0,0 +1 @@ +Trigger controlled single-file frontier v1b experiment. diff --git a/experiments/single-file-frontier-v1c-trigger.md b/experiments/single-file-frontier-v1c-trigger.md new file mode 100644 index 00000000..0e556d2c --- /dev/null +++ b/experiments/single-file-frontier-v1c-trigger.md @@ -0,0 +1 @@ +Trigger balanced v1c single-file experiment on agent websocket.