From 5cbe333b04cf22497381793e308a8ba44796fac6 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 11:39:06 +0800 Subject: [PATCH 01/18] experiment: compare single-file range and frontier v1b --- .../single-file-range-vs-frontier-v1b.yml | 418 ++++++++++++++++++ 1 file changed, 418 insertions(+) create mode 100644 .github/workflows/single-file-range-vs-frontier-v1b.yml diff --git a/.github/workflows/single-file-range-vs-frontier-v1b.yml b/.github/workflows/single-file-range-vs-frontier-v1b.yml new file mode 100644 index 00000000..04d2e72a --- /dev/null +++ b/.github/workflows/single-file-range-vs-frontier-v1b.yml @@ -0,0 +1,418 @@ +name: Single-file Range vs Relevance Frontier v1b + +on: + push: + branches: + - experiment/single-file-frontier-v1b-20260924 + +permissions: + contents: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + FILE_PATH: crates/nession-server/src/server/handler.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + HARNESS_SHA: cfc3d696df6c9ecc833372cf616805d2b931d0af + TYPESAFE_MODEL: jev-latest + TYPESAFE_API_URL: https://api.typesafe.ai/v1/systemone + +jobs: + experiment: + name: Single-file Range vs Frontier v1b + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: cfc3d696df6c9ecc833372cf616805d2b931d0af + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Run single-file experiment + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + run: | + set -euo pipefail + test -n "$TYPESAFE_API_KEY" + mkdir -p artifacts/range artifacts/frontier + python3 - <<'PY' + import json + import os + import sys + import time + from pathlib import Path + + sys.path.insert(0, str(Path("harness/src").resolve())) + from localization_result import ( + build_system_one_range_result, + build_system_one_relevance_frontier_result, + ) + from system_one_code_locator import Trace + from system_one_range_runtime import ( + SystemOneFileDecider, + canonical_file_results, + run_file_runtime, + ) + from system_one_relevance_frontier import ( + RelevanceFrontierDecider, + canonical_results as canonical_frontier_results, + run_frontier_file, + ) + + root = Path("subject") + query = os.environ["QUERY"] + model = os.environ["TYPESAFE_MODEL"] + endpoint = os.environ["TYPESAFE_API_URL"] + key = os.environ["TYPESAFE_API_KEY"] + file_path = os.environ["FILE_PATH"] + subject = { + "repository": os.environ["SUBJECT_REPOSITORY"], + "revision": os.environ["SUBJECT_SHA"], + } + candidate = { + "score": 0.90, + "payload": {"path": file_path, "extension": ".rs"}, + } + + range_params = { + "window_lines": 140, + "parallel_threshold": 0.65, + "max_jumps": 2, + "max_file_epochs": 32, + } + frontier_params = { + "max_rounds": 10, + "max_actions_per_round": 2, + "probe_lines": 128, + "target_region_lines": 64, + "final_window_lines": 24, + "refine_threshold": 0.78, + "candidate_threshold": 0.60, + "gradient_threshold": 0.14, + "volatility_threshold": 0.10, + "stable_delta": 0.06, + "stable_rounds": 2, + "max_frontier_leaves": 18, + "final_max_candidates": 16, + } + + def coverage(state): + rows = sorted((int(a), int(b)) for a, b in state.get("coverage", [])) + merged = [] + for a, b in rows: + if not merged or a > merged[-1][1] + 1: + merged.append([a, b]) + else: + merged[-1][1] = max(merged[-1][1], b) + return sum(b - a + 1 for a, b in merged) / max(1, int(state["line_count"])) + + def evidence_ranges(result): + return [ + (int(e["start_line"]), int(e["end_line"])) + for item in result.get("files", []) + for e in item.get("evidence", []) + ] + + def overlap(left, right): + total = sum(b - a + 1 for a, b in left) + if not total: + return 0.0 + hit = 0 + for a, b in left: + for c, d in right: + hit += max(0, min(b, d) - max(a, c) + 1) + return min(1.0, hit / total) + + def action_counts(state): + counts = {} + for snapshot in state.get("action_history", []): + for action in snapshot.get("actions", []): + kind = action.get("kind") + counts[kind] = counts.get(kind, 0) + 1 + return counts + + range_trace = Trace("artifacts/range/trace.jsonl") + range_decider = SystemOneFileDecider(key, range_trace, endpoint, model) + started = time.perf_counter() + range_state, range_usage = run_file_runtime( + root, query, candidate, range_decider, range_trace, **range_params + ) + range_ms = (time.perf_counter() - started) * 1000 + range_files = canonical_file_results([range_state], 0.65) + range_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [range_state], "result_files": range_files, + "metrics": { + **range_usage, + "reads_executed": range_state["read_count"], + "evidence_regions": sum(len(x["evidence"]) for x in range_files), + "elapsed_ms": round(range_ms, 3), + }, + } + range_canonical = build_system_one_range_result(range_engine, model) + + frontier_trace = Trace("artifacts/frontier/trace.jsonl") + frontier_decider = RelevanceFrontierDecider( + key, frontier_trace, endpoint, model + ) + started = time.perf_counter() + frontier_state, frontier_usage = run_frontier_file( + root, query, candidate, frontier_decider, frontier_trace, + **frontier_params + ) + frontier_ms = (time.perf_counter() - started) * 1000 + frontier_files = canonical_frontier_results([frontier_state]) + frontier_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [frontier_state], "result_files": frontier_files, + "metrics": { + **frontier_usage, + "reads_executed": len(frontier_state["observations"]), + "evidence_regions": sum(len(x["evidence"]) for x in frontier_files), + "elapsed_ms": round(frontier_ms, 3), + }, + } + frontier_canonical = build_system_one_relevance_frontier_result( + frontier_engine, model + ) + + report = { + "query": query, + "file": file_path, + "subject": subject, + "model": model, + "frontier_params": frontier_params, + "range_params": range_params, + "range": { + "metrics": range_engine["metrics"], + "termination": range_state["termination"], + "epochs": range_state["epoch"], + "source_coverage": coverage(range_state), + "evidence": evidence_ranges(range_canonical), + }, + "frontier": { + "metrics": frontier_engine["metrics"], + "termination": frontier_state["termination"], + "rounds": frontier_state["round"], + "frontier_coverage": frontier_state["frontier_coverage"], + "source_coverage": frontier_state["source_coverage"], + "action_counts": action_counts(frontier_state), + "evidence": evidence_ranges(frontier_canonical), + }, + } + report["comparison"] = { + "elapsed_ratio": frontier_ms / max(1.0, range_ms), + "read_ratio": len(frontier_state["observations"]) / max(1, range_state["read_count"]), + "input_token_ratio": frontier_usage["input_tokens"] / max(1, range_usage["input_tokens"]), + "output_token_ratio": frontier_usage["output_tokens"] / max(1, range_usage["output_tokens"]), + "frontier_evidence_covered_by_range": overlap( + evidence_ranges(frontier_canonical), evidence_ranges(range_canonical) + ), + "range_evidence_covered_by_frontier": overlap( + evidence_ranges(range_canonical), evidence_ranges(frontier_canonical) + ), + } + + Path("artifacts/range/result.json").write_text( + json.dumps(range_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/range/localization-result.json").write_text( + json.dumps(range_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/result.json").write_text( + json.dumps(frontier_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/localization-result.json").write_text( + json.dumps(frontier_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/report.json").write_text( + json.dumps(report, indent=2, ensure_ascii=False) + "\n" + ) + print(json.dumps(report, indent=2, ensure_ascii=False)) + PY + + - name: Compare canonical results + run: | + set -euo pipefail + python3 harness/src/compare_localization_results.py \ + --left artifacts/range/localization-result.json \ + --right artifacts/frontier/localization-result.json \ + --output-json artifacts/localization-comparison.json \ + --output-markdown artifacts/localization-comparison.md + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + r = json.load(open("artifacts/report.json")) + print("## Single-file Range vs Frontier v1b") + print("") + print("| Metric | Range | Frontier v1b |") + print("| --- | ---: | ---: |") + print("| elapsed ms | %s | %s |" % (r["range"]["metrics"]["elapsed_ms"], r["frontier"]["metrics"]["elapsed_ms"])) + print("| model calls | %s | %s |" % (r["range"]["metrics"]["model_calls"], r["frontier"]["metrics"]["model_calls"])) + print("| input tokens | %s | %s |" % (r["range"]["metrics"]["input_tokens"], r["frontier"]["metrics"]["input_tokens"])) + print("| output tokens | %s | %s |" % (r["range"]["metrics"]["output_tokens"], r["frontier"]["metrics"]["output_tokens"])) + print("| reads | %s | %s |" % (r["range"]["metrics"]["reads_executed"], r["frontier"]["metrics"]["reads_executed"])) + print("| source coverage | %.1f%% | %.1f%% |" % (r["range"]["source_coverage"] * 100, r["frontier"]["source_coverage"] * 100)) + print("| frontier coverage | - | %.1f%% |" % (r["frontier"]["frontier_coverage"] * 100)) + print("| evidence regions | %s | %s |" % (r["range"]["metrics"]["evidence_regions"], r["frontier"]["metrics"]["evidence_regions"])) + print("| termination | %s | %s |" % (r["range"]["termination"], r["frontier"]["termination"])) + print("") + print("Frontier action counts:", r["frontier"]["action_counts"]) + PY + cat artifacts/localization-comparison.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload experiment + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-frontier-v1b-${{ github.run_id }} + path: artifacts/ + if-no-files-found: error + retention-days: 90 + + blind-quality: + name: Blind quality review + needs: experiment + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: cfc3d696df6c9ecc833372cf616805d2b931d0af + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Download experiment + uses: actions/download-artifact@v4 + with: + name: single-file-range-vs-frontier-v1b-${{ github.run_id }} + path: artifacts + + - name: Prepare anonymous candidates + run: | + set -euo pipefail + mkdir -p artifacts/quality/range artifacts/quality/frontier + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/range/localization-result.json \ + --candidate-id candidate-a \ + --output-candidate artifacts/quality/range/candidate.json \ + --output-prompt artifacts/quality/range/prompt.txt + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/frontier/localization-result.json \ + --candidate-id candidate-b \ + --output-candidate artifacts/quality/frontier/candidate.json \ + --output-prompt artifacts/quality/frontier/prompt.txt + + - name: Evaluate range anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/range/prompt.txt \ + > ../artifacts/quality/range/evaluator.raw.jsonl + + - name: Evaluate frontier anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/frontier/prompt.txt \ + > ../artifacts/quality/frontier/evaluator.raw.jsonl + + - name: Finalize blind scorecards + run: | + set -euo pipefail + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/range/candidate.json \ + --raw-jsonl artifacts/quality/range/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/range/scorecard.json \ + --output-markdown artifacts/quality/range/report.md + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/frontier/candidate.json \ + --raw-jsonl artifacts/quality/frontier/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/frontier/scorecard.json \ + --output-markdown artifacts/quality/frontier/report.md + python3 harness/src/localization_quality_evaluation.py report \ + --candidate-a artifacts/quality/range/scorecard.json \ + --candidate-b artifacts/quality/frontier/scorecard.json \ + --candidate-a-label "Range runtime" \ + --candidate-b-label "Relevance frontier v1b" \ + --output-json artifacts/quality/evaluation-report.json \ + --output-markdown artifacts/quality/evaluation-report.md + cat artifacts/quality/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload quality + if: always() + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-frontier-v1b-quality-${{ github.run_id }} + path: artifacts/quality/ + if-no-files-found: error + retention-days: 90 From 6d94f280547f3156ba5e32ea5cae328e8bc74983 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 11:39:09 +0800 Subject: [PATCH 02/18] experiment: trigger single-file frontier v1b --- experiments/single-file-frontier-v1b-trigger.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/single-file-frontier-v1b-trigger.md diff --git a/experiments/single-file-frontier-v1b-trigger.md b/experiments/single-file-frontier-v1b-trigger.md new file mode 100644 index 00000000..8893a4d2 --- /dev/null +++ b/experiments/single-file-frontier-v1b-trigger.md @@ -0,0 +1 @@ +Trigger controlled single-file frontier v1b experiment. From 06911c2ff7eb17e32464943cf5cdef45c10940b6 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 11:41:05 +0800 Subject: [PATCH 03/18] experiment: run balanced frontier v1c on agent websocket --- .../single-file-range-vs-frontier-v1b.yml | 36 +++++++++---------- 1 file changed, 18 insertions(+), 18 deletions(-) diff --git a/.github/workflows/single-file-range-vs-frontier-v1b.yml b/.github/workflows/single-file-range-vs-frontier-v1b.yml index 04d2e72a..3202f466 100644 --- a/.github/workflows/single-file-range-vs-frontier-v1b.yml +++ b/.github/workflows/single-file-range-vs-frontier-v1b.yml @@ -1,9 +1,9 @@ -name: Single-file Range vs Relevance Frontier v1b +name: Single-file Range vs Relevance Frontier v1c on: push: branches: - - experiment/single-file-frontier-v1b-20260924 + - experiment/single-file-frontier-v1c-20260924 permissions: contents: read @@ -12,7 +12,7 @@ env: QUERY: Help me optimize the websocket connection implementation SUBJECT_REPOSITORY: BestNathan/nession SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df - FILE_PATH: crates/nession-server/src/server/handler.rs + FILE_PATH: crates/nession-agent/src/server/websocket.rs HARNESS_REPOSITORY: BestNathan/system-one-code-explore HARNESS_SHA: cfc3d696df6c9ecc833372cf616805d2b931d0af TYPESAFE_MODEL: jev-latest @@ -20,7 +20,7 @@ env: jobs: experiment: - name: Single-file Range vs Frontier v1b + name: Single-file Range vs Frontier v1c runs-on: ubuntu-24.04 environment: typesafe timeout-minutes: 120 @@ -102,17 +102,17 @@ jobs: frontier_params = { "max_rounds": 10, "max_actions_per_round": 2, - "probe_lines": 128, - "target_region_lines": 64, - "final_window_lines": 24, - "refine_threshold": 0.78, - "candidate_threshold": 0.60, - "gradient_threshold": 0.14, + "probe_lines": 112, + "target_region_lines": 48, + "final_window_lines": 32, + "refine_threshold": 0.72, + "candidate_threshold": 0.55, + "gradient_threshold": 0.15, "volatility_threshold": 0.10, "stable_delta": 0.06, "stable_rounds": 2, - "max_frontier_leaves": 18, - "final_max_candidates": 16, + "max_frontier_leaves": 24, + "final_max_candidates": 24, } def coverage(state): @@ -261,9 +261,9 @@ jobs: python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" import json r = json.load(open("artifacts/report.json")) - print("## Single-file Range vs Frontier v1b") + print("## Single-file Range vs Frontier v1c") print("") - print("| Metric | Range | Frontier v1b |") + print("| Metric | Range | Frontier v1c |") print("| --- | ---: | ---: |") print("| elapsed ms | %s | %s |" % (r["range"]["metrics"]["elapsed_ms"], r["frontier"]["metrics"]["elapsed_ms"])) print("| model calls | %s | %s |" % (r["range"]["metrics"]["model_calls"], r["frontier"]["metrics"]["model_calls"])) @@ -282,7 +282,7 @@ jobs: - name: Upload experiment uses: actions/upload-artifact@v4 with: - name: single-file-range-vs-frontier-v1b-${{ github.run_id }} + name: single-file-range-vs-frontier-v1c-${{ github.run_id }} path: artifacts/ if-no-files-found: error retention-days: 90 @@ -326,7 +326,7 @@ jobs: - name: Download experiment uses: actions/download-artifact@v4 with: - name: single-file-range-vs-frontier-v1b-${{ github.run_id }} + name: single-file-range-vs-frontier-v1c-${{ github.run_id }} path: artifacts - name: Prepare anonymous candidates @@ -403,7 +403,7 @@ jobs: --candidate-a artifacts/quality/range/scorecard.json \ --candidate-b artifacts/quality/frontier/scorecard.json \ --candidate-a-label "Range runtime" \ - --candidate-b-label "Relevance frontier v1b" \ + --candidate-b-label "Relevance frontier v1c" \ --output-json artifacts/quality/evaluation-report.json \ --output-markdown artifacts/quality/evaluation-report.md cat artifacts/quality/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" @@ -412,7 +412,7 @@ jobs: if: always() uses: actions/upload-artifact@v4 with: - name: single-file-range-vs-frontier-v1b-quality-${{ github.run_id }} + name: single-file-range-vs-frontier-v1c-quality-${{ github.run_id }} path: artifacts/quality/ if-no-files-found: error retention-days: 90 From 27e0ee13a32a2ffd0f901ea8051ef1b62a404fc2 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 11:41:12 +0800 Subject: [PATCH 04/18] fix: keep v1b experiment branch trigger --- .github/workflows/single-file-range-vs-frontier-v1b.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/single-file-range-vs-frontier-v1b.yml b/.github/workflows/single-file-range-vs-frontier-v1b.yml index 3202f466..48292520 100644 --- a/.github/workflows/single-file-range-vs-frontier-v1b.yml +++ b/.github/workflows/single-file-range-vs-frontier-v1b.yml @@ -3,7 +3,7 @@ name: Single-file Range vs Relevance Frontier v1c on: push: branches: - - experiment/single-file-frontier-v1c-20260924 + - experiment/single-file-frontier-v1b-20260924 permissions: contents: read From 4ad20589fe064f66c936afb342e0656def5a79b1 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 11:41:15 +0800 Subject: [PATCH 05/18] experiment: trigger balanced frontier v1c --- experiments/single-file-frontier-v1c-trigger.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/single-file-frontier-v1c-trigger.md diff --git a/experiments/single-file-frontier-v1c-trigger.md b/experiments/single-file-frontier-v1c-trigger.md new file mode 100644 index 00000000..0e556d2c --- /dev/null +++ b/experiments/single-file-frontier-v1c-trigger.md @@ -0,0 +1 @@ +Trigger balanced v1c single-file experiment on agent websocket. From a3aade60a96a9c79000a501035c02d4bca6a2077 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:20:56 +0800 Subject: [PATCH 06/18] experiment: add full-read Claude relevance baseline workflow --- .../workflows/full-read-claude-baseline.yml | 156 ++++++++++++++++++ 1 file changed, 156 insertions(+) create mode 100644 .github/workflows/full-read-claude-baseline.yml diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml new file mode 100644 index 00000000..bb56ebe1 --- /dev/null +++ b/.github/workflows/full-read-claude-baseline.yml @@ -0,0 +1,156 @@ +name: Full-read Claude relevance baseline + +on: + push: + branches: + - experiment/full-read-baseline-20260924 + +permissions: + contents: read + actions: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706b2b1a1df + FILE_PATH: crates/nession-agent/src/server/websocket.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + HARNESS_SHA: f93b57d81e9e3f530ec1a26f23bb2472aa90d1b + V1C_RUN_ID: "35952482145" + V1C_ARTIFACT: single-file-range-vs-frontier-v1c-35952482145 + +jobs: + baseline: + name: Build full-read Claude reference field + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: f93b57d81e9e3f530ec1a26f23bb2472aa90d1b + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - name: Setup Python + uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Setup Node + uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Prepare reference prompt + run: | + set -euo pipefail + mkdir -p artifacts/baseline + python3 harness/src/full_read_relevance_baseline.py prepare \ + --source "subject/$FILE_PATH" \ + --query "$QUERY" \ + --window-lines 64 \ + --stride-lines 32 \ + --output-prompt artifacts/baseline/prompt.txt \ + --output-manifest artifacts/baseline/manifest.json + + - name: Generate full-read Claude field + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + test -n "$ANTHROPIC_API_KEY" + test -n "$CLAUDE_MODEL" + claude -p \ + --output-format text \ + --model "$CLAUDE_MODEL" \ + --effort high \ + --max-turns 4 \ + --permission-mode dontAsk \ + --no-session-persistence \ + --setting-sources '' \ + --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' \ + --disable-slash-commands \ + --tools '' \ + --settings '{"disableAllHooks":true}' \ + < artifacts/baseline/prompt.txt \ + > artifacts/baseline/claude-output.txt + + - name: Normalize reference field + run: | + set -euo pipefail + python3 harness/src/full_read_relevance_baseline.py normalize \ + --source "subject/$FILE_PATH" \ + --query "$QUERY" \ + --window-lines 64 \ + --stride-lines 32 \ + --raw-output artifacts/baseline/claude-output.txt \ + --output artifacts/baseline/reference.json + + - name: Download v1c candidate + uses: actions/download-artifact@v5 + with: + name: single-file-range-vs-frontier-v1c-35952482145 + path: artifacts/v1c + run-id: 35952482145 + github-token: ${{ secrets.GITHUB_TOKEN }} + + - name: Project v1c results onto reference field + run: | + set -euo pipefail + python3 harness/src/compare_to_full_read_baseline.py \ + --reference artifacts/baseline/reference.json \ + --candidate artifacts/v1c/frontier/localization-result.json \ + --output-json artifacts/baseline/v1c-frontier-reference-metrics.json + python3 harness/src/compare_to_full_read_baseline.py \ + --reference artifacts/baseline/reference.json \ + --candidate artifacts/v1c/range/localization-result.json \ + --output-json artifacts/baseline/v1c-range-reference-metrics.json + + - name: Summarize + run: | + set -euo pipefail + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + for label, path in [ + ("Range v1c", "artifacts/baseline/v1c-range-reference-metrics.json"), + ("Frontier v1c", "artifacts/baseline/v1c-frontier-reference-metrics.json"), + ]: + r = json.load(open(path)) + m = r["metrics"] + print(f"## {label} vs full-read reference") + print("") + print("| Metric | Value |") + print("| --- | ---: |") + print(f"| weighted relevance recall | {m['weighted_relevance_recall']:.3f} |") + print(f"| high-relevance window recall | {m['high_relevance_window_recall']:.3f} |") + print(f"| core-window recall | {m['core_window_recall']:.3f} |") + print(f"| relevance-weighted precision | {m['relevance_weighted_precision']:.3f} |") + print(f"| source coverage | {r['candidate']['source_coverage']:.3f} |") + print("") + PY + + - name: Upload baseline + uses: actions/upload-artifact@v4 + with: + name: full-read-claude-baseline-$GITHUB_RUN_ID + path: artifacts/baseline + if-no-files-found: error + retention-days: 90 From 60d157b92e6e13410b250ca2f2cc60fae0d28bde Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:21:01 +0800 Subject: [PATCH 07/18] fix: pin frozen subject and stable baseline artifact name --- .github/workflows/full-read-claude-baseline.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml index bb56ebe1..4d124874 100644 --- a/.github/workflows/full-read-claude-baseline.yml +++ b/.github/workflows/full-read-claude-baseline.yml @@ -12,7 +12,7 @@ permissions: env: QUERY: Help me optimize the websocket connection implementation SUBJECT_REPOSITORY: BestNathan/nession - SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706b2b1a1df + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df FILE_PATH: crates/nession-agent/src/server/websocket.rs HARNESS_REPOSITORY: BestNathan/system-one-code-explore HARNESS_SHA: f93b57d81e9e3f530ec1a26f23bb2472aa90d1b @@ -150,7 +150,7 @@ jobs: - name: Upload baseline uses: actions/upload-artifact@v4 with: - name: full-read-claude-baseline-$GITHUB_RUN_ID + name: full-read-claude-baseline path: artifacts/baseline if-no-files-found: error retention-days: 90 From f3baff1f9d00bf89c27ea78037e18eb28c472c7b Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:21:12 +0800 Subject: [PATCH 08/18] experiment: trigger full-read Claude baseline --- experiments/FULL_READ_BASELINE_TRIGGER.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/FULL_READ_BASELINE_TRIGGER.md diff --git a/experiments/FULL_READ_BASELINE_TRIGGER.md b/experiments/FULL_READ_BASELINE_TRIGGER.md new file mode 100644 index 00000000..9cff40dd --- /dev/null +++ b/experiments/FULL_READ_BASELINE_TRIGGER.md @@ -0,0 +1 @@ +trigger: full-read Claude relevance baseline 2026-09-24 From cf28506672ed4c155b62c39df8632f796d451ae2 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:23:22 +0800 Subject: [PATCH 09/18] experiment: add phase0 coarse scan and Choice frontier benchmark --- .../single-file-range-vs-phase0-choice.yml | 417 ++++++++++++++++++ 1 file changed, 417 insertions(+) create mode 100644 .github/workflows/single-file-range-vs-phase0-choice.yml diff --git a/.github/workflows/single-file-range-vs-phase0-choice.yml b/.github/workflows/single-file-range-vs-phase0-choice.yml new file mode 100644 index 00000000..fac721c6 --- /dev/null +++ b/.github/workflows/single-file-range-vs-phase0-choice.yml @@ -0,0 +1,417 @@ +name: Single-file Range vs Phase0 + Choice Frontier + +on: + push: + branches: + - experiment/phase0-choice-frontier-20260924 + +permissions: + contents: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + FILE_PATH: crates/nession-agent/src/server/websocket.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + HARNESS_SHA: fa517f9d5916754030f4d9b0068b075066a7c220 + TYPESAFE_MODEL: jev-latest + TYPESAFE_API_URL: https://api.typesafe.ai/v1/systemone + +jobs: + experiment: + name: Single-file Range vs Frontier phase0-choice + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: fa517f9d5916754030f4d9b0068b075066a7c220 + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Run single-file experiment + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + run: | + set -euo pipefail + test -n "$TYPESAFE_API_KEY" + mkdir -p artifacts/range artifacts/frontier + python3 - <<'PY' + import json + import os + import sys + import time + from pathlib import Path + + sys.path.insert(0, str(Path("harness/src").resolve())) + from localization_result import ( + build_system_one_range_result, + build_system_one_relevance_frontier_result, + ) + from system_one_code_locator import Trace + from system_one_range_runtime import ( + SystemOneFileDecider, + canonical_file_results, + run_file_runtime, + ) + from system_one_relevance_frontier import ( + ChoiceRelevanceFrontierDecider, + canonical_results as canonical_frontier_results, + run_frontier_file_phase0_choice, + ) + + root = Path("subject") + query = os.environ["QUERY"] + model = os.environ["TYPESAFE_MODEL"] + endpoint = os.environ["TYPESAFE_API_URL"] + key = os.environ["TYPESAFE_API_KEY"] + file_path = os.environ["FILE_PATH"] + subject = { + "repository": os.environ["SUBJECT_REPOSITORY"], + "revision": os.environ["SUBJECT_SHA"], + } + candidate = { + "score": 0.90, + "payload": {"path": file_path, "extension": ".rs"}, + } + + range_params = { + "window_lines": 140, + "parallel_threshold": 0.65, + "max_jumps": 2, + "max_file_epochs": 32, + } + frontier_params = { + "max_rounds": 10, + "probe_lines": 112, + "target_region_lines": 48, + "final_window_lines": 32, + "refine_threshold": 0.72, + "candidate_threshold": 0.55, + "gradient_threshold": 0.15, + "volatility_threshold": 0.10, + "stable_delta": 0.06, + "stable_rounds": 2, + "max_frontier_leaves": 24, + "final_max_candidates": 24, + } + + def coverage(state): + rows = sorted((int(a), int(b)) for a, b in state.get("coverage", [])) + merged = [] + for a, b in rows: + if not merged or a > merged[-1][1] + 1: + merged.append([a, b]) + else: + merged[-1][1] = max(merged[-1][1], b) + return sum(b - a + 1 for a, b in merged) / max(1, int(state["line_count"])) + + def evidence_ranges(result): + return [ + (int(e["start_line"]), int(e["end_line"])) + for item in result.get("files", []) + for e in item.get("evidence", []) + ] + + def overlap(left, right): + total = sum(b - a + 1 for a, b in left) + if not total: + return 0.0 + hit = 0 + for a, b in left: + for c, d in right: + hit += max(0, min(b, d) - max(a, c) + 1) + return min(1.0, hit / total) + + def action_counts(state): + counts = {} + for snapshot in state.get("action_history", []): + for action in snapshot.get("actions", []): + kind = action.get("kind") + counts[kind] = counts.get(kind, 0) + 1 + return counts + + range_trace = Trace("artifacts/range/trace.jsonl") + range_decider = SystemOneFileDecider(key, range_trace, endpoint, model) + started = time.perf_counter() + range_state, range_usage = run_file_runtime( + root, query, candidate, range_decider, range_trace, **range_params + ) + range_ms = (time.perf_counter() - started) * 1000 + range_files = canonical_file_results([range_state], 0.65) + range_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [range_state], "result_files": range_files, + "metrics": { + **range_usage, + "reads_executed": range_state["read_count"], + "evidence_regions": sum(len(x["evidence"]) for x in range_files), + "elapsed_ms": round(range_ms, 3), + }, + } + range_canonical = build_system_one_range_result(range_engine, model) + + frontier_trace = Trace("artifacts/frontier/trace.jsonl") + frontier_decider = ChoiceRelevanceFrontierDecider( + key, frontier_trace, endpoint, model + ) + started = time.perf_counter() + frontier_state, frontier_usage = run_frontier_file_phase0_choice( + root, query, candidate, frontier_decider, frontier_trace, + **frontier_params + ) + frontier_ms = (time.perf_counter() - started) * 1000 + frontier_files = canonical_frontier_results([frontier_state]) + frontier_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [frontier_state], "result_files": frontier_files, + "metrics": { + **frontier_usage, + "reads_executed": len(frontier_state["observations"]), + "evidence_regions": sum(len(x["evidence"]) for x in frontier_files), + "elapsed_ms": round(frontier_ms, 3), + }, + } + frontier_canonical = build_system_one_relevance_frontier_result( + frontier_engine, model + ) + + report = { + "query": query, + "file": file_path, + "subject": subject, + "model": model, + "frontier_params": frontier_params, + "range_params": range_params, + "range": { + "metrics": range_engine["metrics"], + "termination": range_state["termination"], + "epochs": range_state["epoch"], + "source_coverage": coverage(range_state), + "evidence": evidence_ranges(range_canonical), + }, + "frontier": { + "metrics": frontier_engine["metrics"], + "termination": frontier_state["termination"], + "rounds": frontier_state["round"], + "frontier_coverage": frontier_state["frontier_coverage"], + "source_coverage": frontier_state["source_coverage"], + "action_counts": action_counts(frontier_state), + "evidence": evidence_ranges(frontier_canonical), + }, + } + report["comparison"] = { + "elapsed_ratio": frontier_ms / max(1.0, range_ms), + "read_ratio": len(frontier_state["observations"]) / max(1, range_state["read_count"]), + "input_token_ratio": frontier_usage["input_tokens"] / max(1, range_usage["input_tokens"]), + "output_token_ratio": frontier_usage["output_tokens"] / max(1, range_usage["output_tokens"]), + "frontier_evidence_covered_by_range": overlap( + evidence_ranges(frontier_canonical), evidence_ranges(range_canonical) + ), + "range_evidence_covered_by_frontier": overlap( + evidence_ranges(range_canonical), evidence_ranges(frontier_canonical) + ), + } + + Path("artifacts/range/result.json").write_text( + json.dumps(range_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/range/localization-result.json").write_text( + json.dumps(range_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/result.json").write_text( + json.dumps(frontier_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/localization-result.json").write_text( + json.dumps(frontier_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/report.json").write_text( + json.dumps(report, indent=2, ensure_ascii=False) + "\n" + ) + print(json.dumps(report, indent=2, ensure_ascii=False)) + PY + + - name: Compare canonical results + run: | + set -euo pipefail + python3 harness/src/compare_localization_results.py \ + --left artifacts/range/localization-result.json \ + --right artifacts/frontier/localization-result.json \ + --output-json artifacts/localization-comparison.json \ + --output-markdown artifacts/localization-comparison.md + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + r = json.load(open("artifacts/report.json")) + print("## Single-file Range vs Frontier phase0-choice") + print("") + print("| Metric | Range | Frontier phase0-choice |") + print("| --- | ---: | ---: |") + print("| elapsed ms | %s | %s |" % (r["range"]["metrics"]["elapsed_ms"], r["frontier"]["metrics"]["elapsed_ms"])) + print("| model calls | %s | %s |" % (r["range"]["metrics"]["model_calls"], r["frontier"]["metrics"]["model_calls"])) + print("| input tokens | %s | %s |" % (r["range"]["metrics"]["input_tokens"], r["frontier"]["metrics"]["input_tokens"])) + print("| output tokens | %s | %s |" % (r["range"]["metrics"]["output_tokens"], r["frontier"]["metrics"]["output_tokens"])) + print("| reads | %s | %s |" % (r["range"]["metrics"]["reads_executed"], r["frontier"]["metrics"]["reads_executed"])) + print("| source coverage | %.1f%% | %.1f%% |" % (r["range"]["source_coverage"] * 100, r["frontier"]["source_coverage"] * 100)) + print("| frontier coverage | - | %.1f%% |" % (r["frontier"]["frontier_coverage"] * 100)) + print("| evidence regions | %s | %s |" % (r["range"]["metrics"]["evidence_regions"], r["frontier"]["metrics"]["evidence_regions"])) + print("| termination | %s | %s |" % (r["range"]["termination"], r["frontier"]["termination"])) + print("") + print("Frontier action counts:", r["frontier"]["action_counts"]) + PY + cat artifacts/localization-comparison.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload experiment + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-${{ github.run_id }} + path: artifacts/ + if-no-files-found: error + retention-days: 90 + + blind-quality: + name: Blind quality review + needs: experiment + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: fa517f9d5916754030f4d9b0068b075066a7c220 + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Download experiment + uses: actions/download-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-${{ github.run_id }} + path: artifacts + + - name: Prepare anonymous candidates + run: | + set -euo pipefail + mkdir -p artifacts/quality/range artifacts/quality/frontier + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/range/localization-result.json \ + --candidate-id candidate-a \ + --output-candidate artifacts/quality/range/candidate.json \ + --output-prompt artifacts/quality/range/prompt.txt + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/frontier/localization-result.json \ + --candidate-id candidate-b \ + --output-candidate artifacts/quality/frontier/candidate.json \ + --output-prompt artifacts/quality/frontier/prompt.txt + + - name: Evaluate range anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/range/prompt.txt \ + > ../artifacts/quality/range/evaluator.raw.jsonl + + - name: Evaluate frontier anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/frontier/prompt.txt \ + > ../artifacts/quality/frontier/evaluator.raw.jsonl + + - name: Finalize blind scorecards + run: | + set -euo pipefail + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/range/candidate.json \ + --raw-jsonl artifacts/quality/range/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/range/scorecard.json \ + --output-markdown artifacts/quality/range/report.md + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/frontier/candidate.json \ + --raw-jsonl artifacts/quality/frontier/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/frontier/scorecard.json \ + --output-markdown artifacts/quality/frontier/report.md + python3 harness/src/localization_quality_evaluation.py report \ + --candidate-a artifacts/quality/range/scorecard.json \ + --candidate-b artifacts/quality/frontier/scorecard.json \ + --candidate-a-label "Range runtime" \ + --candidate-b-label "Relevance frontier phase0-choice" \ + --output-json artifacts/quality/evaluation-report.json \ + --output-markdown artifacts/quality/evaluation-report.md + cat artifacts/quality/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload quality + if: always() + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-quality-${{ github.run_id }} + path: artifacts/quality/ + if-no-files-found: error + retention-days: 90 From 2525d2c9a787fd86c45c4e9982ce5951f76b8f8d Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:23:31 +0800 Subject: [PATCH 10/18] fix: record Choice policy action counts --- .github/workflows/single-file-range-vs-phase0-choice.yml | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/.github/workflows/single-file-range-vs-phase0-choice.yml b/.github/workflows/single-file-range-vs-phase0-choice.yml index fac721c6..7352ef5e 100644 --- a/.github/workflows/single-file-range-vs-phase0-choice.yml +++ b/.github/workflows/single-file-range-vs-phase0-choice.yml @@ -144,9 +144,15 @@ jobs: def action_counts(state): counts = {} for snapshot in state.get("action_history", []): + policy_action = snapshot.get("policy", {}).get("action") + if isinstance(policy_action, dict): + kind = policy_action.get("kind") + if kind: + counts[kind] = counts.get(kind, 0) + 1 for action in snapshot.get("actions", []): kind = action.get("kind") - counts[kind] = counts.get(kind, 0) + 1 + if kind: + counts[kind] = counts.get(kind, 0) + 1 return counts range_trace = Trace("artifacts/range/trace.jsonl") From 19bf460e1c9e09766ea0bbd2ebfb5db46a025158 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:23:35 +0800 Subject: [PATCH 11/18] experiment: trigger phase0 Choice frontier benchmark --- experiments/PHASE0_CHOICE_FRONTIER_TRIGGER.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/PHASE0_CHOICE_FRONTIER_TRIGGER.md diff --git a/experiments/PHASE0_CHOICE_FRONTIER_TRIGGER.md b/experiments/PHASE0_CHOICE_FRONTIER_TRIGGER.md new file mode 100644 index 00000000..239f4320 --- /dev/null +++ b/experiments/PHASE0_CHOICE_FRONTIER_TRIGGER.md @@ -0,0 +1 @@ +trigger: phase0 coarse scan + Choice policy frontier 2026-09-24 From 1b6f536c4e2ca56ad36ecb12fd8218863672bd32 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:26:53 +0800 Subject: [PATCH 12/18] experiment: benchmark Phase0 + multi-action Choice frontier v2 --- .../single-file-range-vs-phase0-choice-v2.yml | 424 ++++++++++++++++++ 1 file changed, 424 insertions(+) create mode 100644 .github/workflows/single-file-range-vs-phase0-choice-v2.yml diff --git a/.github/workflows/single-file-range-vs-phase0-choice-v2.yml b/.github/workflows/single-file-range-vs-phase0-choice-v2.yml new file mode 100644 index 00000000..15d6ec59 --- /dev/null +++ b/.github/workflows/single-file-range-vs-phase0-choice-v2.yml @@ -0,0 +1,424 @@ +name: Single-file Range vs Phase0 + Choice Frontier v2 + +on: + push: + branches: + - experiment/phase0-choice-v2-frontier-v2-20260924 + +permissions: + contents: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + FILE_PATH: crates/nession-agent/src/server/websocket.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + HARNESS_SHA: 0aa52ee0376c18b8452d2becf584a6f919f8b018 + TYPESAFE_MODEL: jev-latest + TYPESAFE_API_URL: https://api.typesafe.ai/v1/systemone + +jobs: + experiment: + name: Single-file Range vs Frontier phase0-choice-v2 + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: 0aa52ee0376c18b8452d2becf584a6f919f8b018 + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Run single-file experiment + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + run: | + set -euo pipefail + test -n "$TYPESAFE_API_KEY" + mkdir -p artifacts/range artifacts/frontier + python3 - <<'PY' + import json + import os + import sys + import time + from pathlib import Path + + sys.path.insert(0, str(Path("harness/src").resolve())) + from localization_result import ( + build_system_one_range_result, + build_system_one_relevance_frontier_result, + ) + from system_one_code_locator import Trace + from system_one_range_runtime import ( + SystemOneFileDecider, + canonical_file_results, + run_file_runtime, + ) + from system_one_relevance_frontier import ( + ChoiceRelevanceFrontierDecider, + canonical_results as canonical_frontier_results, + run_frontier_file_phase0_choice_v2, + ) + + root = Path("subject") + query = os.environ["QUERY"] + model = os.environ["TYPESAFE_MODEL"] + endpoint = os.environ["TYPESAFE_API_URL"] + key = os.environ["TYPESAFE_API_KEY"] + file_path = os.environ["FILE_PATH"] + subject = { + "repository": os.environ["SUBJECT_REPOSITORY"], + "revision": os.environ["SUBJECT_SHA"], + } + candidate = { + "score": 0.90, + "payload": {"path": file_path, "extension": ".rs"}, + } + + range_params = { + "window_lines": 140, + "parallel_threshold": 0.65, + "max_jumps": 2, + "max_file_epochs": 32, + } + frontier_params = { + "max_rounds": 10, + "max_actions_per_choice": 6, + "probe_lines": 112, + "target_region_lines": 48, + "final_window_lines": 32, + "refine_threshold": 0.72, + "candidate_threshold": 0.55, + "gradient_threshold": 0.15, + "volatility_threshold": 0.10, + "stable_delta": 0.06, + "stable_rounds": 2, + "max_frontier_leaves": 24, + "final_max_candidates": 24, + } + + def coverage(state): + rows = sorted((int(a), int(b)) for a, b in state.get("coverage", [])) + merged = [] + for a, b in rows: + if not merged or a > merged[-1][1] + 1: + merged.append([a, b]) + else: + merged[-1][1] = max(merged[-1][1], b) + return sum(b - a + 1 for a, b in merged) / max(1, int(state["line_count"])) + + def evidence_ranges(result): + return [ + (int(e["start_line"]), int(e["end_line"])) + for item in result.get("files", []) + for e in item.get("evidence", []) + ] + + def overlap(left, right): + total = sum(b - a + 1 for a, b in left) + if not total: + return 0.0 + hit = 0 + for a, b in left: + for c, d in right: + hit += max(0, min(b, d) - max(a, c) + 1) + return min(1.0, hit / total) + + def action_counts(state): + counts = {} + for snapshot in state.get("action_history", []): + policy_action = snapshot.get("policy", {}).get("action") + if isinstance(policy_action, dict): + kind = policy_action.get("kind") + if kind: + counts[kind] = counts.get(kind, 0) + 1 + for action in snapshot.get("actions", []): + kind = action.get("kind") + if kind: + counts[kind] = counts.get(kind, 0) + 1 + return counts + + range_trace = Trace("artifacts/range/trace.jsonl") + range_decider = SystemOneFileDecider(key, range_trace, endpoint, model) + started = time.perf_counter() + range_state, range_usage = run_file_runtime( + root, query, candidate, range_decider, range_trace, **range_params + ) + range_ms = (time.perf_counter() - started) * 1000 + range_files = canonical_file_results([range_state], 0.65) + range_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [range_state], "result_files": range_files, + "metrics": { + **range_usage, + "reads_executed": range_state["read_count"], + "evidence_regions": sum(len(x["evidence"]) for x in range_files), + "elapsed_ms": round(range_ms, 3), + }, + } + range_canonical = build_system_one_range_result(range_engine, model) + + frontier_trace = Trace("artifacts/frontier/trace.jsonl") + frontier_decider = ChoiceRelevanceFrontierDecider( + key, frontier_trace, endpoint, model + ) + started = time.perf_counter() + frontier_state, frontier_usage = run_frontier_file_phase0_choice_v2( + root, query, candidate, frontier_decider, frontier_trace, + **frontier_params + ) + frontier_ms = (time.perf_counter() - started) * 1000 + frontier_files = canonical_frontier_results([frontier_state]) + frontier_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [frontier_state], "result_files": frontier_files, + "metrics": { + **frontier_usage, + "reads_executed": len(frontier_state["observations"]), + "evidence_regions": sum(len(x["evidence"]) for x in frontier_files), + "elapsed_ms": round(frontier_ms, 3), + }, + } + frontier_canonical = build_system_one_relevance_frontier_result( + frontier_engine, model + ) + + report = { + "query": query, + "file": file_path, + "subject": subject, + "model": model, + "frontier_params": frontier_params, + "range_params": range_params, + "range": { + "metrics": range_engine["metrics"], + "termination": range_state["termination"], + "epochs": range_state["epoch"], + "source_coverage": coverage(range_state), + "evidence": evidence_ranges(range_canonical), + }, + "frontier": { + "metrics": frontier_engine["metrics"], + "termination": frontier_state["termination"], + "rounds": frontier_state["round"], + "frontier_coverage": frontier_state["frontier_coverage"], + "source_coverage": frontier_state["source_coverage"], + "action_counts": action_counts(frontier_state), + "evidence": evidence_ranges(frontier_canonical), + }, + } + report["comparison"] = { + "elapsed_ratio": frontier_ms / max(1.0, range_ms), + "read_ratio": len(frontier_state["observations"]) / max(1, range_state["read_count"]), + "input_token_ratio": frontier_usage["input_tokens"] / max(1, range_usage["input_tokens"]), + "output_token_ratio": frontier_usage["output_tokens"] / max(1, range_usage["output_tokens"]), + "frontier_evidence_covered_by_range": overlap( + evidence_ranges(frontier_canonical), evidence_ranges(range_canonical) + ), + "range_evidence_covered_by_frontier": overlap( + evidence_ranges(range_canonical), evidence_ranges(frontier_canonical) + ), + } + + Path("artifacts/range/result.json").write_text( + json.dumps(range_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/range/localization-result.json").write_text( + json.dumps(range_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/result.json").write_text( + json.dumps(frontier_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/localization-result.json").write_text( + json.dumps(frontier_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/report.json").write_text( + json.dumps(report, indent=2, ensure_ascii=False) + "\n" + ) + print(json.dumps(report, indent=2, ensure_ascii=False)) + PY + + - name: Compare canonical results + run: | + set -euo pipefail + python3 harness/src/compare_localization_results.py \ + --left artifacts/range/localization-result.json \ + --right artifacts/frontier/localization-result.json \ + --output-json artifacts/localization-comparison.json \ + --output-markdown artifacts/localization-comparison.md + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + r = json.load(open("artifacts/report.json")) + print("## Single-file Range vs Frontier phase0-choice-v2") + print("") + print("| Metric | Range | Frontier phase0-choice-v2 |") + print("| --- | ---: | ---: |") + print("| elapsed ms | %s | %s |" % (r["range"]["metrics"]["elapsed_ms"], r["frontier"]["metrics"]["elapsed_ms"])) + print("| model calls | %s | %s |" % (r["range"]["metrics"]["model_calls"], r["frontier"]["metrics"]["model_calls"])) + print("| input tokens | %s | %s |" % (r["range"]["metrics"]["input_tokens"], r["frontier"]["metrics"]["input_tokens"])) + print("| output tokens | %s | %s |" % (r["range"]["metrics"]["output_tokens"], r["frontier"]["metrics"]["output_tokens"])) + print("| reads | %s | %s |" % (r["range"]["metrics"]["reads_executed"], r["frontier"]["metrics"]["reads_executed"])) + print("| source coverage | %.1f%% | %.1f%% |" % (r["range"]["source_coverage"] * 100, r["frontier"]["source_coverage"] * 100)) + print("| frontier coverage | - | %.1f%% |" % (r["frontier"]["frontier_coverage"] * 100)) + print("| evidence regions | %s | %s |" % (r["range"]["metrics"]["evidence_regions"], r["frontier"]["metrics"]["evidence_regions"])) + print("| termination | %s | %s |" % (r["range"]["termination"], r["frontier"]["termination"])) + print("") + print("Frontier action counts:", r["frontier"]["action_counts"]) + PY + cat artifacts/localization-comparison.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload experiment + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-v2-v2-${{ github.run_id }} + path: artifacts/ + if-no-files-found: error + retention-days: 90 + + blind-quality: + name: Blind quality review + needs: experiment + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: 0aa52ee0376c18b8452d2becf584a6f919f8b018 + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Download experiment + uses: actions/download-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-v2-v2-${{ github.run_id }} + path: artifacts + + - name: Prepare anonymous candidates + run: | + set -euo pipefail + mkdir -p artifacts/quality/range artifacts/quality/frontier + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/range/localization-result.json \ + --candidate-id candidate-a \ + --output-candidate artifacts/quality/range/candidate.json \ + --output-prompt artifacts/quality/range/prompt.txt + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/frontier/localization-result.json \ + --candidate-id candidate-b \ + --output-candidate artifacts/quality/frontier/candidate.json \ + --output-prompt artifacts/quality/frontier/prompt.txt + + - name: Evaluate range anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/range/prompt.txt \ + > ../artifacts/quality/range/evaluator.raw.jsonl + + - name: Evaluate frontier anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/frontier/prompt.txt \ + > ../artifacts/quality/frontier/evaluator.raw.jsonl + + - name: Finalize blind scorecards + run: | + set -euo pipefail + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/range/candidate.json \ + --raw-jsonl artifacts/quality/range/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/range/scorecard.json \ + --output-markdown artifacts/quality/range/report.md + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/frontier/candidate.json \ + --raw-jsonl artifacts/quality/frontier/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/frontier/scorecard.json \ + --output-markdown artifacts/quality/frontier/report.md + python3 harness/src/localization_quality_evaluation.py report \ + --candidate-a artifacts/quality/range/scorecard.json \ + --candidate-b artifacts/quality/frontier/scorecard.json \ + --candidate-a-label "Range runtime" \ + --candidate-b-label "Relevance frontier phase0-choice-v2" \ + --output-json artifacts/quality/evaluation-report.json \ + --output-markdown artifacts/quality/evaluation-report.md + cat artifacts/quality/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload quality + if: always() + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-phase0-choice-v2-v2-quality-${{ github.run_id }} + path: artifacts/quality/ + if-no-files-found: error + retention-days: 90 From 671eb3cdd69da98d79e20fc2b65aae6e48b2a0bf Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:27:02 +0800 Subject: [PATCH 13/18] fix: pin v2 benchmark branch trigger --- .../workflows/single-file-range-vs-phase0-choice-v2.yml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/.github/workflows/single-file-range-vs-phase0-choice-v2.yml b/.github/workflows/single-file-range-vs-phase0-choice-v2.yml index 15d6ec59..b7aa242d 100644 --- a/.github/workflows/single-file-range-vs-phase0-choice-v2.yml +++ b/.github/workflows/single-file-range-vs-phase0-choice-v2.yml @@ -3,7 +3,7 @@ name: Single-file Range vs Phase0 + Choice Frontier v2 on: push: branches: - - experiment/phase0-choice-v2-frontier-v2-20260924 + - experiment/phase0-choice-frontier-v2-20260924 permissions: contents: read @@ -288,7 +288,7 @@ jobs: - name: Upload experiment uses: actions/upload-artifact@v4 with: - name: single-file-range-vs-phase0-choice-v2-v2-${{ github.run_id }} + name: single-file-range-vs-phase0-choice-v2-${{ github.run_id }} path: artifacts/ if-no-files-found: error retention-days: 90 @@ -332,7 +332,7 @@ jobs: - name: Download experiment uses: actions/download-artifact@v4 with: - name: single-file-range-vs-phase0-choice-v2-v2-${{ github.run_id }} + name: single-file-range-vs-phase0-choice-v2-${{ github.run_id }} path: artifacts - name: Prepare anonymous candidates @@ -418,7 +418,7 @@ jobs: if: always() uses: actions/upload-artifact@v4 with: - name: single-file-range-vs-phase0-choice-v2-v2-quality-${{ github.run_id }} + name: single-file-range-vs-phase0-choice-v2-quality-${{ github.run_id }} path: artifacts/quality/ if-no-files-found: error retention-days: 90 From e527d7438fd8d35a64313c91923de2746d17eee5 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:27:06 +0800 Subject: [PATCH 14/18] experiment: trigger phase0 Choice frontier v2 benchmark --- experiments/PHASE0_CHOICE_FRONTIER_V2_TRIGGER.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/PHASE0_CHOICE_FRONTIER_V2_TRIGGER.md diff --git a/experiments/PHASE0_CHOICE_FRONTIER_V2_TRIGGER.md b/experiments/PHASE0_CHOICE_FRONTIER_V2_TRIGGER.md new file mode 100644 index 00000000..aef25395 --- /dev/null +++ b/experiments/PHASE0_CHOICE_FRONTIER_V2_TRIGGER.md @@ -0,0 +1 @@ +trigger: phase0 coarse scan + multi-action Choice frontier v2 2026-09-24 From 4d1bdf640ad6c942eaf98185e88ddd471fd29b52 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:27:45 +0800 Subject: [PATCH 15/18] fix: pin corrected Choice frontier harness --- .github/workflows/single-file-range-vs-phase0-choice-v2.yml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/.github/workflows/single-file-range-vs-phase0-choice-v2.yml b/.github/workflows/single-file-range-vs-phase0-choice-v2.yml index b7aa242d..59874d10 100644 --- a/.github/workflows/single-file-range-vs-phase0-choice-v2.yml +++ b/.github/workflows/single-file-range-vs-phase0-choice-v2.yml @@ -14,7 +14,7 @@ env: SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df FILE_PATH: crates/nession-agent/src/server/websocket.rs HARNESS_REPOSITORY: BestNathan/system-one-code-explore - HARNESS_SHA: 0aa52ee0376c18b8452d2becf584a6f919f8b018 + HARNESS_SHA: 79223f693add092a1de6b8d8bc5d00a1b9774c41 TYPESAFE_MODEL: jev-latest TYPESAFE_API_URL: https://api.typesafe.ai/v1/systemone @@ -29,7 +29,7 @@ jobs: uses: actions/checkout@v4 with: repository: BestNathan/system-one-code-explore - ref: 0aa52ee0376c18b8452d2becf584a6f919f8b018 + ref: 79223f693add092a1de6b8d8bc5d00a1b9774c41 path: harness fetch-depth: 1 persist-credentials: false @@ -304,7 +304,7 @@ jobs: uses: actions/checkout@v4 with: repository: BestNathan/system-one-code-explore - ref: 0aa52ee0376c18b8452d2becf584a6f919f8b018 + ref: 79223f693add092a1de6b8d8bc5d00a1b9774c41 path: harness fetch-depth: 1 persist-credentials: false From 3b4fbabc85078bb918bea87f82a5028ac4330088 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:27:49 +0800 Subject: [PATCH 16/18] experiment: retry phase0 Choice frontier v2 --- experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY.md diff --git a/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY.md b/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY.md new file mode 100644 index 00000000..c72e3042 --- /dev/null +++ b/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY.md @@ -0,0 +1 @@ +retry after harness test fixes From 0d456c08e0e31967335edd329cde31e92362bdb8 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:28:06 +0800 Subject: [PATCH 17/18] fix: pin v2 harness test revision --- .github/workflows/single-file-range-vs-phase0-choice-v2.yml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/.github/workflows/single-file-range-vs-phase0-choice-v2.yml b/.github/workflows/single-file-range-vs-phase0-choice-v2.yml index 59874d10..c99b815f 100644 --- a/.github/workflows/single-file-range-vs-phase0-choice-v2.yml +++ b/.github/workflows/single-file-range-vs-phase0-choice-v2.yml @@ -14,7 +14,7 @@ env: SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df FILE_PATH: crates/nession-agent/src/server/websocket.rs HARNESS_REPOSITORY: BestNathan/system-one-code-explore - HARNESS_SHA: 79223f693add092a1de6b8d8bc5d00a1b9774c41 + HARNESS_SHA: b2264c9f14d63c2617ac76b335f43e4121da1d8b TYPESAFE_MODEL: jev-latest TYPESAFE_API_URL: https://api.typesafe.ai/v1/systemone @@ -29,7 +29,7 @@ jobs: uses: actions/checkout@v4 with: repository: BestNathan/system-one-code-explore - ref: 79223f693add092a1de6b8d8bc5d00a1b9774c41 + ref: b2264c9f14d63c2617ac76b335f43e4121da1d8b path: harness fetch-depth: 1 persist-credentials: false @@ -304,7 +304,7 @@ jobs: uses: actions/checkout@v4 with: repository: BestNathan/system-one-code-explore - ref: 79223f693add092a1de6b8d8bc5d00a1b9774c41 + ref: b2264c9f14d63c2617ac76b335f43e4121da1d8b path: harness fetch-depth: 1 persist-credentials: false From 655587a3e62069b7eab32caaec20fe64361aa40e Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:28:10 +0800 Subject: [PATCH 18/18] experiment: retry phase0 Choice frontier v2 again --- experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY2.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY2.md diff --git a/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY2.md b/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY2.md new file mode 100644 index 00000000..dabbe41f --- /dev/null +++ b/experiments/PHASE0_CHOICE_FRONTIER_V2_RETRY2.md @@ -0,0 +1 @@ +retry v2 after multi-action test fixture fix