From 5cbe333b04cf22497381793e308a8ba44796fac6 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 11:39:06 +0800 Subject: [PATCH 01/19] experiment: compare single-file range and frontier v1b --- .../single-file-range-vs-frontier-v1b.yml | 418 ++++++++++++++++++ 1 file changed, 418 insertions(+) create mode 100644 .github/workflows/single-file-range-vs-frontier-v1b.yml diff --git a/.github/workflows/single-file-range-vs-frontier-v1b.yml b/.github/workflows/single-file-range-vs-frontier-v1b.yml new file mode 100644 index 00000000..04d2e72a --- /dev/null +++ b/.github/workflows/single-file-range-vs-frontier-v1b.yml @@ -0,0 +1,418 @@ +name: Single-file Range vs Relevance Frontier v1b + +on: + push: + branches: + - experiment/single-file-frontier-v1b-20260924 + +permissions: + contents: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + FILE_PATH: crates/nession-server/src/server/handler.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + HARNESS_SHA: cfc3d696df6c9ecc833372cf616805d2b931d0af + TYPESAFE_MODEL: jev-latest + TYPESAFE_API_URL: https://api.typesafe.ai/v1/systemone + +jobs: + experiment: + name: Single-file Range vs Frontier v1b + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: cfc3d696df6c9ecc833372cf616805d2b931d0af + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Run single-file experiment + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + run: | + set -euo pipefail + test -n "$TYPESAFE_API_KEY" + mkdir -p artifacts/range artifacts/frontier + python3 - <<'PY' + import json + import os + import sys + import time + from pathlib import Path + + sys.path.insert(0, str(Path("harness/src").resolve())) + from localization_result import ( + build_system_one_range_result, + build_system_one_relevance_frontier_result, + ) + from system_one_code_locator import Trace + from system_one_range_runtime import ( + SystemOneFileDecider, + canonical_file_results, + run_file_runtime, + ) + from system_one_relevance_frontier import ( + RelevanceFrontierDecider, + canonical_results as canonical_frontier_results, + run_frontier_file, + ) + + root = Path("subject") + query = os.environ["QUERY"] + model = os.environ["TYPESAFE_MODEL"] + endpoint = os.environ["TYPESAFE_API_URL"] + key = os.environ["TYPESAFE_API_KEY"] + file_path = os.environ["FILE_PATH"] + subject = { + "repository": os.environ["SUBJECT_REPOSITORY"], + "revision": os.environ["SUBJECT_SHA"], + } + candidate = { + "score": 0.90, + "payload": {"path": file_path, "extension": ".rs"}, + } + + range_params = { + "window_lines": 140, + "parallel_threshold": 0.65, + "max_jumps": 2, + "max_file_epochs": 32, + } + frontier_params = { + "max_rounds": 10, + "max_actions_per_round": 2, + "probe_lines": 128, + "target_region_lines": 64, + "final_window_lines": 24, + "refine_threshold": 0.78, + "candidate_threshold": 0.60, + "gradient_threshold": 0.14, + "volatility_threshold": 0.10, + "stable_delta": 0.06, + "stable_rounds": 2, + "max_frontier_leaves": 18, + "final_max_candidates": 16, + } + + def coverage(state): + rows = sorted((int(a), int(b)) for a, b in state.get("coverage", [])) + merged = [] + for a, b in rows: + if not merged or a > merged[-1][1] + 1: + merged.append([a, b]) + else: + merged[-1][1] = max(merged[-1][1], b) + return sum(b - a + 1 for a, b in merged) / max(1, int(state["line_count"])) + + def evidence_ranges(result): + return [ + (int(e["start_line"]), int(e["end_line"])) + for item in result.get("files", []) + for e in item.get("evidence", []) + ] + + def overlap(left, right): + total = sum(b - a + 1 for a, b in left) + if not total: + return 0.0 + hit = 0 + for a, b in left: + for c, d in right: + hit += max(0, min(b, d) - max(a, c) + 1) + return min(1.0, hit / total) + + def action_counts(state): + counts = {} + for snapshot in state.get("action_history", []): + for action in snapshot.get("actions", []): + kind = action.get("kind") + counts[kind] = counts.get(kind, 0) + 1 + return counts + + range_trace = Trace("artifacts/range/trace.jsonl") + range_decider = SystemOneFileDecider(key, range_trace, endpoint, model) + started = time.perf_counter() + range_state, range_usage = run_file_runtime( + root, query, candidate, range_decider, range_trace, **range_params + ) + range_ms = (time.perf_counter() - started) * 1000 + range_files = canonical_file_results([range_state], 0.65) + range_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [range_state], "result_files": range_files, + "metrics": { + **range_usage, + "reads_executed": range_state["read_count"], + "evidence_regions": sum(len(x["evidence"]) for x in range_files), + "elapsed_ms": round(range_ms, 3), + }, + } + range_canonical = build_system_one_range_result(range_engine, model) + + frontier_trace = Trace("artifacts/frontier/trace.jsonl") + frontier_decider = RelevanceFrontierDecider( + key, frontier_trace, endpoint, model + ) + started = time.perf_counter() + frontier_state, frontier_usage = run_frontier_file( + root, query, candidate, frontier_decider, frontier_trace, + **frontier_params + ) + frontier_ms = (time.perf_counter() - started) * 1000 + frontier_files = canonical_frontier_results([frontier_state]) + frontier_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [frontier_state], "result_files": frontier_files, + "metrics": { + **frontier_usage, + "reads_executed": len(frontier_state["observations"]), + "evidence_regions": sum(len(x["evidence"]) for x in frontier_files), + "elapsed_ms": round(frontier_ms, 3), + }, + } + frontier_canonical = build_system_one_relevance_frontier_result( + frontier_engine, model + ) + + report = { + "query": query, + "file": file_path, + "subject": subject, + "model": model, + "frontier_params": frontier_params, + "range_params": range_params, + "range": { + "metrics": range_engine["metrics"], + "termination": range_state["termination"], + "epochs": range_state["epoch"], + "source_coverage": coverage(range_state), + "evidence": evidence_ranges(range_canonical), + }, + "frontier": { + "metrics": frontier_engine["metrics"], + "termination": frontier_state["termination"], + "rounds": frontier_state["round"], + "frontier_coverage": frontier_state["frontier_coverage"], + "source_coverage": frontier_state["source_coverage"], + "action_counts": action_counts(frontier_state), + "evidence": evidence_ranges(frontier_canonical), + }, + } + report["comparison"] = { + "elapsed_ratio": frontier_ms / max(1.0, range_ms), + "read_ratio": len(frontier_state["observations"]) / max(1, range_state["read_count"]), + "input_token_ratio": frontier_usage["input_tokens"] / max(1, range_usage["input_tokens"]), + "output_token_ratio": frontier_usage["output_tokens"] / max(1, range_usage["output_tokens"]), + "frontier_evidence_covered_by_range": overlap( + evidence_ranges(frontier_canonical), evidence_ranges(range_canonical) + ), + "range_evidence_covered_by_frontier": overlap( + evidence_ranges(range_canonical), evidence_ranges(frontier_canonical) + ), + } + + Path("artifacts/range/result.json").write_text( + json.dumps(range_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/range/localization-result.json").write_text( + json.dumps(range_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/result.json").write_text( + json.dumps(frontier_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/localization-result.json").write_text( + json.dumps(frontier_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/report.json").write_text( + json.dumps(report, indent=2, ensure_ascii=False) + "\n" + ) + print(json.dumps(report, indent=2, ensure_ascii=False)) + PY + + - name: Compare canonical results + run: | + set -euo pipefail + python3 harness/src/compare_localization_results.py \ + --left artifacts/range/localization-result.json \ + --right artifacts/frontier/localization-result.json \ + --output-json artifacts/localization-comparison.json \ + --output-markdown artifacts/localization-comparison.md + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + r = json.load(open("artifacts/report.json")) + print("## Single-file Range vs Frontier v1b") + print("") + print("| Metric | Range | Frontier v1b |") + print("| --- | ---: | ---: |") + print("| elapsed ms | %s | %s |" % (r["range"]["metrics"]["elapsed_ms"], r["frontier"]["metrics"]["elapsed_ms"])) + print("| model calls | %s | %s |" % (r["range"]["metrics"]["model_calls"], r["frontier"]["metrics"]["model_calls"])) + print("| input tokens | %s | %s |" % (r["range"]["metrics"]["input_tokens"], r["frontier"]["metrics"]["input_tokens"])) + print("| output tokens | %s | %s |" % (r["range"]["metrics"]["output_tokens"], r["frontier"]["metrics"]["output_tokens"])) + print("| reads | %s | %s |" % (r["range"]["metrics"]["reads_executed"], r["frontier"]["metrics"]["reads_executed"])) + print("| source coverage | %.1f%% | %.1f%% |" % (r["range"]["source_coverage"] * 100, r["frontier"]["source_coverage"] * 100)) + print("| frontier coverage | - | %.1f%% |" % (r["frontier"]["frontier_coverage"] * 100)) + print("| evidence regions | %s | %s |" % (r["range"]["metrics"]["evidence_regions"], r["frontier"]["metrics"]["evidence_regions"])) + print("| termination | %s | %s |" % (r["range"]["termination"], r["frontier"]["termination"])) + print("") + print("Frontier action counts:", r["frontier"]["action_counts"]) + PY + cat artifacts/localization-comparison.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload experiment + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-frontier-v1b-${{ github.run_id }} + path: artifacts/ + if-no-files-found: error + retention-days: 90 + + blind-quality: + name: Blind quality review + needs: experiment + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: cfc3d696df6c9ecc833372cf616805d2b931d0af + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Download experiment + uses: actions/download-artifact@v4 + with: + name: single-file-range-vs-frontier-v1b-${{ github.run_id }} + path: artifacts + + - name: Prepare anonymous candidates + run: | + set -euo pipefail + mkdir -p artifacts/quality/range artifacts/quality/frontier + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/range/localization-result.json \ + --candidate-id candidate-a \ + --output-candidate artifacts/quality/range/candidate.json \ + --output-prompt artifacts/quality/range/prompt.txt + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/frontier/localization-result.json \ + --candidate-id candidate-b \ + --output-candidate artifacts/quality/frontier/candidate.json \ + --output-prompt artifacts/quality/frontier/prompt.txt + + - name: Evaluate range anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/range/prompt.txt \ + > ../artifacts/quality/range/evaluator.raw.jsonl + + - name: Evaluate frontier anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/frontier/prompt.txt \ + > ../artifacts/quality/frontier/evaluator.raw.jsonl + + - name: Finalize blind scorecards + run: | + set -euo pipefail + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/range/candidate.json \ + --raw-jsonl artifacts/quality/range/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/range/scorecard.json \ + --output-markdown artifacts/quality/range/report.md + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/frontier/candidate.json \ + --raw-jsonl artifacts/quality/frontier/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/frontier/scorecard.json \ + --output-markdown artifacts/quality/frontier/report.md + python3 harness/src/localization_quality_evaluation.py report \ + --candidate-a artifacts/quality/range/scorecard.json \ + --candidate-b artifacts/quality/frontier/scorecard.json \ + --candidate-a-label "Range runtime" \ + --candidate-b-label "Relevance frontier v1b" \ + --output-json artifacts/quality/evaluation-report.json \ + --output-markdown artifacts/quality/evaluation-report.md + cat artifacts/quality/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload quality + if: always() + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-frontier-v1b-quality-${{ github.run_id }} + path: artifacts/quality/ + if-no-files-found: error + retention-days: 90 From 6d94f280547f3156ba5e32ea5cae328e8bc74983 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 11:39:09 +0800 Subject: [PATCH 02/19] experiment: trigger single-file frontier v1b --- experiments/single-file-frontier-v1b-trigger.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/single-file-frontier-v1b-trigger.md diff --git a/experiments/single-file-frontier-v1b-trigger.md b/experiments/single-file-frontier-v1b-trigger.md new file mode 100644 index 00000000..8893a4d2 --- /dev/null +++ b/experiments/single-file-frontier-v1b-trigger.md @@ -0,0 +1 @@ +Trigger controlled single-file frontier v1b experiment. From 06911c2ff7eb17e32464943cf5cdef45c10940b6 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 11:41:05 +0800 Subject: [PATCH 03/19] experiment: run balanced frontier v1c on agent websocket --- .../single-file-range-vs-frontier-v1b.yml | 36 +++++++++---------- 1 file changed, 18 insertions(+), 18 deletions(-) diff --git a/.github/workflows/single-file-range-vs-frontier-v1b.yml b/.github/workflows/single-file-range-vs-frontier-v1b.yml index 04d2e72a..3202f466 100644 --- a/.github/workflows/single-file-range-vs-frontier-v1b.yml +++ b/.github/workflows/single-file-range-vs-frontier-v1b.yml @@ -1,9 +1,9 @@ -name: Single-file Range vs Relevance Frontier v1b +name: Single-file Range vs Relevance Frontier v1c on: push: branches: - - experiment/single-file-frontier-v1b-20260924 + - experiment/single-file-frontier-v1c-20260924 permissions: contents: read @@ -12,7 +12,7 @@ env: QUERY: Help me optimize the websocket connection implementation SUBJECT_REPOSITORY: BestNathan/nession SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df - FILE_PATH: crates/nession-server/src/server/handler.rs + FILE_PATH: crates/nession-agent/src/server/websocket.rs HARNESS_REPOSITORY: BestNathan/system-one-code-explore HARNESS_SHA: cfc3d696df6c9ecc833372cf616805d2b931d0af TYPESAFE_MODEL: jev-latest @@ -20,7 +20,7 @@ env: jobs: experiment: - name: Single-file Range vs Frontier v1b + name: Single-file Range vs Frontier v1c runs-on: ubuntu-24.04 environment: typesafe timeout-minutes: 120 @@ -102,17 +102,17 @@ jobs: frontier_params = { "max_rounds": 10, "max_actions_per_round": 2, - "probe_lines": 128, - "target_region_lines": 64, - "final_window_lines": 24, - "refine_threshold": 0.78, - "candidate_threshold": 0.60, - "gradient_threshold": 0.14, + "probe_lines": 112, + "target_region_lines": 48, + "final_window_lines": 32, + "refine_threshold": 0.72, + "candidate_threshold": 0.55, + "gradient_threshold": 0.15, "volatility_threshold": 0.10, "stable_delta": 0.06, "stable_rounds": 2, - "max_frontier_leaves": 18, - "final_max_candidates": 16, + "max_frontier_leaves": 24, + "final_max_candidates": 24, } def coverage(state): @@ -261,9 +261,9 @@ jobs: python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" import json r = json.load(open("artifacts/report.json")) - print("## Single-file Range vs Frontier v1b") + print("## Single-file Range vs Frontier v1c") print("") - print("| Metric | Range | Frontier v1b |") + print("| Metric | Range | Frontier v1c |") print("| --- | ---: | ---: |") print("| elapsed ms | %s | %s |" % (r["range"]["metrics"]["elapsed_ms"], r["frontier"]["metrics"]["elapsed_ms"])) print("| model calls | %s | %s |" % (r["range"]["metrics"]["model_calls"], r["frontier"]["metrics"]["model_calls"])) @@ -282,7 +282,7 @@ jobs: - name: Upload experiment uses: actions/upload-artifact@v4 with: - name: single-file-range-vs-frontier-v1b-${{ github.run_id }} + name: single-file-range-vs-frontier-v1c-${{ github.run_id }} path: artifacts/ if-no-files-found: error retention-days: 90 @@ -326,7 +326,7 @@ jobs: - name: Download experiment uses: actions/download-artifact@v4 with: - name: single-file-range-vs-frontier-v1b-${{ github.run_id }} + name: single-file-range-vs-frontier-v1c-${{ github.run_id }} path: artifacts - name: Prepare anonymous candidates @@ -403,7 +403,7 @@ jobs: --candidate-a artifacts/quality/range/scorecard.json \ --candidate-b artifacts/quality/frontier/scorecard.json \ --candidate-a-label "Range runtime" \ - --candidate-b-label "Relevance frontier v1b" \ + --candidate-b-label "Relevance frontier v1c" \ --output-json artifacts/quality/evaluation-report.json \ --output-markdown artifacts/quality/evaluation-report.md cat artifacts/quality/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" @@ -412,7 +412,7 @@ jobs: if: always() uses: actions/upload-artifact@v4 with: - name: single-file-range-vs-frontier-v1b-quality-${{ github.run_id }} + name: single-file-range-vs-frontier-v1c-quality-${{ github.run_id }} path: artifacts/quality/ if-no-files-found: error retention-days: 90 From 27e0ee13a32a2ffd0f901ea8051ef1b62a404fc2 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 11:41:12 +0800 Subject: [PATCH 04/19] fix: keep v1b experiment branch trigger --- .github/workflows/single-file-range-vs-frontier-v1b.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/single-file-range-vs-frontier-v1b.yml b/.github/workflows/single-file-range-vs-frontier-v1b.yml index 3202f466..48292520 100644 --- a/.github/workflows/single-file-range-vs-frontier-v1b.yml +++ b/.github/workflows/single-file-range-vs-frontier-v1b.yml @@ -3,7 +3,7 @@ name: Single-file Range vs Relevance Frontier v1c on: push: branches: - - experiment/single-file-frontier-v1c-20260924 + - experiment/single-file-frontier-v1b-20260924 permissions: contents: read From 4ad20589fe064f66c936afb342e0656def5a79b1 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 11:41:15 +0800 Subject: [PATCH 05/19] experiment: trigger balanced frontier v1c --- experiments/single-file-frontier-v1c-trigger.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/single-file-frontier-v1c-trigger.md diff --git a/experiments/single-file-frontier-v1c-trigger.md b/experiments/single-file-frontier-v1c-trigger.md new file mode 100644 index 00000000..0e556d2c --- /dev/null +++ b/experiments/single-file-frontier-v1c-trigger.md @@ -0,0 +1 @@ +Trigger balanced v1c single-file experiment on agent websocket. From a3aade60a96a9c79000a501035c02d4bca6a2077 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:20:56 +0800 Subject: [PATCH 06/19] experiment: add full-read Claude relevance baseline workflow --- .../workflows/full-read-claude-baseline.yml | 156 ++++++++++++++++++ 1 file changed, 156 insertions(+) create mode 100644 .github/workflows/full-read-claude-baseline.yml diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml new file mode 100644 index 00000000..bb56ebe1 --- /dev/null +++ b/.github/workflows/full-read-claude-baseline.yml @@ -0,0 +1,156 @@ +name: Full-read Claude relevance baseline + +on: + push: + branches: + - experiment/full-read-baseline-20260924 + +permissions: + contents: read + actions: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706b2b1a1df + FILE_PATH: crates/nession-agent/src/server/websocket.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + HARNESS_SHA: f93b57d81e9e3f530ec1a26f23bb2472aa90d1b + V1C_RUN_ID: "35952482145" + V1C_ARTIFACT: single-file-range-vs-frontier-v1c-35952482145 + +jobs: + baseline: + name: Build full-read Claude reference field + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: f93b57d81e9e3f530ec1a26f23bb2472aa90d1b + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - name: Setup Python + uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Setup Node + uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Prepare reference prompt + run: | + set -euo pipefail + mkdir -p artifacts/baseline + python3 harness/src/full_read_relevance_baseline.py prepare \ + --source "subject/$FILE_PATH" \ + --query "$QUERY" \ + --window-lines 64 \ + --stride-lines 32 \ + --output-prompt artifacts/baseline/prompt.txt \ + --output-manifest artifacts/baseline/manifest.json + + - name: Generate full-read Claude field + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + test -n "$ANTHROPIC_API_KEY" + test -n "$CLAUDE_MODEL" + claude -p \ + --output-format text \ + --model "$CLAUDE_MODEL" \ + --effort high \ + --max-turns 4 \ + --permission-mode dontAsk \ + --no-session-persistence \ + --setting-sources '' \ + --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' \ + --disable-slash-commands \ + --tools '' \ + --settings '{"disableAllHooks":true}' \ + < artifacts/baseline/prompt.txt \ + > artifacts/baseline/claude-output.txt + + - name: Normalize reference field + run: | + set -euo pipefail + python3 harness/src/full_read_relevance_baseline.py normalize \ + --source "subject/$FILE_PATH" \ + --query "$QUERY" \ + --window-lines 64 \ + --stride-lines 32 \ + --raw-output artifacts/baseline/claude-output.txt \ + --output artifacts/baseline/reference.json + + - name: Download v1c candidate + uses: actions/download-artifact@v5 + with: + name: single-file-range-vs-frontier-v1c-35952482145 + path: artifacts/v1c + run-id: 35952482145 + github-token: ${{ secrets.GITHUB_TOKEN }} + + - name: Project v1c results onto reference field + run: | + set -euo pipefail + python3 harness/src/compare_to_full_read_baseline.py \ + --reference artifacts/baseline/reference.json \ + --candidate artifacts/v1c/frontier/localization-result.json \ + --output-json artifacts/baseline/v1c-frontier-reference-metrics.json + python3 harness/src/compare_to_full_read_baseline.py \ + --reference artifacts/baseline/reference.json \ + --candidate artifacts/v1c/range/localization-result.json \ + --output-json artifacts/baseline/v1c-range-reference-metrics.json + + - name: Summarize + run: | + set -euo pipefail + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + for label, path in [ + ("Range v1c", "artifacts/baseline/v1c-range-reference-metrics.json"), + ("Frontier v1c", "artifacts/baseline/v1c-frontier-reference-metrics.json"), + ]: + r = json.load(open(path)) + m = r["metrics"] + print(f"## {label} vs full-read reference") + print("") + print("| Metric | Value |") + print("| --- | ---: |") + print(f"| weighted relevance recall | {m['weighted_relevance_recall']:.3f} |") + print(f"| high-relevance window recall | {m['high_relevance_window_recall']:.3f} |") + print(f"| core-window recall | {m['core_window_recall']:.3f} |") + print(f"| relevance-weighted precision | {m['relevance_weighted_precision']:.3f} |") + print(f"| source coverage | {r['candidate']['source_coverage']:.3f} |") + print("") + PY + + - name: Upload baseline + uses: actions/upload-artifact@v4 + with: + name: full-read-claude-baseline-$GITHUB_RUN_ID + path: artifacts/baseline + if-no-files-found: error + retention-days: 90 From 60d157b92e6e13410b250ca2f2cc60fae0d28bde Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:21:01 +0800 Subject: [PATCH 07/19] fix: pin frozen subject and stable baseline artifact name --- .github/workflows/full-read-claude-baseline.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml index bb56ebe1..4d124874 100644 --- a/.github/workflows/full-read-claude-baseline.yml +++ b/.github/workflows/full-read-claude-baseline.yml @@ -12,7 +12,7 @@ permissions: env: QUERY: Help me optimize the websocket connection implementation SUBJECT_REPOSITORY: BestNathan/nession - SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706b2b1a1df + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df FILE_PATH: crates/nession-agent/src/server/websocket.rs HARNESS_REPOSITORY: BestNathan/system-one-code-explore HARNESS_SHA: f93b57d81e9e3f530ec1a26f23bb2472aa90d1b @@ -150,7 +150,7 @@ jobs: - name: Upload baseline uses: actions/upload-artifact@v4 with: - name: full-read-claude-baseline-$GITHUB_RUN_ID + name: full-read-claude-baseline path: artifacts/baseline if-no-files-found: error retention-days: 90 From f3baff1f9d00bf89c27ea78037e18eb28c472c7b Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:21:12 +0800 Subject: [PATCH 08/19] experiment: trigger full-read Claude baseline --- experiments/FULL_READ_BASELINE_TRIGGER.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/FULL_READ_BASELINE_TRIGGER.md diff --git a/experiments/FULL_READ_BASELINE_TRIGGER.md b/experiments/FULL_READ_BASELINE_TRIGGER.md new file mode 100644 index 00000000..9cff40dd --- /dev/null +++ b/experiments/FULL_READ_BASELINE_TRIGGER.md @@ -0,0 +1 @@ +trigger: full-read Claude relevance baseline 2026-09-24 From fb623fe542ea57c06f573d82ec0c46f9760e8fd3 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:24:11 +0800 Subject: [PATCH 09/19] fix: checkout baseline harness branch reliably --- .github/workflows/full-read-claude-baseline.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml index 4d124874..9a36473a 100644 --- a/.github/workflows/full-read-claude-baseline.yml +++ b/.github/workflows/full-read-claude-baseline.yml @@ -30,7 +30,7 @@ jobs: uses: actions/checkout@v4 with: repository: BestNathan/system-one-code-explore - ref: f93b57d81e9e3f530ec1a26f23bb2472aa90d1b + ref: experiment/full-read-baseline-20260924 path: harness fetch-depth: 1 persist-credentials: false From e2383ae5bb11f1a9352762be2f837c77afe86050 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:24:17 +0800 Subject: [PATCH 10/19] experiment: retry full-read baseline --- experiments/FULL_READ_BASELINE_RETRY.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/FULL_READ_BASELINE_RETRY.md diff --git a/experiments/FULL_READ_BASELINE_RETRY.md b/experiments/FULL_READ_BASELINE_RETRY.md new file mode 100644 index 00000000..fc0d0d64 --- /dev/null +++ b/experiments/FULL_READ_BASELINE_RETRY.md @@ -0,0 +1 @@ +retry after checkout fix From 7097f20c3dd2b98440dc2b43957c488099a4be8f Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:28:25 +0800 Subject: [PATCH 11/19] fix: allow Claude Code proxy model for full-read baseline --- .github/workflows/full-read-claude-baseline.yml | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml index 9a36473a..95e8f040 100644 --- a/.github/workflows/full-read-claude-baseline.yml +++ b/.github/workflows/full-read-claude-baseline.yml @@ -73,6 +73,7 @@ jobs: env: ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT: "1" run: | set -euo pipefail test -n "$ANTHROPIC_API_KEY" From 54674e9468e1db3f5d1a36b9a650f16c07bd87a0 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:28:29 +0800 Subject: [PATCH 12/19] experiment: retry full-read baseline with proxy model --- experiments/FULL_READ_BASELINE_PROXY_MODEL_TRIGGER.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/FULL_READ_BASELINE_PROXY_MODEL_TRIGGER.md diff --git a/experiments/FULL_READ_BASELINE_PROXY_MODEL_TRIGGER.md b/experiments/FULL_READ_BASELINE_PROXY_MODEL_TRIGGER.md new file mode 100644 index 00000000..94001239 --- /dev/null +++ b/experiments/FULL_READ_BASELINE_PROXY_MODEL_TRIGGER.md @@ -0,0 +1 @@ +trigger after proxy model catalog fix From 9ffb931915d4b2e76b910261359d8b0b536bac73 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:32:16 +0800 Subject: [PATCH 13/19] fix: use Claude Code Sonnet for full-read reference baseline --- .github/workflows/full-read-claude-baseline.yml | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml index 95e8f040..f6593518 100644 --- a/.github/workflows/full-read-claude-baseline.yml +++ b/.github/workflows/full-read-claude-baseline.yml @@ -72,15 +72,13 @@ jobs: - name: Generate full-read Claude field env: ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} - CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT: "1" run: | set -euo pipefail test -n "$ANTHROPIC_API_KEY" - test -n "$CLAUDE_MODEL" claude -p \ --output-format text \ - --model "$CLAUDE_MODEL" \ + --model sonnet \ --effort high \ --max-turns 4 \ --permission-mode dontAsk \ From c50f542f132249be4cb2e5d80cc2e99e6ab14b41 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:32:20 +0800 Subject: [PATCH 14/19] experiment: retry full-read baseline with Sonnet --- experiments/FULL_READ_BASELINE_SONNET_TRIGGER.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/FULL_READ_BASELINE_SONNET_TRIGGER.md diff --git a/experiments/FULL_READ_BASELINE_SONNET_TRIGGER.md b/experiments/FULL_READ_BASELINE_SONNET_TRIGGER.md new file mode 100644 index 00000000..2806140b --- /dev/null +++ b/experiments/FULL_READ_BASELINE_SONNET_TRIGGER.md @@ -0,0 +1 @@ +retry full-read baseline with Claude Code Sonnet From a9046bc309e1e43269192bf1d4a5b17155e6e5d2 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:37:18 +0800 Subject: [PATCH 15/19] fix: pin active Claude Sonnet 4.6 model for baseline --- .github/workflows/full-read-claude-baseline.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml index f6593518..8b26b91d 100644 --- a/.github/workflows/full-read-claude-baseline.yml +++ b/.github/workflows/full-read-claude-baseline.yml @@ -78,7 +78,7 @@ jobs: test -n "$ANTHROPIC_API_KEY" claude -p \ --output-format text \ - --model sonnet \ + --model claude-sonnet-4-6 \ --effort high \ --max-turns 4 \ --permission-mode dontAsk \ From 98a5cb77a62bdfa743329574c247c171b3cf2329 Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 12:37:22 +0800 Subject: [PATCH 16/19] experiment: retry full-read baseline with Sonnet 4.6 --- experiments/FULL_READ_BASELINE_SONNET46_TRIGGER.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/FULL_READ_BASELINE_SONNET46_TRIGGER.md diff --git a/experiments/FULL_READ_BASELINE_SONNET46_TRIGGER.md b/experiments/FULL_READ_BASELINE_SONNET46_TRIGGER.md new file mode 100644 index 00000000..86be6d7f --- /dev/null +++ b/experiments/FULL_READ_BASELINE_SONNET46_TRIGGER.md @@ -0,0 +1 @@ +retry full-read baseline with explicit Claude Sonnet 4.6 From fd30d24e1d2c3ea7472e5ebc8f05ccd94fe27ecf Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 13:13:51 +0800 Subject: [PATCH 17/19] fix: run full-read CC baseline through ds model routing --- .../workflows/full-read-claude-baseline.yml | 48 ++++++++++++++----- 1 file changed, 36 insertions(+), 12 deletions(-) diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml index 8b26b91d..051b0354 100644 --- a/.github/workflows/full-read-claude-baseline.yml +++ b/.github/workflows/full-read-claude-baseline.yml @@ -1,4 +1,4 @@ -name: Full-read Claude relevance baseline +name: Full-read CC relevance baseline (ds) on: push: @@ -15,13 +15,12 @@ env: SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df FILE_PATH: crates/nession-agent/src/server/websocket.rs HARNESS_REPOSITORY: BestNathan/system-one-code-explore - HARNESS_SHA: f93b57d81e9e3f530ec1a26f23bb2472aa90d1b V1C_RUN_ID: "35952482145" V1C_ARTIFACT: single-file-range-vs-frontier-v1c-35952482145 jobs: baseline: - name: Build full-read Claude reference field + name: Build full-read CC reference field with ds model runs-on: ubuntu-24.04 environment: ds timeout-minutes: 120 @@ -69,17 +68,19 @@ jobs: --output-prompt artifacts/baseline/prompt.txt \ --output-manifest artifacts/baseline/manifest.json - - name: Generate full-read Claude field + - name: Generate full-read CC field through ds env: ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} - CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT: "1" + ANTHROPIC_AUTH_TOKEN: ${{ secrets.ANTHROPIC_AUTH_TOKEN || secrets.ANTHROPIC_API_KEY }} + ANTHROPIC_BASE_URL: ${{ vars.ANTHROPIC_BASE_URL || secrets.ANTHROPIC_BASE_URL }} + ANTHROPIC_MODEL: ${{ vars.CLAUDE_MODEL || 'deepseek-flash' }} run: | set -euo pipefail - test -n "$ANTHROPIC_API_KEY" + test -n "$ANTHROPIC_BASE_URL" + test -n "$ANTHROPIC_MODEL" + echo "Claude Code evaluator model: $ANTHROPIC_MODEL" claude -p \ --output-format text \ - --model claude-sonnet-4-6 \ - --effort high \ --max-turns 4 \ --permission-mode dontAsk \ --no-session-persistence \ @@ -90,7 +91,23 @@ jobs: --tools '' \ --settings '{"disableAllHooks":true}' \ < artifacts/baseline/prompt.txt \ - > artifacts/baseline/claude-output.txt + > artifacts/baseline/cc-output.txt + + - name: Record evaluator provenance + env: + CC_MODEL: ${{ vars.CLAUDE_MODEL || 'deepseek-flash' }} + run: | + python3 - <<'PY' + import json, os + from pathlib import Path + Path("artifacts/baseline/evaluator.json").write_text( + json.dumps({ + "runtime": "claude-code", + "environment": "ds", + "model": os.environ["CC_MODEL"], + }, indent=2) + "\n" + ) + PY - name: Normalize reference field run: | @@ -100,7 +117,7 @@ jobs: --query "$QUERY" \ --window-lines 64 \ --stride-lines 32 \ - --raw-output artifacts/baseline/claude-output.txt \ + --raw-output artifacts/baseline/cc-output.txt \ --output artifacts/baseline/reference.json - name: Download v1c candidate @@ -128,13 +145,20 @@ jobs: set -euo pipefail python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" import json + evaluator = json.load(open("artifacts/baseline/evaluator.json")) + print("## Full-read CC reference") + print("") + print("Runtime: `%s`, environment: `%s`, model: `%s`" % ( + evaluator["runtime"], evaluator["environment"], evaluator["model"] + )) + print("") for label, path in [ ("Range v1c", "artifacts/baseline/v1c-range-reference-metrics.json"), ("Frontier v1c", "artifacts/baseline/v1c-frontier-reference-metrics.json"), ]: r = json.load(open(path)) m = r["metrics"] - print(f"## {label} vs full-read reference") + print(f"### {label}") print("") print("| Metric | Value |") print("| --- | ---: |") @@ -149,7 +173,7 @@ jobs: - name: Upload baseline uses: actions/upload-artifact@v4 with: - name: full-read-claude-baseline + name: full-read-cc-ds-baseline-${{ github.run_id }} path: artifacts/baseline if-no-files-found: error retention-days: 90 From c5769d91c294ae1481b43866fe038e3b8fd0ce7f Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 13:16:08 +0800 Subject: [PATCH 18/19] experiment: trigger full-read CC ds baseline --- experiments/FULL_READ_DS_BASELINE_TRIGGER.md | 1 + 1 file changed, 1 insertion(+) create mode 100644 experiments/FULL_READ_DS_BASELINE_TRIGGER.md diff --git a/experiments/FULL_READ_DS_BASELINE_TRIGGER.md b/experiments/FULL_READ_DS_BASELINE_TRIGGER.md new file mode 100644 index 00000000..d99c474b --- /dev/null +++ b/experiments/FULL_READ_DS_BASELINE_TRIGGER.md @@ -0,0 +1 @@ +trigger: full-read Claude Code reference field through ds model routing 2026-09-24 From 0b9ca304ee42efc1b4d1b8161c90971505614aaa Mon Sep 17 00:00:00 2001 From: Nathan <308719298@qq.com> Date: Thu, 24 Sep 2026 13:21:13 +0800 Subject: [PATCH 19/19] fix: retry ds full-read baseline until canonical geometry validates --- .../workflows/full-read-claude-baseline.yml | 60 +++++++++++-------- 1 file changed, 35 insertions(+), 25 deletions(-) diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml index 051b0354..7c7a344d 100644 --- a/.github/workflows/full-read-claude-baseline.yml +++ b/.github/workflows/full-read-claude-baseline.yml @@ -68,30 +68,51 @@ jobs: --output-prompt artifacts/baseline/prompt.txt \ --output-manifest artifacts/baseline/manifest.json - - name: Generate full-read CC field through ds + - name: Generate and validate full-read CC field through ds env: ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} ANTHROPIC_AUTH_TOKEN: ${{ secrets.ANTHROPIC_AUTH_TOKEN || secrets.ANTHROPIC_API_KEY }} ANTHROPIC_BASE_URL: ${{ vars.ANTHROPIC_BASE_URL || secrets.ANTHROPIC_BASE_URL }} ANTHROPIC_MODEL: ${{ vars.CLAUDE_MODEL || 'deepseek-flash' }} + CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT: "1" run: | set -euo pipefail test -n "$ANTHROPIC_BASE_URL" test -n "$ANTHROPIC_MODEL" echo "Claude Code evaluator model: $ANTHROPIC_MODEL" - claude -p \ - --output-format text \ - --max-turns 4 \ - --permission-mode dontAsk \ - --no-session-persistence \ - --setting-sources '' \ - --strict-mcp-config \ - --mcp-config '{"mcpServers":{}}' \ - --disable-slash-commands \ - --tools '' \ - --settings '{"disableAllHooks":true}' \ - < artifacts/baseline/prompt.txt \ - > artifacts/baseline/cc-output.txt + + rm -f artifacts/baseline/reference.json + for attempt in 1 2 3; do + echo "Full-read reference attempt $attempt" + claude -p \ + --output-format text \ + --max-turns 4 \ + --permission-mode dontAsk \ + --no-session-persistence \ + --setting-sources '' \ + --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' \ + --disable-slash-commands \ + --tools '' \ + --settings '{"disableAllHooks":true}' \ + < artifacts/baseline/prompt.txt \ + > "artifacts/baseline/cc-output-attempt-${attempt}.txt" || true + + if python3 harness/src/full_read_relevance_baseline.py normalize \ + --source "subject/$FILE_PATH" \ + --query "$QUERY" \ + --window-lines 64 \ + --stride-lines 32 \ + --raw-output "artifacts/baseline/cc-output-attempt-${attempt}.txt" \ + --output artifacts/baseline/reference.json; then + cp "artifacts/baseline/cc-output-attempt-${attempt}.txt" \ + artifacts/baseline/cc-output.txt + echo "$attempt" > artifacts/baseline/accepted-attempt.txt + break + fi + rm -f artifacts/baseline/reference.json + done + test -s artifacts/baseline/reference.json - name: Record evaluator provenance env: @@ -109,17 +130,6 @@ jobs: ) PY - - name: Normalize reference field - run: | - set -euo pipefail - python3 harness/src/full_read_relevance_baseline.py normalize \ - --source "subject/$FILE_PATH" \ - --query "$QUERY" \ - --window-lines 64 \ - --stride-lines 32 \ - --raw-output artifacts/baseline/cc-output.txt \ - --output artifacts/baseline/reference.json - - name: Download v1c candidate uses: actions/download-artifact@v5 with: