Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
19 commits
Select commit Hold shift + click to select a range
5cbe333
experiment: compare single-file range and frontier v1b
BestNathan Sep 24, 2026
6d94f28
experiment: trigger single-file frontier v1b
BestNathan Sep 24, 2026
06911c2
experiment: run balanced frontier v1c on agent websocket
BestNathan Sep 24, 2026
27e0ee1
fix: keep v1b experiment branch trigger
BestNathan Sep 24, 2026
4ad2058
experiment: trigger balanced frontier v1c
BestNathan Sep 24, 2026
a3aade6
experiment: add full-read Claude relevance baseline workflow
BestNathan Sep 24, 2026
60d157b
fix: pin frozen subject and stable baseline artifact name
BestNathan Sep 24, 2026
f3baff1
experiment: trigger full-read Claude baseline
BestNathan Sep 24, 2026
fb623fe
fix: checkout baseline harness branch reliably
BestNathan Sep 24, 2026
e2383ae
experiment: retry full-read baseline
BestNathan Sep 24, 2026
7097f20
fix: allow Claude Code proxy model for full-read baseline
BestNathan Sep 24, 2026
54674e9
experiment: retry full-read baseline with proxy model
BestNathan Sep 24, 2026
9ffb931
fix: use Claude Code Sonnet for full-read reference baseline
BestNathan Sep 24, 2026
c50f542
experiment: retry full-read baseline with Sonnet
BestNathan Sep 24, 2026
a9046bc
fix: pin active Claude Sonnet 4.6 model for baseline
BestNathan Sep 24, 2026
98a5cb7
experiment: retry full-read baseline with Sonnet 4.6
BestNathan Sep 24, 2026
fd30d24
fix: run full-read CC baseline through ds model routing
BestNathan Sep 24, 2026
c5769d9
experiment: trigger full-read CC ds baseline
BestNathan Sep 24, 2026
0b9ca30
fix: retry ds full-read baseline until canonical geometry validates
BestNathan Sep 24, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
189 changes: 189 additions & 0 deletions .github/workflows/full-read-claude-baseline.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,189 @@
name: Full-read CC relevance baseline (ds)

on:
push:
branches:
- experiment/full-read-baseline-20260924

permissions:
contents: read
actions: read

env:
QUERY: Help me optimize the websocket connection implementation
SUBJECT_REPOSITORY: BestNathan/nession
SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df
FILE_PATH: crates/nession-agent/src/server/websocket.rs
HARNESS_REPOSITORY: BestNathan/system-one-code-explore
V1C_RUN_ID: "35952482145"
V1C_ARTIFACT: single-file-range-vs-frontier-v1c-35952482145

jobs:
baseline:
name: Build full-read CC reference field with ds model
runs-on: ubuntu-24.04
environment: ds
timeout-minutes: 120
steps:
- name: Checkout harness
uses: actions/checkout@v4
with:
repository: BestNathan/system-one-code-explore
ref: experiment/full-read-baseline-20260924
path: harness
fetch-depth: 1
persist-credentials: false

- name: Checkout frozen subject
uses: actions/checkout@v4
with:
repository: BestNathan/nession
ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df
path: subject
fetch-depth: 1
persist-credentials: false

- name: Setup Python
uses: actions/setup-python@v5
with:
python-version: "3.13"

- name: Setup Node
uses: actions/setup-node@v4
with:
node-version: "22.14.0"

- name: Install pinned Claude Code
run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278

- name: Prepare reference prompt
run: |
set -euo pipefail
mkdir -p artifacts/baseline
python3 harness/src/full_read_relevance_baseline.py prepare \
--source "subject/$FILE_PATH" \
--query "$QUERY" \
--window-lines 64 \
--stride-lines 32 \
--output-prompt artifacts/baseline/prompt.txt \
--output-manifest artifacts/baseline/manifest.json

- name: Generate and validate full-read CC field through ds
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
ANTHROPIC_AUTH_TOKEN: ${{ secrets.ANTHROPIC_AUTH_TOKEN || secrets.ANTHROPIC_API_KEY }}
ANTHROPIC_BASE_URL: ${{ vars.ANTHROPIC_BASE_URL || secrets.ANTHROPIC_BASE_URL }}
ANTHROPIC_MODEL: ${{ vars.CLAUDE_MODEL || 'deepseek-flash' }}
CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT: "1"
run: |
set -euo pipefail
test -n "$ANTHROPIC_BASE_URL"
test -n "$ANTHROPIC_MODEL"
echo "Claude Code evaluator model: $ANTHROPIC_MODEL"

rm -f artifacts/baseline/reference.json
for attempt in 1 2 3; do
echo "Full-read reference attempt $attempt"
claude -p \
--output-format text \
--max-turns 4 \
--permission-mode dontAsk \
--no-session-persistence \
--setting-sources '' \
--strict-mcp-config \
--mcp-config '{"mcpServers":{}}' \
--disable-slash-commands \
--tools '' \
--settings '{"disableAllHooks":true}' \
< artifacts/baseline/prompt.txt \
> "artifacts/baseline/cc-output-attempt-${attempt}.txt" || true

if python3 harness/src/full_read_relevance_baseline.py normalize \
--source "subject/$FILE_PATH" \
--query "$QUERY" \
--window-lines 64 \
--stride-lines 32 \
--raw-output "artifacts/baseline/cc-output-attempt-${attempt}.txt" \
--output artifacts/baseline/reference.json; then
cp "artifacts/baseline/cc-output-attempt-${attempt}.txt" \
artifacts/baseline/cc-output.txt
echo "$attempt" > artifacts/baseline/accepted-attempt.txt
break
fi
rm -f artifacts/baseline/reference.json
done
test -s artifacts/baseline/reference.json

- name: Record evaluator provenance
env:
CC_MODEL: ${{ vars.CLAUDE_MODEL || 'deepseek-flash' }}
run: |
python3 - <<'PY'
import json, os
from pathlib import Path
Path("artifacts/baseline/evaluator.json").write_text(
json.dumps({
"runtime": "claude-code",
"environment": "ds",
"model": os.environ["CC_MODEL"],
}, indent=2) + "\n"
)
PY

- name: Download v1c candidate
uses: actions/download-artifact@v5
with:
name: single-file-range-vs-frontier-v1c-35952482145
path: artifacts/v1c
run-id: 35952482145
github-token: ${{ secrets.GITHUB_TOKEN }}

- name: Project v1c results onto reference field
run: |
set -euo pipefail
python3 harness/src/compare_to_full_read_baseline.py \
--reference artifacts/baseline/reference.json \
--candidate artifacts/v1c/frontier/localization-result.json \
--output-json artifacts/baseline/v1c-frontier-reference-metrics.json
python3 harness/src/compare_to_full_read_baseline.py \
--reference artifacts/baseline/reference.json \
--candidate artifacts/v1c/range/localization-result.json \
--output-json artifacts/baseline/v1c-range-reference-metrics.json

- name: Summarize
run: |
set -euo pipefail
python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY"
import json
evaluator = json.load(open("artifacts/baseline/evaluator.json"))
print("## Full-read CC reference")
print("")
print("Runtime: `%s`, environment: `%s`, model: `%s`" % (
evaluator["runtime"], evaluator["environment"], evaluator["model"]
))
print("")
for label, path in [
("Range v1c", "artifacts/baseline/v1c-range-reference-metrics.json"),
("Frontier v1c", "artifacts/baseline/v1c-frontier-reference-metrics.json"),
]:
r = json.load(open(path))
m = r["metrics"]
print(f"### {label}")
print("")
print("| Metric | Value |")
print("| --- | ---: |")
print(f"| weighted relevance recall | {m['weighted_relevance_recall']:.3f} |")
print(f"| high-relevance window recall | {m['high_relevance_window_recall']:.3f} |")
print(f"| core-window recall | {m['core_window_recall']:.3f} |")
print(f"| relevance-weighted precision | {m['relevance_weighted_precision']:.3f} |")
print(f"| source coverage | {r['candidate']['source_coverage']:.3f} |")
print("")
PY

- name: Upload baseline
uses: actions/upload-artifact@v4
with:
name: full-read-cc-ds-baseline-${{ github.run_id }}
path: artifacts/baseline
if-no-files-found: error
retention-days: 90
Loading
Loading