diff --git a/.github/workflows/full-read-claude-baseline.yml b/.github/workflows/full-read-claude-baseline.yml new file mode 100644 index 00000000..7c7a344d --- /dev/null +++ b/.github/workflows/full-read-claude-baseline.yml @@ -0,0 +1,189 @@ +name: Full-read CC relevance baseline (ds) + +on: + push: + branches: + - experiment/full-read-baseline-20260924 + +permissions: + contents: read + actions: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + FILE_PATH: crates/nession-agent/src/server/websocket.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + V1C_RUN_ID: "35952482145" + V1C_ARTIFACT: single-file-range-vs-frontier-v1c-35952482145 + +jobs: + baseline: + name: Build full-read CC reference field with ds model + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: experiment/full-read-baseline-20260924 + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - name: Setup Python + uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Setup Node + uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Prepare reference prompt + run: | + set -euo pipefail + mkdir -p artifacts/baseline + python3 harness/src/full_read_relevance_baseline.py prepare \ + --source "subject/$FILE_PATH" \ + --query "$QUERY" \ + --window-lines 64 \ + --stride-lines 32 \ + --output-prompt artifacts/baseline/prompt.txt \ + --output-manifest artifacts/baseline/manifest.json + + - name: Generate and validate full-read CC field through ds + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + ANTHROPIC_AUTH_TOKEN: ${{ secrets.ANTHROPIC_AUTH_TOKEN || secrets.ANTHROPIC_API_KEY }} + ANTHROPIC_BASE_URL: ${{ vars.ANTHROPIC_BASE_URL || secrets.ANTHROPIC_BASE_URL }} + ANTHROPIC_MODEL: ${{ vars.CLAUDE_MODEL || 'deepseek-flash' }} + CLAUDE_CODE_DISABLE_UNKNOWN_MODEL_WINDOW_ENFORCEMENT: "1" + run: | + set -euo pipefail + test -n "$ANTHROPIC_BASE_URL" + test -n "$ANTHROPIC_MODEL" + echo "Claude Code evaluator model: $ANTHROPIC_MODEL" + + rm -f artifacts/baseline/reference.json + for attempt in 1 2 3; do + echo "Full-read reference attempt $attempt" + claude -p \ + --output-format text \ + --max-turns 4 \ + --permission-mode dontAsk \ + --no-session-persistence \ + --setting-sources '' \ + --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' \ + --disable-slash-commands \ + --tools '' \ + --settings '{"disableAllHooks":true}' \ + < artifacts/baseline/prompt.txt \ + > "artifacts/baseline/cc-output-attempt-${attempt}.txt" || true + + if python3 harness/src/full_read_relevance_baseline.py normalize \ + --source "subject/$FILE_PATH" \ + --query "$QUERY" \ + --window-lines 64 \ + --stride-lines 32 \ + --raw-output "artifacts/baseline/cc-output-attempt-${attempt}.txt" \ + --output artifacts/baseline/reference.json; then + cp "artifacts/baseline/cc-output-attempt-${attempt}.txt" \ + artifacts/baseline/cc-output.txt + echo "$attempt" > artifacts/baseline/accepted-attempt.txt + break + fi + rm -f artifacts/baseline/reference.json + done + test -s artifacts/baseline/reference.json + + - name: Record evaluator provenance + env: + CC_MODEL: ${{ vars.CLAUDE_MODEL || 'deepseek-flash' }} + run: | + python3 - <<'PY' + import json, os + from pathlib import Path + Path("artifacts/baseline/evaluator.json").write_text( + json.dumps({ + "runtime": "claude-code", + "environment": "ds", + "model": os.environ["CC_MODEL"], + }, indent=2) + "\n" + ) + PY + + - name: Download v1c candidate + uses: actions/download-artifact@v5 + with: + name: single-file-range-vs-frontier-v1c-35952482145 + path: artifacts/v1c + run-id: 35952482145 + github-token: ${{ secrets.GITHUB_TOKEN }} + + - name: Project v1c results onto reference field + run: | + set -euo pipefail + python3 harness/src/compare_to_full_read_baseline.py \ + --reference artifacts/baseline/reference.json \ + --candidate artifacts/v1c/frontier/localization-result.json \ + --output-json artifacts/baseline/v1c-frontier-reference-metrics.json + python3 harness/src/compare_to_full_read_baseline.py \ + --reference artifacts/baseline/reference.json \ + --candidate artifacts/v1c/range/localization-result.json \ + --output-json artifacts/baseline/v1c-range-reference-metrics.json + + - name: Summarize + run: | + set -euo pipefail + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + evaluator = json.load(open("artifacts/baseline/evaluator.json")) + print("## Full-read CC reference") + print("") + print("Runtime: `%s`, environment: `%s`, model: `%s`" % ( + evaluator["runtime"], evaluator["environment"], evaluator["model"] + )) + print("") + for label, path in [ + ("Range v1c", "artifacts/baseline/v1c-range-reference-metrics.json"), + ("Frontier v1c", "artifacts/baseline/v1c-frontier-reference-metrics.json"), + ]: + r = json.load(open(path)) + m = r["metrics"] + print(f"### {label}") + print("") + print("| Metric | Value |") + print("| --- | ---: |") + print(f"| weighted relevance recall | {m['weighted_relevance_recall']:.3f} |") + print(f"| high-relevance window recall | {m['high_relevance_window_recall']:.3f} |") + print(f"| core-window recall | {m['core_window_recall']:.3f} |") + print(f"| relevance-weighted precision | {m['relevance_weighted_precision']:.3f} |") + print(f"| source coverage | {r['candidate']['source_coverage']:.3f} |") + print("") + PY + + - name: Upload baseline + uses: actions/upload-artifact@v4 + with: + name: full-read-cc-ds-baseline-${{ github.run_id }} + path: artifacts/baseline + if-no-files-found: error + retention-days: 90 diff --git a/.github/workflows/single-file-range-vs-frontier-v1b.yml b/.github/workflows/single-file-range-vs-frontier-v1b.yml new file mode 100644 index 00000000..48292520 --- /dev/null +++ b/.github/workflows/single-file-range-vs-frontier-v1b.yml @@ -0,0 +1,418 @@ +name: Single-file Range vs Relevance Frontier v1c + +on: + push: + branches: + - experiment/single-file-frontier-v1b-20260924 + +permissions: + contents: read + +env: + QUERY: Help me optimize the websocket connection implementation + SUBJECT_REPOSITORY: BestNathan/nession + SUBJECT_SHA: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + FILE_PATH: crates/nession-agent/src/server/websocket.rs + HARNESS_REPOSITORY: BestNathan/system-one-code-explore + HARNESS_SHA: cfc3d696df6c9ecc833372cf616805d2b931d0af + TYPESAFE_MODEL: jev-latest + TYPESAFE_API_URL: https://api.typesafe.ai/v1/systemone + +jobs: + experiment: + name: Single-file Range vs Frontier v1c + runs-on: ubuntu-24.04 + environment: typesafe + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: cfc3d696df6c9ecc833372cf616805d2b931d0af + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Run single-file experiment + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY || vars.TYPESAFE_API_KEY }} + run: | + set -euo pipefail + test -n "$TYPESAFE_API_KEY" + mkdir -p artifacts/range artifacts/frontier + python3 - <<'PY' + import json + import os + import sys + import time + from pathlib import Path + + sys.path.insert(0, str(Path("harness/src").resolve())) + from localization_result import ( + build_system_one_range_result, + build_system_one_relevance_frontier_result, + ) + from system_one_code_locator import Trace + from system_one_range_runtime import ( + SystemOneFileDecider, + canonical_file_results, + run_file_runtime, + ) + from system_one_relevance_frontier import ( + RelevanceFrontierDecider, + canonical_results as canonical_frontier_results, + run_frontier_file, + ) + + root = Path("subject") + query = os.environ["QUERY"] + model = os.environ["TYPESAFE_MODEL"] + endpoint = os.environ["TYPESAFE_API_URL"] + key = os.environ["TYPESAFE_API_KEY"] + file_path = os.environ["FILE_PATH"] + subject = { + "repository": os.environ["SUBJECT_REPOSITORY"], + "revision": os.environ["SUBJECT_SHA"], + } + candidate = { + "score": 0.90, + "payload": {"path": file_path, "extension": ".rs"}, + } + + range_params = { + "window_lines": 140, + "parallel_threshold": 0.65, + "max_jumps": 2, + "max_file_epochs": 32, + } + frontier_params = { + "max_rounds": 10, + "max_actions_per_round": 2, + "probe_lines": 112, + "target_region_lines": 48, + "final_window_lines": 32, + "refine_threshold": 0.72, + "candidate_threshold": 0.55, + "gradient_threshold": 0.15, + "volatility_threshold": 0.10, + "stable_delta": 0.06, + "stable_rounds": 2, + "max_frontier_leaves": 24, + "final_max_candidates": 24, + } + + def coverage(state): + rows = sorted((int(a), int(b)) for a, b in state.get("coverage", [])) + merged = [] + for a, b in rows: + if not merged or a > merged[-1][1] + 1: + merged.append([a, b]) + else: + merged[-1][1] = max(merged[-1][1], b) + return sum(b - a + 1 for a, b in merged) / max(1, int(state["line_count"])) + + def evidence_ranges(result): + return [ + (int(e["start_line"]), int(e["end_line"])) + for item in result.get("files", []) + for e in item.get("evidence", []) + ] + + def overlap(left, right): + total = sum(b - a + 1 for a, b in left) + if not total: + return 0.0 + hit = 0 + for a, b in left: + for c, d in right: + hit += max(0, min(b, d) - max(a, c) + 1) + return min(1.0, hit / total) + + def action_counts(state): + counts = {} + for snapshot in state.get("action_history", []): + for action in snapshot.get("actions", []): + kind = action.get("kind") + counts[kind] = counts.get(kind, 0) + 1 + return counts + + range_trace = Trace("artifacts/range/trace.jsonl") + range_decider = SystemOneFileDecider(key, range_trace, endpoint, model) + started = time.perf_counter() + range_state, range_usage = run_file_runtime( + root, query, candidate, range_decider, range_trace, **range_params + ) + range_ms = (time.perf_counter() - started) * 1000 + range_files = canonical_file_results([range_state], 0.65) + range_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [range_state], "result_files": range_files, + "metrics": { + **range_usage, + "reads_executed": range_state["read_count"], + "evidence_regions": sum(len(x["evidence"]) for x in range_files), + "elapsed_ms": round(range_ms, 3), + }, + } + range_canonical = build_system_one_range_result(range_engine, model) + + frontier_trace = Trace("artifacts/frontier/trace.jsonl") + frontier_decider = RelevanceFrontierDecider( + key, frontier_trace, endpoint, model + ) + started = time.perf_counter() + frontier_state, frontier_usage = run_frontier_file( + root, query, candidate, frontier_decider, frontier_trace, + **frontier_params + ) + frontier_ms = (time.perf_counter() - started) * 1000 + frontier_files = canonical_frontier_results([frontier_state]) + frontier_engine = { + "query": query, "model": model, "subject": subject, + "file_states": [frontier_state], "result_files": frontier_files, + "metrics": { + **frontier_usage, + "reads_executed": len(frontier_state["observations"]), + "evidence_regions": sum(len(x["evidence"]) for x in frontier_files), + "elapsed_ms": round(frontier_ms, 3), + }, + } + frontier_canonical = build_system_one_relevance_frontier_result( + frontier_engine, model + ) + + report = { + "query": query, + "file": file_path, + "subject": subject, + "model": model, + "frontier_params": frontier_params, + "range_params": range_params, + "range": { + "metrics": range_engine["metrics"], + "termination": range_state["termination"], + "epochs": range_state["epoch"], + "source_coverage": coverage(range_state), + "evidence": evidence_ranges(range_canonical), + }, + "frontier": { + "metrics": frontier_engine["metrics"], + "termination": frontier_state["termination"], + "rounds": frontier_state["round"], + "frontier_coverage": frontier_state["frontier_coverage"], + "source_coverage": frontier_state["source_coverage"], + "action_counts": action_counts(frontier_state), + "evidence": evidence_ranges(frontier_canonical), + }, + } + report["comparison"] = { + "elapsed_ratio": frontier_ms / max(1.0, range_ms), + "read_ratio": len(frontier_state["observations"]) / max(1, range_state["read_count"]), + "input_token_ratio": frontier_usage["input_tokens"] / max(1, range_usage["input_tokens"]), + "output_token_ratio": frontier_usage["output_tokens"] / max(1, range_usage["output_tokens"]), + "frontier_evidence_covered_by_range": overlap( + evidence_ranges(frontier_canonical), evidence_ranges(range_canonical) + ), + "range_evidence_covered_by_frontier": overlap( + evidence_ranges(range_canonical), evidence_ranges(frontier_canonical) + ), + } + + Path("artifacts/range/result.json").write_text( + json.dumps(range_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/range/localization-result.json").write_text( + json.dumps(range_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/result.json").write_text( + json.dumps(frontier_engine, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/frontier/localization-result.json").write_text( + json.dumps(frontier_canonical, indent=2, ensure_ascii=False) + "\n" + ) + Path("artifacts/report.json").write_text( + json.dumps(report, indent=2, ensure_ascii=False) + "\n" + ) + print(json.dumps(report, indent=2, ensure_ascii=False)) + PY + + - name: Compare canonical results + run: | + set -euo pipefail + python3 harness/src/compare_localization_results.py \ + --left artifacts/range/localization-result.json \ + --right artifacts/frontier/localization-result.json \ + --output-json artifacts/localization-comparison.json \ + --output-markdown artifacts/localization-comparison.md + python3 - <<'PY' >> "$GITHUB_STEP_SUMMARY" + import json + r = json.load(open("artifacts/report.json")) + print("## Single-file Range vs Frontier v1c") + print("") + print("| Metric | Range | Frontier v1c |") + print("| --- | ---: | ---: |") + print("| elapsed ms | %s | %s |" % (r["range"]["metrics"]["elapsed_ms"], r["frontier"]["metrics"]["elapsed_ms"])) + print("| model calls | %s | %s |" % (r["range"]["metrics"]["model_calls"], r["frontier"]["metrics"]["model_calls"])) + print("| input tokens | %s | %s |" % (r["range"]["metrics"]["input_tokens"], r["frontier"]["metrics"]["input_tokens"])) + print("| output tokens | %s | %s |" % (r["range"]["metrics"]["output_tokens"], r["frontier"]["metrics"]["output_tokens"])) + print("| reads | %s | %s |" % (r["range"]["metrics"]["reads_executed"], r["frontier"]["metrics"]["reads_executed"])) + print("| source coverage | %.1f%% | %.1f%% |" % (r["range"]["source_coverage"] * 100, r["frontier"]["source_coverage"] * 100)) + print("| frontier coverage | - | %.1f%% |" % (r["frontier"]["frontier_coverage"] * 100)) + print("| evidence regions | %s | %s |" % (r["range"]["metrics"]["evidence_regions"], r["frontier"]["metrics"]["evidence_regions"])) + print("| termination | %s | %s |" % (r["range"]["termination"], r["frontier"]["termination"])) + print("") + print("Frontier action counts:", r["frontier"]["action_counts"]) + PY + cat artifacts/localization-comparison.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload experiment + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-frontier-v1c-${{ github.run_id }} + path: artifacts/ + if-no-files-found: error + retention-days: 90 + + blind-quality: + name: Blind quality review + needs: experiment + runs-on: ubuntu-24.04 + environment: ds + timeout-minutes: 120 + steps: + - name: Checkout harness + uses: actions/checkout@v4 + with: + repository: BestNathan/system-one-code-explore + ref: cfc3d696df6c9ecc833372cf616805d2b931d0af + path: harness + fetch-depth: 1 + persist-credentials: false + + - name: Checkout frozen subject + uses: actions/checkout@v4 + with: + repository: BestNathan/nession + ref: 7ac9b6e0c2bb43c52f83e7dd706c0c0dc0d7a1df + path: subject + fetch-depth: 1 + persist-credentials: false + + - uses: actions/setup-node@v4 + with: + node-version: "22.14.0" + + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + + - name: Install pinned Claude Code + run: npm install --global --no-fund --no-audit @anthropic-ai/claude-code@2.1.278 + + - name: Download experiment + uses: actions/download-artifact@v4 + with: + name: single-file-range-vs-frontier-v1c-${{ github.run_id }} + path: artifacts + + - name: Prepare anonymous candidates + run: | + set -euo pipefail + mkdir -p artifacts/quality/range artifacts/quality/frontier + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/range/localization-result.json \ + --candidate-id candidate-a \ + --output-candidate artifacts/quality/range/candidate.json \ + --output-prompt artifacts/quality/range/prompt.txt + python3 harness/src/localization_quality_evaluation.py prepare \ + --input artifacts/frontier/localization-result.json \ + --candidate-id candidate-b \ + --output-candidate artifacts/quality/frontier/candidate.json \ + --output-prompt artifacts/quality/frontier/prompt.txt + + - name: Evaluate range anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/range/prompt.txt \ + > ../artifacts/quality/range/evaluator.raw.jsonl + + - name: Evaluate frontier anonymously + working-directory: subject + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + CLAUDE_MODEL: ${{ vars.CLAUDE_MODEL }} + run: | + set -euo pipefail + claude -p \ + --output-format stream-json --verbose \ + --model "$CLAUDE_MODEL" --effort high --max-turns 80 \ + --permission-mode dontAsk --no-session-persistence \ + --setting-sources '' --strict-mcp-config \ + --mcp-config '{"mcpServers":{}}' --disable-slash-commands \ + --tools 'Bash,Read,Glob,Grep' \ + --allowedTools 'Bash(*)' 'Read' 'Glob' 'Grep' \ + --disallowedTools 'Edit' 'Write' 'Agent' 'Task' 'WebFetch' 'WebSearch' 'EndConversation' 'mcp__*' \ + --settings '{"disableAllHooks":true}' \ + < ../artifacts/quality/frontier/prompt.txt \ + > ../artifacts/quality/frontier/evaluator.raw.jsonl + + - name: Finalize blind scorecards + run: | + set -euo pipefail + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/range/candidate.json \ + --raw-jsonl artifacts/quality/range/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/range/scorecard.json \ + --output-markdown artifacts/quality/range/report.md + python3 harness/src/localization_quality_evaluation.py finalize \ + --candidate artifacts/quality/frontier/candidate.json \ + --raw-jsonl artifacts/quality/frontier/evaluator.raw.jsonl \ + --subject-root subject --model "$CLAUDE_MODEL" \ + --output-json artifacts/quality/frontier/scorecard.json \ + --output-markdown artifacts/quality/frontier/report.md + python3 harness/src/localization_quality_evaluation.py report \ + --candidate-a artifacts/quality/range/scorecard.json \ + --candidate-b artifacts/quality/frontier/scorecard.json \ + --candidate-a-label "Range runtime" \ + --candidate-b-label "Relevance frontier v1c" \ + --output-json artifacts/quality/evaluation-report.json \ + --output-markdown artifacts/quality/evaluation-report.md + cat artifacts/quality/evaluation-report.md >> "$GITHUB_STEP_SUMMARY" + + - name: Upload quality + if: always() + uses: actions/upload-artifact@v4 + with: + name: single-file-range-vs-frontier-v1c-quality-${{ github.run_id }} + path: artifacts/quality/ + if-no-files-found: error + retention-days: 90 diff --git a/experiments/FULL_READ_BASELINE_PROXY_MODEL_TRIGGER.md b/experiments/FULL_READ_BASELINE_PROXY_MODEL_TRIGGER.md new file mode 100644 index 00000000..94001239 --- /dev/null +++ b/experiments/FULL_READ_BASELINE_PROXY_MODEL_TRIGGER.md @@ -0,0 +1 @@ +trigger after proxy model catalog fix diff --git a/experiments/FULL_READ_BASELINE_RETRY.md b/experiments/FULL_READ_BASELINE_RETRY.md new file mode 100644 index 00000000..fc0d0d64 --- /dev/null +++ b/experiments/FULL_READ_BASELINE_RETRY.md @@ -0,0 +1 @@ +retry after checkout fix diff --git a/experiments/FULL_READ_BASELINE_SONNET46_TRIGGER.md b/experiments/FULL_READ_BASELINE_SONNET46_TRIGGER.md new file mode 100644 index 00000000..86be6d7f --- /dev/null +++ b/experiments/FULL_READ_BASELINE_SONNET46_TRIGGER.md @@ -0,0 +1 @@ +retry full-read baseline with explicit Claude Sonnet 4.6 diff --git a/experiments/FULL_READ_BASELINE_SONNET_TRIGGER.md b/experiments/FULL_READ_BASELINE_SONNET_TRIGGER.md new file mode 100644 index 00000000..2806140b --- /dev/null +++ b/experiments/FULL_READ_BASELINE_SONNET_TRIGGER.md @@ -0,0 +1 @@ +retry full-read baseline with Claude Code Sonnet diff --git a/experiments/FULL_READ_BASELINE_TRIGGER.md b/experiments/FULL_READ_BASELINE_TRIGGER.md new file mode 100644 index 00000000..9cff40dd --- /dev/null +++ b/experiments/FULL_READ_BASELINE_TRIGGER.md @@ -0,0 +1 @@ +trigger: full-read Claude relevance baseline 2026-09-24 diff --git a/experiments/FULL_READ_DS_BASELINE_TRIGGER.md b/experiments/FULL_READ_DS_BASELINE_TRIGGER.md new file mode 100644 index 00000000..d99c474b --- /dev/null +++ b/experiments/FULL_READ_DS_BASELINE_TRIGGER.md @@ -0,0 +1 @@ +trigger: full-read Claude Code reference field through ds model routing 2026-09-24 diff --git a/experiments/single-file-frontier-v1b-trigger.md b/experiments/single-file-frontier-v1b-trigger.md new file mode 100644 index 00000000..8893a4d2 --- /dev/null +++ b/experiments/single-file-frontier-v1b-trigger.md @@ -0,0 +1 @@ +Trigger controlled single-file frontier v1b experiment. diff --git a/experiments/single-file-frontier-v1c-trigger.md b/experiments/single-file-frontier-v1c-trigger.md new file mode 100644 index 00000000..0e556d2c --- /dev/null +++ b/experiments/single-file-frontier-v1c-trigger.md @@ -0,0 +1 @@ +Trigger balanced v1c single-file experiment on agent websocket.