diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 2501aa4f..b01fdefe 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -5,6 +5,10 @@ updates: schedule: interval: "weekly" open-pull-requests-limit: 5 + # This file reproduces retained campaigns and must match their observed inventory. + # New campaigns use benchmark-requirements-windows-py312-current.txt instead. + exclude-paths: + - "eval/configs/benchmark-requirements-windows-py312.txt" # MCP 2 changes the server API; migrate and qualify it separately. ignore: - dependency-name: "mcp" diff --git a/docs/BENCHMARK_EXPANSION_RUNBOOK.md b/docs/BENCHMARK_EXPANSION_RUNBOOK.md index dec2189e..d38b4351 100644 --- a/docs/BENCHMARK_EXPANSION_RUNBOOK.md +++ b/docs/BENCHMARK_EXPANSION_RUNBOOK.md @@ -15,21 +15,57 @@ are explicit unscored attempts, not fabricated failures or product quality concl ## Environment and freeze -Use a separate Python 3.12 environment. The tested Windows package versions are in +Use a separate Python 3.12 environment. Retained campaigns keep their tested Windows versions in [`benchmark-requirements-windows-py312.txt`](../eval/configs/benchmark-requirements-windows-py312.txt), with the observed distribution inventory in [`benchmark-environment-windows-py312.json`](../eval/configs/benchmark-environment-windows-py312.json). -This is an exact version lock for this platform, not a cross-platform wheel-hash lock. Optional -packages are confined to the evaluation environment; the NumPy-only core dependency contract is unchanged. +Keep both files unchanged when reproducing those campaigns from their frozen source revision. +They are historical evidence, not the dependency recommendation for new work. + +For a new campaign, install +[`benchmark-requirements-windows-py312-current.txt`](../eval/configs/benchmark-requirements-windows-py312-current.txt) +and capture the actual installed distributions into a new private inventory. The current file +includes PyJWT 2.15.1; its package pins do not establish that a new campaign was run or qualified. +These are version pins for Windows, not a cross-platform wheel-hash lock. Optional packages are +confined to the evaluation environment; the NumPy-only core dependency contract is unchanged. ```powershell -uv venv .private-eval/benchmark-20260915/venv --python 3.12 --seed -$py = '.private-eval/benchmark-20260915/venv/Scripts/python.exe' -uv pip install --python $py -r eval/configs/benchmark-requirements-windows-py312.txt +uv venv .private-eval/benchmark-current/venv --python 3.12 --seed +$py = '.private-eval/benchmark-current/venv/Scripts/python.exe' +uv pip install --python $py -r eval/configs/benchmark-requirements-windows-py312-current.txt uv pip install --python $py --no-deps -e . & $py -m eval.coding_corpus --verify +$dependencyLock = '.private-eval/benchmark-current/environment.json' +$captureEnvironment = @' +import importlib.metadata as metadata +import json +import platform +import re +import sys +from pathlib import Path + +names = {re.sub(r"[-_.]+", "-", item.metadata["Name"]).lower() + for item in metadata.distributions() if item.metadata["Name"]} +inventory = { + "schema": "engraphis-benchmark-environment/v1", + "python": platform.python_version(), + "platform": platform.platform(), + "distributions": {name: metadata.version(name) for name in sorted(names)}, +} +target = Path(sys.argv[1]) +target.parent.mkdir(parents=True, exist_ok=True) +with target.open("x", encoding="utf-8", newline="\n") as stream: + json.dump(inventory, stream, indent=2, sort_keys=True) + stream.write("\n") +'@ +$captureEnvironment | & $py - $dependencyLock ``` +Choose a new directory for each campaign. Inventory capture refuses to overwrite an existing +file. Pass that exact `$dependencyLock` to preparation and subsequent commands for this new +campaign. Installing changed packages requires a fresh inventory and manifest; never edit the +retained observed inventory to pretend that a different environment produced an old result. + The 400 scenarios have 40 family labels and the frozen 80/80/240 split. They are generated from shared templates. Family counts therefore do not imply 40 independent real repositories. The corpus is `implementation_team`, and cannot satisfy independent-human acceptance or @@ -67,9 +103,9 @@ Prepare into new paths after all relevant code and dependency changes are finish ```powershell & $py -m eval.benchmark_campaign --prepare ` - --manifest .private-eval/benchmark-20260915/campaign.json ` - --companion .private-eval/benchmark-20260915/comparisons.json ` - --dependency-lock eval/configs/benchmark-environment-windows-py312.json + --manifest .private-eval/benchmark-current/campaign.json ` + --companion .private-eval/benchmark-current/comparisons.json ` + --dependency-lock $dependencyLock ``` The original, retired API-route inputs are preserved at @@ -229,7 +265,7 @@ companion cells remain explicitly counted. Freeze selection before held-out work ```powershell & $py -m eval.benchmark_campaign --manifest --companion ` - --dependency-lock eval/configs/benchmark-environment-windows-py312.json ` + --dependency-lock ` --results --freeze-selection --selection-receipt ``` diff --git a/eval/configs/benchmark-requirements-windows-py312-current.txt b/eval/configs/benchmark-requirements-windows-py312-current.txt new file mode 100644 index 00000000..20bf8b51 --- /dev/null +++ b/eval/configs/benchmark-requirements-windows-py312-current.txt @@ -0,0 +1,116 @@ +# Dependencies for new Windows Python 3.12 campaigns; capture a fresh observed inventory. +# Retained campaigns keep benchmark-requirements-windows-py312.txt unchanged. +annotated-doc==0.0.5 +annotated-types==0.8.0 +anyio==4.15.1 +attrs==26.1.0 +av==18.1.0 +backoff==2.2.1 +certifi==2026.7.22 +cffi==2.1.1 +charset-normalizer==3.5.1 +click==8.5.0 +cloudpickle==3.1.2 +colorama==0.4.6 +cryptography==50.0.1 +ctranslate2==4.8.2 +distro==1.9.0 +fastapi==0.141.1 +faster-whisper==1.2.1 +filelock==3.32.7 +flatbuffers==25.12.19 +fsspec==2026.7.0 +graphiti-core==0.29.3 +greenlet==3.5.6 +grpcio==1.84.0 +h11==0.16.0 +h2==4.4.1 +hf-xet==1.6.0 +hpack==4.2.0 +httpcore==1.0.9 +httptools==0.8.0 +httpx==0.28.1 +httpx-sse==0.4.3 +huggingface_hub==1.31.0 +hyperframe==6.1.0 +idna==3.19 +iniconfig==2.3.0 +jinja2==3.1.6 +jiter==0.17.0 +joblib==1.6.0 +jsonschema==4.26.0 +jsonschema-specifications==2025.9.1 +markdown-it-py==4.2.0 +markupsafe==3.0.3 +mcp==1.30.0 +mdurl==0.1.2 +mem0ai==2.0.20 +mpmath==1.3.0 +narwhals==2.26.0 +neo4j==6.3.1 +networkx==3.6.1 +nodeenv==1.10.0 +numpy==2.4.5 +onnxruntime==1.30.0 +openai==2.36.0 +packaging==26.3 +pillow==12.3.0 +pip==26.2.1 +pluggy==1.6.0 +portalocker==3.2.0 +posthog==7.54.0 +protobuf==6.33.6 +psutil==7.2.2 +psycopg==3.3.5 +psycopg-binary==3.3.5 +pycparser==3.0 +pydantic==2.13.5 +pydantic-settings==2.15.0 +pydantic_core==2.46.5 +pygments==2.21.0 +pyjwt==2.15.1 +pypdf==6.18.1 +pyright==1.1.413 +pytesseract==0.3.13 +pytest==9.1.1 +pytest-asyncio==1.4.0 +python-dotenv==1.2.3 +python-multipart==0.0.32 +pytz==2026.3.post1 +pywin32==312 +pyyaml==6.0.3 +qdrant-client==1.19.0 +referencing==0.37.0 +regex==2026.9.10 +requests==2.34.2 +rich==15.0.0 +rpds-py==2026.6.3 +ruff==0.16.7 +safetensors==0.8.0 +scikit-learn==1.9.1 +scipy==1.18.1 +sentence-transformers==6.0.0 +setuptools==84.0.0 +shellingham==1.5.4 +sniffio==1.3.1 +sqlalchemy==2.0.54 +sqlite-vec==0.1.9 +sse-starlette==3.4.11 +starlette==1.6.0 +sympy==1.14.0 +tenacity==9.1.4 +threadpoolctl==3.7.0 +tokenizers==0.23.2 +torch==2.13.0 +tqdm==4.70.1 +transformers==5.17.0 +tree-sitter==0.26.0 +tree-sitter-language-pack==1.17.0 +typer==0.27.2 +typing-inspection==0.4.4 +typing_extensions==4.16.0 +tzdata==2026.4 +urllib3==2.8.0 +uvicorn==0.53.0 +watchfiles==1.2.0 +websockets==17.1