diff --git a/.github/workflows/_llm_server.yml b/.github/workflows/_llm_server.yml index 12d04423112..fbd6a71e659 100644 --- a/.github/workflows/_llm_server.yml +++ b/.github/workflows/_llm_server.yml @@ -44,6 +44,7 @@ jobs: python -m pytest -q \ examples/llm_server/python/tests \ + examples/llm_server/evals/terminal_bench/tests \ examples/models/muse-glimmer/tests/test_serve.py cmake -S . -B cmake-out \ diff --git a/.github/workflows/llm-server-evals.yml b/.github/workflows/llm-server-evals.yml new file mode 100644 index 00000000000..6c385056ecb --- /dev/null +++ b/.github/workflows/llm-server-evals.yml @@ -0,0 +1,92 @@ +name: LLM Server Evaluations + +on: + schedule: + - cron: '17 6 * * *' + workflow_dispatch: + inputs: + suite: + description: Terminal-Bench task subset + type: choice + options: [smoke, nightly] + default: smoke + +permissions: + contents: read + +jobs: + terminal-bench: + if: ${{ vars.LLM_SERVER_EVAL_TARGETS != '' }} + strategy: + fail-fast: false + matrix: + target: ${{ fromJSON(vars.LLM_SERVER_EVAL_TARGETS || '[{"name":"unconfigured","runner":"ubuntu-22.04"}]') }} + name: Terminal-Bench (${{ matrix.target.name }}) + runs-on: ${{ matrix.target.runner }} + timeout-minutes: 180 + concurrency: + group: llm-server-evals-${{ matrix.target.name }} + cancel-in-progress: false + env: + EVAL_CONFIG: ${{ matrix.target.config }} + EVAL_PREPARE: ${{ matrix.target.prepare }} + EVAL_SUITE: ${{ inputs.suite || 'nightly' }} + EVAL_TARGET: ${{ matrix.target.name }} + steps: + - name: Initialize artifacts + shell: bash + run: | + set -euo pipefail + EVAL_ARTIFACTS="${RUNNER_TEMP}/llm-server-evals-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}-${EVAL_TARGET}" + mkdir -p "${EVAL_ARTIFACTS}" + echo "EVAL_ARTIFACTS=${EVAL_ARTIFACTS}" >> "${GITHUB_ENV}" + if [[ "${EVAL_CONFIG}" != /* ]]; then + EVAL_CONFIG="${GITHUB_WORKSPACE}/${EVAL_CONFIG}" + fi + echo "EVAL_CONFIG=${EVAL_CONFIG}" >> "${GITHUB_ENV}" + printf '%s\n' "${GITHUB_SHA}" "${RUNNER_OS}" "${RUNNER_ARCH}" > "${EVAL_ARTIFACTS}/ci.txt" + - uses: actions/checkout@v4 + with: + submodules: recursive + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + - name: Prepare the current checkout and model + shell: bash + run: | + set -euo pipefail + mkdir -p "${EVAL_ARTIFACTS}" + test -f "${EVAL_PREPARE}" + cp "${EVAL_PREPARE}" "${EVAL_ARTIFACTS}/prepare.sh" + bash "${EVAL_PREPARE}" "${GITHUB_WORKSPACE}" "${EVAL_CONFIG}" 2>&1 | tee "${EVAL_ARTIFACTS}/prepare.log" + test -f "${EVAL_CONFIG}" + cp "${EVAL_CONFIG}" "${EVAL_ARTIFACTS}/model.toml" + - name: Set up Terminal-Bench + shell: bash + run: | + set -euo pipefail + bash examples/llm_server/evals/setup.sh terminal-bench --ci 2>&1 | tee "${EVAL_ARTIFACTS}/setup.log" + - name: Evaluate + shell: bash + run: | + set -euo pipefail + bash examples/llm_server/evals/run.sh terminal-bench \ + --config "${EVAL_CONFIG}" --suite "${EVAL_SUITE}" \ + --executorch-root "${GITHUB_WORKSPACE}" \ + --run-dir "${EVAL_ARTIFACTS}/run" 2>&1 | tee "${EVAL_ARTIFACTS}/evaluation.log" + - name: Publish summary + if: always() + shell: bash + run: | + if [[ -f "${EVAL_ARTIFACTS}/run/summary.md" ]]; then + cat "${EVAL_ARTIFACTS}/run/summary.md" >> "${GITHUB_STEP_SUMMARY}" + else + echo 'Evaluation did not produce a summary; inspect preparation and setup logs.' >> "${GITHUB_STEP_SUMMARY}" + fi + - uses: actions/upload-artifact@v4 + if: always() + with: + name: terminal-bench-${{ matrix.target.name }} + path: ${{ env.EVAL_ARTIFACTS }} + if-no-files-found: warn + retention-days: 30 diff --git a/examples/llm_server/README.md b/examples/llm_server/README.md index 3d7f7e3d64d..f8af7fa9177 100644 --- a/examples/llm_server/README.md +++ b/examples/llm_server/README.md @@ -8,6 +8,7 @@ examples/llm_server/ spec/ # language-neutral OpenAI contract ExecuTorch targets conformance/ # one test suite every language server must pass python/ # Python server implementation (current) + evals/ # task accuracy and performance through the server # cpp/ # future: no-Python single-binary server ``` @@ -106,3 +107,9 @@ Reliability guidance: `tools` were included in the request. - If a request fails with `unsupported_parameter`, remove or disable that OpenAI knob in your pi/client config. + +## Evaluate with Terminal-Bench + +The [evaluation guide](evals/README.md) provides setup and run commands for +Terminal-Bench, configuration templates in `evals/configs/`, recorded task and +performance metrics, and periodic CI on provisioned Linux GPU and macOS runners. diff --git a/examples/llm_server/evals/README.md b/examples/llm_server/evals/README.md new file mode 100644 index 00000000000..aa04fa5f9e1 --- /dev/null +++ b/examples/llm_server/evals/README.md @@ -0,0 +1,146 @@ +# LLM server evaluations + +Run end-to-end evaluations through the ExecuTorch LLM server. Terminal-Bench is +available today; additional harnesses such as tau2 and GuideLLM can be added as +sibling directories with their own dependencies and metrics. TOML configurations +live in `configs/`. + +## Setup + +From the repository root: + +```bash +bash examples/llm_server/evals/setup.sh terminal-bench +``` + +Setup reuses a working Docker installation, or installs Docker Engine on Ubuntu +and Colima with the Docker CLI on macOS. Installing system packages can require +administrator authentication; macOS installation requires Homebrew. On a +provisioned CI runner, add `--ci` to require existing Docker without changing +system services. + +The script creates an isolated Python 3.12 environment with Harbor 0.22.0 and +downloads a pinned Terminal-Bench task subset. Harbor installs mini-SWE-agent +2.4.6 inside each task container when the trial starts. No host installation of +the agent is needed. Setup can be rerun; it preserves an unchanged task cache +and reports modifications instead of overwriting them. + +Dependencies and tasks live under `~/.cache/executorch-evals` (or +`$XDG_CACHE_HOME/executorch-evals`). Set `EXECUTORCH_EVAL_CACHE` consistently for +setup and run to use another location. Setup does not replace your ExecuTorch +Python environment. Prepare an exported model, compatible native worker, and +[server environment](../python/README.md) +before running an evaluation. + +## Configure once + +```bash +cp examples/llm_server/evals/configs/terminal-bench.example.toml examples/llm_server/evals/configs/terminal-bench.local.toml +``` + +Files named `*.local.toml` are ignored by Git so machine-specific paths stay local. +Edit the worker, model, tokenizer, and server Python paths. Set `max_context` to +a capacity supported by the export and choose `max_output_tokens` independently. +The example uses a 2,048-token context for an integration check; use a qualified +model and appropriate context/output budgets for task-quality evaluation. + +## Run + +```bash +bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --suite smoke +``` + +The command checks prerequisites, starts a fresh server for each trial, probes +the server from a Docker container, runs Harbor, and cleans up its processes. +Linux automatically maps `host.docker.internal` to the host gateway for Harbor +task containers. The launcher uses `host.lima.internal` for a Colima context on +macOS and `host.docker.internal` for Docker Desktop. Use `agent_base_url` for a +custom container route; the server must listen on an address containers can reach. + +`smoke` runs `fix-git`; `nightly` adds `openssl-selfsigned-cert`. Both are pinned +subsets, not a full Terminal-Bench score. Use `--task /path/to/task` for a custom +task, or `--task-root /path/to/tasks` with a named suite. + +```bash +# Inspect dependencies without running a trial. +bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --check +# Record the configuration and commands without launching processes. +bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --dry-run +# Run the same subset used by periodic CI. +bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --suite nightly +``` + +`--check` checks host prerequisites. The container route is tested after the +server starts during an actual run; a particular task's custom network can still +require additional configuration. See the [Terminal-Bench guide](terminal_bench/README.md) +for budget controls, launcher configuration, and resuming runs. + +## Results + +The command prints a fresh results directory for every invocation, under +`/terminal-bench/runs` unless `output_root` is configured. It contains +`summary.md`, `results.json`, `results.tsv`, resolved configuration, source and +asset checksums, exact commands, server logs, agent trajectories, and verifier +outputs. Setup dependency versions and the selected task revision are recorded. + +Results distinguish task rewards from infrastructure errors and report executed +tool observations, token usage, and prefill/decode timings. Reward zero with no +tool use does not establish a meaningful agent rollout. + +## Adding an evaluation + +Add a sibling package with its own `__main__.py`, `setup.sh`, `run.sh`, +requirements, documentation, and tests; place its TOML examples in `configs/` and register its entry points +in the two top-level scripts. Keep harness-specific behavior and dependencies inside that package. +Extract shared utilities when a second integration needs them. Server protocol +tests remain beside the server implementation in `../python/tests`. + +## Periodic CI on Linux GPU and macOS + +The [LLM Server Evaluations workflow](../../../.github/workflows/llm-server-evals.yml) +runs the nightly subset at 06:17 UTC and supports manual dispatch. It becomes +active when the repository variable `LLM_SERVER_EVAL_TARGETS` is configured. +Each matrix entry selects its own runner, model configuration, and preparation +script. For example, after provisioning runners with these labels: + +```json +[ + { + "name": "linux-gpu", + "runner": ["self-hosted", "Linux", "X64", "executorch-evals-gpu"], + "config": "examples/llm_server/evals/configs/linux-gpu.local.toml", + "prepare": "/opt/executorch-evals/prepare.sh" + }, + { + "name": "macos-mlx", + "runner": ["self-hosted", "macOS", "ARM64", "executorch-evals-mlx"], + "config": "examples/llm_server/evals/configs/macos-mlx.local.toml", + "prepare": "/Users/runner/executorch-evals/prepare.sh" + } +] +``` + +Both runners need working Docker and Compose accessible by the runner user. The +Linux GPU runner also needs its model backend and drivers. The macOS runner +needs its selected model backend (for example, Metal/MLX) and Docker Desktop or Colima with virtualization support; +an ordinary hosted macOS VM is not assumed to meet these requirements. + +The preparation script receives the current checkout and configuration paths: + +```bash +bash /path/to/prepare.sh "$GITHUB_WORKSPACE" "$GITHUB_WORKSPACE/examples/llm_server/evals/configs/macos-mlx.local.toml" +``` + +CI resolves relative configuration paths against the checkout before calling +the preparation script. It must build the compatible worker from that checkout, prepare the server +environment, obtain pinned model/tokenizer assets, and write or update the +configuration to those paths. Cached weights may be reused; a worker built from +an older checkout would not evaluate the C++ changes under test. Keep model +revision, quantization, context, output allowance, and task settings fixed when +comparing runs. Report Linux and macOS performance separately. + +CI invokes the same setup and run scripts as local development. It uploads +configuration, preparation/setup logs, and evaluation artifacts even on failure. +Infrastructure errors fail the job; task rewards are reported without a quality +threshold until a reliable model baseline is established. Lightweight driver +tests continue to run in the existing LLM server CI workflow. diff --git a/examples/llm_server/evals/__init__.py b/examples/llm_server/evals/__init__.py new file mode 100644 index 00000000000..2e41cd717f6 --- /dev/null +++ b/examples/llm_server/evals/__init__.py @@ -0,0 +1,5 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. diff --git a/examples/llm_server/evals/configs/.gitignore b/examples/llm_server/evals/configs/.gitignore new file mode 100644 index 00000000000..a4ce64c7a7c --- /dev/null +++ b/examples/llm_server/evals/configs/.gitignore @@ -0,0 +1 @@ +*.local.toml diff --git a/examples/llm_server/evals/configs/terminal-bench.example.toml b/examples/llm_server/evals/configs/terminal-bench.example.toml new file mode 100644 index 00000000000..76bdc5bd340 --- /dev/null +++ b/examples/llm_server/evals/configs/terminal-bench.example.toml @@ -0,0 +1,27 @@ +# Copy to terminal-bench.local.toml and adjust paths (relative to this file). +# setup.sh downloads the tasks for the named suites. +suite = "smoke" +worker_bin = "/path/to/model_worker" +model_path = "/path/to/model.pte" +tokenizer_path = "/path/to/tokenizer.json" +hf_tokenizer = "/path/to/pinned-hf-tokenizer" +server_python = "/path/to/executorch-env/bin/python" +model_id = "qwen3" + +max_context = 2048 +max_output_tokens = 512 +step_limit = 100 +attempts = 1 +# Enable only when the worker advertises named-session support. +session_affinity = false +# Optional model-specific launcher; it must accept the common server flags. +# server_module = "executorch.examples.llm_server.python.server" +# server_arg = ["--assistant-header=<|im_start|>assistant\n"] +thinking = false + +host = "0.0.0.0" +port = 8000 +# run.sh chooses the Docker/Colima host route. Override for custom networking: +# agent_base_url = "http://host.docker.internal:8000/v1" +# Results default to the evaluation cache. Override if desired: +# output_root = "/path/to/evaluation-results" diff --git a/examples/llm_server/evals/run.sh b/examples/llm_server/evals/run.sh new file mode 100644 index 00000000000..96cf04fee22 --- /dev/null +++ b/examples/llm_server/evals/run.sh @@ -0,0 +1,18 @@ +#!/usr/bin/env bash +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. +set -euo pipefail + +evals_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd) +case "${1:---help}" in + terminal-bench) harness=terminal_bench ;; + --help|-h) + echo "Usage: bash examples/llm_server/evals/run.sh terminal-bench [driver options]" + exit 0 ;; + *) echo "Unsupported evaluation: $1" >&2; exit 2 ;; +esac +shift +exec bash "${evals_dir}/${harness}/run.sh" "$@" diff --git a/examples/llm_server/evals/setup.sh b/examples/llm_server/evals/setup.sh new file mode 100644 index 00000000000..77ed908ab93 --- /dev/null +++ b/examples/llm_server/evals/setup.sh @@ -0,0 +1,18 @@ +#!/usr/bin/env bash +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. +set -euo pipefail + +evals_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd) +case "${1:---help}" in + terminal-bench) harness=terminal_bench ;; + --help|-h) + echo "Usage: bash examples/llm_server/evals/setup.sh terminal-bench [--ci]" + exit 0 ;; + *) echo "Unsupported evaluation: $1" >&2; exit 2 ;; +esac +shift +exec bash "${evals_dir}/${harness}/setup.sh" "$@" diff --git a/examples/llm_server/evals/terminal_bench/README.md b/examples/llm_server/evals/terminal_bench/README.md new file mode 100644 index 00000000000..ac166afeef2 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/README.md @@ -0,0 +1,103 @@ +# Terminal-Bench with the LLM server + +Use the [evaluation quick start](../README.md) to install Docker and Harbor and +download the pinned tasks. With an exported model and configured worker, run: + +```bash +bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --suite smoke +``` + +Copy [the example configuration](../configs/terminal-bench.example.toml), fill in +local model/tokenizer/worker paths, and select your budgets. The driver starts +and stops the Python server and worker, invokes Harbor, and records a fresh +results directory for every invocation. It defaults to one attempt per task. + +Install the [server dependencies](../../python/requirements.txt) in the interpreter +selected by `server_python`. Setup installs [evaluation dependencies](requirements.txt) +in a separate Python 3.12 environment; `harbor_bin` can select a custom installation. +Use `--suite smoke` or `--suite nightly` for the downloaded task subsets, or +`--task` for a local directory containing `task.toml` and `instruction.md`. +Tasks retain their own container images, verifier, and time limits. + +Harbor 0.22.0 installs mini-SWE-agent 2.4.6 in each task container. Model export, +worker builds, and the server environment remain separate from harness setup. + +## Configuration and controls + +```bash +# Check local dependencies and Docker without starting a trial. +bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --check +# Record resolved settings and exact commands without starting processes. +bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --dry-run +# Resume an unchanged run, retaining completed trials. +bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --run-dir /path/to/run --resume +``` + +TOML keys match CLI options with underscores. Paths are relative to the config +file; explicit CLI options override the corresponding config value, including +lists. Omitting `run_dir` creates a unique directory beneath `output_root`. +Resume compares settings, source revision/diff, driver, task contents, and +worker/model/tokenizer checksums. Code or asset changes require a fresh run. + +| Control | Meaning | +| --- | --- | +| `max_context` | Context limit enforced by the Python server; must fit the worker and export | +| `max_output_tokens` | Per-response generation allowance reserved within that context | +| `step_limit` | Agent iteration cap | +| `attempts` | Independent trials per task, each with a fresh server | +| `session_affinity` | Opt into a named session for the whole task; requires worker support | + +Each admitted prompt must satisfy `prompt_tokens + max_output_tokens <= max_context`. +The context flag cannot enlarge the exported model or change the worker's own +capacity. The model's real HF tokenizer is required for templating and counting. + +The default launcher is `executorch.examples.llm_server.python.server`. For a +model-specific launcher, set `server_module` and pass additional options through +`server_arg`, a list of `--name=value` strings. It must accept the common worker, +model, tokenizer, host, port, and context flags shown by `--dry-run`. For example, +`server_arg = ["--assistant-header=<|im_start|>assistant\n"]` configures the generic +server's assistant header. Match its tool parser to the model's output format. +Worker-specific settings belong in that model's launcher or an executable wrapper +selected by `worker_bin`; the driver does not assume a cache layout or worker API. + +The container-visible `agent_base_url` must point to this server's port and end +in `/v1`. The launcher selects `host.lima.internal` for Colima and +`host.docker.internal` for Docker Desktop. On Linux, the driver supplies a Harbor +Compose overlay mapping `host.docker.internal` to the Docker host gateway. +`--check` validates host prerequisites; actual runs probe `/health` from a +container before starting Harbor. A task with custom networking may need a +different `agent_base_url`. Inspect `container-check.log`, `server.log`, and +`harbor.log` when diagnosing failures. + +## Execution and evidence + +```text +Harbor / mini-SWE-agent / task container + → OpenAI HTTP → Python LLM server → worker → model +``` + +The harness owns task execution, full conversation history, and verification. +Every trial starts a fresh server and worker. With `session_affinity = true`, +it also resets a unique named session before execution and closes it afterward. +Without affinity, it supports workers that only accept anonymous requests. +Processes are cleaned up on completion or interruption. Trials run sequentially. + +Each run saves `manifest.json`, `plan.json`, `summary.md`, `results.json`, and +`results.tsv`. Each trial saves exact commands, server/Harbor logs, the agent +trajectory, verifier outputs, and `outcome.json`. Results include task reward, +agent exit reason, observed tool responses, token usage, peak prompt length, +and prefill/decode timings. Startup and task wall time are recorded separately. +Cumulative input tokens measure traffic; peak prompt length measures context +pressure. Compare timings only across like-for-like models and hardware. + +Reward zero with `RepeatedFormatError` and no tool observations means the model +did not execute a meaningful task rollout. An earlier Qwen3-0.6B smoke run had +this outcome; it validated connectivity, not task quality. Use a qualified model +and suitable context/output budgets before interpreting accuracy results. +Missing server metrics, context violations, or failed cleanup mark a trial as a +harness error. Infrastructure failures return a nonzero status; valid scored +trials return zero even when their reward is zero. + +Driver tests cover configuration, process isolation, interruption, resume, +container routing, metric reporting, and the real Python HTTP server with a +fixture worker. Fixture rewards are not Terminal-Bench scores. diff --git a/examples/llm_server/evals/terminal_bench/__init__.py b/examples/llm_server/evals/terminal_bench/__init__.py new file mode 100644 index 00000000000..2e41cd717f6 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/__init__.py @@ -0,0 +1,5 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. diff --git a/examples/llm_server/evals/terminal_bench/__main__.py b/examples/llm_server/evals/terminal_bench/__main__.py new file mode 100644 index 00000000000..7d860e3bb85 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/__main__.py @@ -0,0 +1,14 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +import signal + +from .runner import main + + +if __name__ == "__main__": + signal.signal(signal.SIGTERM, signal.default_int_handler) + raise SystemExit(main()) diff --git a/examples/llm_server/evals/terminal_bench/prepare.py b/examples/llm_server/evals/terminal_bench/prepare.py new file mode 100644 index 00000000000..1f02c97fed9 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/prepare.py @@ -0,0 +1,95 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +"""Fetch the pinned task subset without modifying an existing task checkout.""" + +import fcntl +import json +import os +import platform +import subprocess +import tempfile +from pathlib import Path + +from .suite import HARNESS_CACHE, SUITES, TASK_ROOT + + +def prepare_tasks(destination: Path = TASK_ROOT) -> None: + dataset = SUITES["dataset"] + tasks = sorted({task for suite in SUITES["suites"].values() for task in suite}) + destination.parent.mkdir(parents=True, exist_ok=True) + with (destination.parent / ".setup.lock").open("a") as lock: + fcntl.flock(lock, fcntl.LOCK_EX) + if not destination.exists(): + with tempfile.TemporaryDirectory(dir=destination.parent) as temp: + checkout = Path(temp) / "tasks" + checkout.mkdir() + commands = [ + ["init", "--quiet"], + ["remote", "add", "origin", dataset["repository"]], + ["sparse-checkout", "init", "--cone"], + ["sparse-checkout", "set", *tasks], + [ + "fetch", + "--depth=1", + "origin", + dataset["revision"], + ], + ["checkout", "--detach", "FETCH_HEAD"], + ] + for command in commands: + subprocess.run(["git", "-C", str(checkout), *command], check=True) + checkout.rename(destination) + head = subprocess.check_output( + ["git", "-C", str(destination), "rev-parse", "HEAD"], text=True + ).strip() + dirty = subprocess.check_output( + ["git", "-C", str(destination), "status", "--porcelain"], text=True + ).strip() + if head != dataset["revision"] or dirty: + raise RuntimeError( + f"Task cache is modified: {destination}. Use a fresh EXECUTORCH_EVAL_CACHE." + ) + for task in tasks: + for filename in ("task.toml", "instruction.md"): + if not (destination / task / filename).is_file(): + raise RuntimeError( + f"Incomplete task cache: {destination / task / filename}" + ) + print(f"Pinned tasks ready: {destination}") + + +def main() -> None: + prepare_tasks() + environment = { + "dataset": SUITES["dataset"], + "platform": platform.platform(), + "machine": platform.machine(), + "uv": subprocess.check_output(["uv", "--version"], text=True).strip(), + "docker": json.loads( + subprocess.check_output( + ["docker", "version", "--format", "{{json .}}"], text=True + ) + ), + "python": subprocess.check_output( + [str(HARNESS_CACHE / "venv/bin/python"), "--version"], text=True + ).strip(), + "packages": subprocess.check_output( + ["uv", "pip", "freeze", "--python", str(HARNESS_CACHE / "venv/bin/python")], + text=True, + ).splitlines(), + "docker_context": subprocess.check_output( + ["docker", "context", "show"], text=True + ).strip(), + "docker_config": os.environ.get("DOCKER_CONFIG"), + } + (HARNESS_CACHE / "environment.json").write_text( + json.dumps(environment, indent=2) + "\n" + ) + + +if __name__ == "__main__": + main() diff --git a/examples/llm_server/evals/terminal_bench/requirements.txt b/examples/llm_server/evals/terminal_bench/requirements.txt new file mode 100644 index 00000000000..8f1ee673ddd --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/requirements.txt @@ -0,0 +1,3 @@ +# Install separately from the model server; Harbor requires Python 3.12+. +harbor==0.22.0 +tomli>=2.0; python_version < "3.11" diff --git a/examples/llm_server/evals/terminal_bench/run.sh b/examples/llm_server/evals/terminal_bench/run.sh new file mode 100644 index 00000000000..720789c708c --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/run.sh @@ -0,0 +1,40 @@ +#!/usr/bin/env bash +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. +set -euo pipefail + +evals_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd) +repo_dir=$(cd -- "${evals_dir}/../../.." && pwd) +harness_name=terminal-bench +harness=terminal_bench +export EXECUTORCH_EVAL_CACHE="${EXECUTORCH_EVAL_CACHE:-${XDG_CACHE_HOME:-${HOME}/.cache}/executorch-evals}" + + +if [[ "$(uname -s)" == Linux ]] && ! docker info >/dev/null 2>&1 && + [[ " $(id -nG) " != *" docker "* ]] && + [[ " $(id -nG "$(id -un)") " == *" docker "* ]]; then + # setup.sh may have just added Docker access while the caller's shell still + # has its original supplementary groups. + printf -v retry '%q ' bash "${evals_dir}/run.sh" "${harness_name}" "$@" + exec sg docker -c "${retry}" +fi + +eval_python="${EXECUTORCH_EVAL_CACHE}/terminal-bench/venv/bin/python" +if [[ ! -x "${eval_python}" ]]; then + echo "Run bash ${evals_dir}/setup.sh terminal-bench first." >&2 + exit 2 +fi +if [[ -z "${EXECUTORCH_EVAL_SERVER_PYTHON:-}" ]]; then + EXECUTORCH_EVAL_SERVER_PYTHON=$(command -v python || command -v python3) + export EXECUTORCH_EVAL_SERVER_PYTHON +fi +if [[ -z "${EXECUTORCH_EVAL_AGENT_HOST:-}" ]] && [[ "$(uname -s)" == Darwin ]]; then + if [[ "$(docker context show 2>/dev/null || true)" == colima* ]]; then + export EXECUTORCH_EVAL_AGENT_HOST=host.lima.internal + fi +fi +export PYTHONPATH="${repo_dir}/src${PYTHONPATH:+:${PYTHONPATH}}" +exec "${eval_python}" -m "executorch.examples.llm_server.evals.${harness}" "$@" diff --git a/examples/llm_server/evals/terminal_bench/runner.py b/examples/llm_server/evals/terminal_bench/runner.py new file mode 100644 index 00000000000..4cc13950396 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/runner.py @@ -0,0 +1,1020 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +"""Run Terminal-Bench tasks through the ExecuTorch LLM server.""" + +from __future__ import annotations + +import argparse +import csv +import fcntl +import hashlib +import json +import os +import re +import shutil +import signal +import socket +import subprocess +import sys +import time +import urllib.error +import urllib.parse +import urllib.request +import uuid +from dataclasses import asdict, dataclass +from datetime import datetime, timezone +from pathlib import Path + +from .suite import HARNESS_CACHE, SUITES, SUITES_PATH, TASK_ROOT + +HARBOR_VERSION = "0.22.0" +AGENT_VERSION = "2.4.6" +PROBE_IMAGE = "curlimages/curl:8.12.1" +SERVER_MODULE = "executorch.examples.llm_server.python.server" +REPO_ROOT = Path(__file__).resolve().parents[4] +HTTP = urllib.request.build_opener(urllib.request.ProxyHandler({})) + + +@dataclass(frozen=True) +class Trial: + key: str + task: str + attempt: int + session_id: str + + +def parser() -> argparse.ArgumentParser: + p = argparse.ArgumentParser(description=__doc__) + p.add_argument( + "--config", type=Path, help="TOML configuration; CLI values override it" + ) + p.add_argument( + "--task", + type=Path, + action="append", + help="Local Harbor task directory; repeat for a subset", + ) + p.add_argument("--suite", choices=tuple(SUITES["suites"])) + p.add_argument("--task-root", type=Path, default=TASK_ROOT) + p.add_argument( + "--run-dir", type=Path, help="Existing directory for --resume, or a new run" + ) + p.add_argument("--output-root", type=Path, default=HARNESS_CACHE / "runs") + p.add_argument("--executorch-root", type=Path, default=REPO_ROOT) + p.add_argument( + "--server-python", + default=os.environ.get("EXECUTORCH_EVAL_SERVER_PYTHON", sys.executable), + ) + p.add_argument("--server-module", default=SERVER_MODULE) + p.add_argument( + "--server-arg", + action="append", + default=[], + help="Extra launcher option as --name=value; repeat for model-specific settings", + ) + p.add_argument( + "--session-affinity", + action="store_true", + help="Reuse a named session across turns; requires a worker with named-session support", + ) + p.add_argument("--worker-bin", type=Path, required=True) + p.add_argument("--model-path", type=Path, required=True) + p.add_argument("--tokenizer-path", type=Path, required=True) + p.add_argument( + "--hf-tokenizer", + required=True, + help="Prefer a pinned local tokenizer directory", + ) + p.add_argument("--model-id", default="qwen3") + p.add_argument("--max-context", type=int, required=True) + p.add_argument("--max-output-tokens", type=int, default=512) + p.add_argument("--step-limit", type=int, default=100) + p.add_argument("--attempts", type=int, default=1) + p.add_argument( + "--thinking", + action="store_true", + help="Keep the model template's thinking default", + ) + p.add_argument( + "--host", + default="0.0.0.0", + help="Server listen address, reachable from task containers", + ) + p.add_argument("--port", type=int, default=8000) + p.add_argument( + "--agent-base-url", + help="Container-visible URL; defaults to host.docker.internal (host.lima.internal with the Colima launcher)", + ) + p.add_argument("--harbor-bin", default=str(HARNESS_CACHE / "venv/bin/harbor")) + p.add_argument("--startup-timeout", type=float, default=180) + p.add_argument("--resume", action="store_true") + mode = p.add_mutually_exclusive_group() + mode.add_argument( + "--dry-run", + action="store_true", + help="Write the manifest and exact launch plans only", + ) + mode.add_argument( + "--check", + action="store_true", + help="Check prerequisites without starting a trial", + ) + return p + + +def config_arguments(p: argparse.ArgumentParser, argv: list[str]) -> list[str]: + probe = argparse.ArgumentParser(add_help=False) + probe.add_argument("--config", type=Path) + config, _ = probe.parse_known_args(argv) + if config.config is None: + return argv + if sys.version_info >= (3, 11): + import tomllib + else: + import tomli as tomllib + + path = config.config.expanduser().resolve() + with path.open("rb") as stream: + values = tomllib.load(stream) + actions = {action.dest: action for action in p._actions} + overrides = {arg.split("=", 1)[0] for arg in argv if arg.startswith("--")} + result = [] + for key, value in values.items(): + if key not in actions or key in { + "help", + "config", + "resume", + "check", + "dry_run", + }: + raise ValueError(f"unknown or command-only configuration key: {key}") + action = actions[key] + option = action.option_strings[0] + if ( + option in overrides + or (key == "task" and "--suite" in overrides) + or (key == "suite" and "--task" in overrides) + ): + continue + result.extend(config_value_arguments(key, value, action, path.parent)) + return result + argv + + +def config_value_arguments(key, value, action, base: Path) -> list[str]: + option = action.option_strings[0] + paths = { + "task", + "task_root", + "run_dir", + "output_root", + "executorch_root", + "worker_bin", + "model_path", + "tokenizer_path", + } + if isinstance(action, argparse._StoreTrueAction): + if not isinstance(value, bool): + raise ValueError(f"{key} must be a boolean") + if value: + return [option] + return [] + result = [] + items = value if isinstance(value, list) else [value] + if not items or any(isinstance(item, (bool, dict, list)) for item in items): + raise ValueError(f"invalid configuration value: {key}") + strings = [] + for item in items: + text = str(item) + if key in paths or ( + key in {"server_python", "harbor_bin", "hf_tokenizer"} + and (text.startswith(("~", ".", "/")) or (base / text).exists()) + ): + text = str((base / Path(text).expanduser()).resolve()) + strings.append(text) + if isinstance(action, argparse._AppendAction): + result.extend(f"{option}={item}" for item in strings) + elif action.nargs == "+": + result.extend([option, *strings]) + elif len(strings) == 1: + result.append(f"{option}={strings[0]}") + else: + raise ValueError(f"{key} expects a single value") + return result + + +def validate(args) -> None: + if args.resume and args.run_dir is None: + raise ValueError("--resume requires an explicit --run-dir") + args.output_root = args.output_root.expanduser().resolve() + if args.run_dir is None: + stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") + args.run_dir = args.output_root / f"{stamp}-{uuid.uuid4().hex[:8]}" + args.run_dir = args.run_dir.expanduser().resolve() + args.executorch_root = args.executorch_root.resolve() + args.worker_bin = args.worker_bin.resolve() + args.model_path = args.model_path.resolve() + args.tokenizer_path = args.tokenizer_path.resolve() + resolve_tasks(args) + if not 0 < args.max_output_tokens < args.max_context: + raise ValueError( + "output reserve must be positive and smaller than --max-context" + ) + if min(args.step_limit, args.attempts) < 1: + raise ValueError("step and attempt limits must be positive") + if not 0 < args.port < 65536 or args.startup_timeout <= 0: + raise ValueError("invalid port or startup timeout") + if len(set(args.task)) != len(args.task): + raise ValueError("duplicate tasks") + validate_server_options(args) + + +def resolve_tasks(args) -> None: + args.task_root = args.task_root.expanduser().resolve() + if args.task and args.suite: + raise ValueError("select either --task or --suite") + if not args.task: + args.suite = args.suite or "smoke" + args.task = [args.task_root / name for name in SUITES["suites"][args.suite]] + args.task = [path.resolve() for path in args.task] + for task in args.task: + if ( + not (task / "task.toml").is_file() + or not (task / "instruction.md").is_file() + ): + raise ValueError( + f"not a Harbor task directory: {task}; run evals/setup.sh terminal-bench or supply --task" + ) + + +def validate_server_options(args) -> None: + if args.agent_base_url is None: + agent_host = os.environ.get( + "EXECUTORCH_EVAL_AGENT_HOST", "host.docker.internal" + ) + args.agent_base_url = f"http://{agent_host}:{args.port}/v1" + url = urllib.parse.urlsplit(args.agent_base_url) + if ( + url.scheme != "http" + or not url.hostname + or url.path.rstrip("/") != "/v1" + or url.query + or url.fragment + or url.username + ): + raise ValueError("--agent-base-url must be an HTTP URL ending in /v1") + if (url.port or 80) != args.port: + raise ValueError("agent URL port must match the managed server port") + controlled = { + "worker_bin", + "model_path", + "tokenizer_path", + "hf_tokenizer", + "model_id", + "host", + "port", + "max_context", + "no_think", + "num_runners", + } + for flag in args.server_arg: + name = flag.split("=", 1)[0].lstrip("-").replace("-", "_") + if name in controlled or not flag.startswith("--") or "=" not in flag: + raise ValueError( + f"server argument is controlled by the benchmark or malformed: {flag}" + ) + + +def source_revision(root: Path) -> dict: + try: + commit = subprocess.check_output( + ["git", "-C", str(root), "rev-parse", "HEAD"], + text=True, + stderr=subprocess.DEVNULL, + ).strip() + diff = subprocess.check_output(["git", "-C", str(root), "diff", "HEAD", "--"]) + return { + "commit": commit, + "tracked_diff_sha256": hashlib.sha256(diff).hexdigest(), + } + except (OSError, subprocess.CalledProcessError): + return {"commit": None} + + +def task_digest(task: Path) -> str: + digest = hashlib.sha256() + for path in sorted(task.rglob("*")): + if path.is_file(): + digest.update(str(path.relative_to(task)).encode() + b"\0") + digest.update(path.read_bytes()) + return digest.hexdigest() + + +def file_digest(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def make_manifest(args) -> dict: + settings = { + k: v for k, v in vars(args).items() if k not in {"dry_run", "check", "resume"} + } + settings = json.loads(json.dumps(settings, default=str)) + assets = {} + for key, path in [ + ("worker_bin", args.worker_bin), + ("model_path", args.model_path), + ("tokenizer_path", args.tokenizer_path), + ]: + assets[key] = ( + { + "path": str(path), + "size": path.stat().st_size, + "mtime_ns": path.stat().st_mtime_ns, + "sha256": file_digest(path), + } + if path.is_file() + else {"path": str(path), "missing": True} + ) + setup_record = HARNESS_CACHE / "environment.json" + uses_setup_environment = ( + Path(args.harbor_bin).resolve() == (HARNESS_CACHE / "venv/bin/harbor").resolve() + ) + return { + "schema_version": 1, + "settings": settings, + "harbor_version": HARBOR_VERSION, + "agent_version": AGENT_VERSION, + "suites_sha256": file_digest(SUITES_PATH), + "task_source": ( + SUITES["dataset"] if args.task_root == TASK_ROOT and args.suite else None + ), + "setup_environment": ( + json.loads(setup_record.read_text()) + if uses_setup_environment and setup_record.is_file() + else None + ), + "executorch": source_revision(args.executorch_root), + "benchmark": source_revision(Path(__file__).parent), + "driver_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(), + "assets": assets, + "tasks": [ + {"path": str(task), "sha256": task_digest(task)} for task in args.task + ], + } + + +def trials(args) -> list[Trial]: + result = [] + run_tag = hashlib.sha256(str(args.run_dir).encode()).hexdigest()[:10] + for index, task in enumerate(args.task): + name = re.sub(r"[^a-zA-Z0-9_-]", "-", task.name)[:48] + for attempt in range(1, args.attempts + 1): + key = f"{index:03d}-{name}-a{attempt}" + result.append(Trial(key, str(task), attempt, f"tb-{run_tag}-{key}")) + return result + + +def agent_config(args, trial: Trial) -> dict: + return { + "agent": {"step_limit": args.step_limit}, + "model": { + "model_kwargs": { + "temperature": 0, + "extra_headers": ( + {"x-session-affinity": trial.session_id} + if args.session_affinity + else {} + ), + } + }, + } + + +def server_environment(args) -> dict[str, str]: + env = os.environ.copy() + source = args.executorch_root / "src" + package_parent = ( + source if (source / "executorch").is_dir() else args.executorch_root.parent + ) + env["PYTHONPATH"] = str(package_parent) + ( + os.pathsep + env["PYTHONPATH"] if env.get("PYTHONPATH") else "" + ) + env["PYTHONUNBUFFERED"] = "1" + return env + + +def server_command(args, trial: Trial) -> list[str]: + command = [ + args.server_python, + "-m", + args.server_module, + "--worker-bin", + str(args.worker_bin), + "--model-path", + str(args.model_path), + "--tokenizer-path", + str(args.tokenizer_path), + "--hf-tokenizer", + args.hf_tokenizer, + "--model-id", + args.model_id, + "--host", + args.host, + "--port", + str(args.port), + "--max-context", + str(args.max_context), + ] + if not args.thinking: + command.append("--no-think") + command.extend(args.server_arg) + return command + + +def harbor_command(args, trial: Trial, execution: Path) -> list[str]: + no_proxy = ",".join( + [ + "localhost", + "127.0.0.1", + "host.docker.internal", + "host.lima.internal", + "172.17.0.1", + urllib.parse.urlsplit(args.agent_base_url).hostname or "", + ] + ) + command = [ + args.harbor_bin, + "run", + "--path", + trial.task, + "--agent", + "mini-swe-agent", + "--model", + f"openai/{args.model_id}", + "--ak", + f"version={AGENT_VERSION}", + "--ak", + f"max_tokens={args.max_output_tokens}", + "--ak", + f"config_file={execution / 'mini-swe.yaml'}", + "--ae", + "MSWEA_API_KEY=dummy-local-key", + "--ae", + "OPENAI_API_KEY=dummy-local-key", + "--ae", + f"OPENAI_BASE_URL={args.agent_base_url.rstrip('/')}", + "--ae", + f"NO_PROXY={no_proxy}", + "--ae", + f"no_proxy={no_proxy}", + "--n-attempts", + "1", + "--n-concurrent", + "1", + "--max-retries", + "0", + "--jobs-dir", + str(execution / "jobs"), + "--job-name", + "trial", + ] + if needs_host_mapping(args): + command.extend(["--extra-docker-compose", str(execution / "host-gateway.yaml")]) + return command + + +def needs_host_mapping(args) -> bool: + return ( + sys.platform == "linux" + and urllib.parse.urlsplit(args.agent_base_url).hostname + == "host.docker.internal" + ) + + +def check_container_connection(args, execution: Path) -> None: + name = f"executorch-eval-probe-{uuid.uuid4().hex}" + command = ["docker", "run", "--rm", "--name", name] + if needs_host_mapping(args): + command.extend(["--add-host", "host.docker.internal:host-gateway"]) + command.extend( + [ + PROBE_IMAGE, + "--noproxy", + "*", + "--fail", + "--silent", + "--show-error", + "--max-time", + "15", + args.agent_base_url.rstrip("/").removesuffix("/v1") + "/health", + ] + ) + json_write(execution / "container-check-command.json", command) + try: + with (execution / "container-check.log").open("w") as log: + result = subprocess.run( + command, stdout=subprocess.PIPE, stderr=log, text=True, timeout=120 + ) + log.write(result.stdout) + if result.returncode or json.loads(result.stdout).get("status") != "ok": + raise RuntimeError( + f"Task-container route to the server failed; see {execution / 'container-check.log'}" + ) + finally: + subprocess.run( + ["docker", "rm", "--force", name], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + timeout=20, + ) + + +def json_write(path: Path, value) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + temp = path.with_suffix(path.suffix + ".tmp") + temp.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") + temp.replace(path) + + +def check_prerequisites(args) -> list[str]: + errors = [] + for label, command in [ + ("Harbor", args.harbor_bin), + ("server Python", args.server_python), + ("Docker", "docker"), + ]: + if shutil.which(command) is None: + errors.append(f"{label} executable not found: {command}") + for path in [args.worker_bin, args.model_path, args.tokenizer_path]: + if not path.is_file(): + errors.append(f"missing runtime asset: {path}") + if args.worker_bin.is_file() and not os.access(args.worker_bin, os.X_OK): + errors.append(f"worker is not executable: {args.worker_bin}") + checks = [] + if shutil.which(args.harbor_bin): + checks.append( + ("Harbor version", [args.harbor_bin, "--version"], None, HARBOR_VERSION) + ) + if shutil.which("docker"): + checks.append(("Docker Compose", ["docker", "compose", "version"], None, None)) + checks.append( + ( + "Docker daemon", + ["docker", "info", "--format", "{{.ServerVersion}}"], + None, + None, + ) + ) + if shutil.which(args.server_python): + checks.append( + ( + "ExecuTorch server", + [args.server_python, "-m", args.server_module, "--help"], + server_environment(args), + "--max-context", + ) + ) + return errors + run_prerequisite_checks(checks) + + +def run_prerequisite_checks(checks) -> list[str]: + errors = [] + for label, command, env, expected in checks: + try: + completed = subprocess.run( + command, + env=env, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + timeout=30, + ) + matches = ( + completed.stdout.strip() == expected + if label == "Harbor version" + else expected is None or expected in completed.stdout + ) + if completed.returncode or not matches: + errors.append(f"{label} check failed: {completed.stdout[-2000:]}") + except (OSError, subprocess.TimeoutExpired) as error: + errors.append(f"{label} check failed: {error}") + return errors + + +def local_url(args) -> str: + host = "127.0.0.1" if args.host == "0.0.0.0" else args.host + return f"http://{host}:{args.port}" + + +def request_json(url: str, method: str = "GET") -> dict: + with HTTP.open(urllib.request.Request(url, method=method), timeout=5) as response: + return json.load(response) + + +def wait_ready(args, process: subprocess.Popen) -> float: + started = time.monotonic() + last_error = "no health response" + while time.monotonic() - started < args.startup_timeout: + if process.poll() is not None: + raise RuntimeError( + f"server exited with code {process.returncode}; see server.log" + ) + try: + if request_json(local_url(args) + "/health").get("status") == "ok": + return time.monotonic() - started + except (OSError, ValueError) as error: + last_error = str(error) + time.sleep(0.25) + raise TimeoutError(f"server readiness timed out: {last_error}") + + +def stop_process(process: subprocess.Popen | None, stop_signal=signal.SIGTERM) -> None: + if process is None: + return + # The server can exit before its worker, so also terminate its process group. + try: + os.killpg(process.pid, stop_signal) + except ProcessLookupError: + pass + try: + process.wait(timeout=20) + except subprocess.TimeoutExpired: + try: + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + process.wait(timeout=10) + + +def parse_service_log( + path: Path, session_id: str, max_context: int, output_reserve: int +) -> dict: + turns = [] + integer_fields = { + "prompt_tokens", + "reused_prompt_tokens", + "prefilled_prompt_tokens", + "completion_tokens", + } + for line in path.read_text(errors="replace").splitlines(): + marker = "llm_turn_stats " + if marker not in line: + continue + row = dict( + part.split("=", 1) + for part in line.split(marker, 1)[1].split() + if "=" in part + ) + if row.get("session_id") != session_id: + continue + for key in integer_fields & row.keys(): + row[key] = int(row[key]) + for key in ("prefill_ms", "decode_ms", "total_ms"): + if key in row: + row[key] = float(row[key]) + turns.append(row) + input_tokens = sum(t["prompt_tokens"] for t in turns) + reused = sum(t["reused_prompt_tokens"] for t in turns) + return { + "turns": turns, + "completed_turns": len(turns), + "peak_prompt_tokens": max((t["prompt_tokens"] for t in turns), default=0), + "input_tokens": input_tokens, + "reused_tokens": reused, + "prefilled_tokens": sum(t["prefilled_prompt_tokens"] for t in turns), + "completion_tokens": sum(t["completion_tokens"] for t in turns), + "prefix_cache_hits": sum(t.get("reason") == "prefix_cache" for t in turns), + "continuation_hits": sum(t.get("reason") == "exact_prefix" for t in turns), + "reuse_fraction": reused / input_tokens if input_tokens else None, + "prefill_ms": sum(t.get("prefill_ms", 0) for t in turns), + "decode_ms": sum(t.get("decode_ms", 0) for t in turns), + "generation_ms": sum(t.get("total_ms", 0) for t in turns), + "context_violations": sum( + t["prompt_tokens"] + output_reserve > max_context for t in turns + ), + } + + +def read_harbor_result(execution: Path, returncode: int) -> dict: + files = sorted((execution / "jobs" / "trial").glob("*/result.json")) + if len(files) != 1: + return { + "status": "harness_error", + "error": f"expected one trial result, found {len(files)}", + "harbor_returncode": returncode, + } + data = json.loads(files[0].read_text()) + reward = ((data.get("verifier_result") or {}).get("rewards") or {}).get("reward") + exception = data.get("exception_info") or {} + result = { + "status": ( + "trial_error" + if exception or returncode + else ("scored" if reward is not None else "harness_error") + ), + "reward": reward, + "exception": exception.get("exception_type"), + "harbor_returncode": returncode, + "harbor_result": str(files[0]), + "agent_metrics": data.get("agent_result"), + } + trajectory = files[0].parent / "agent" / "mini-swe-agent.trajectory.json" + if trajectory.exists(): + data = json.loads(trajectory.read_text()) + info = data.get("info", {}) + messages = data.get("messages", []) + result["trajectory"] = str(trajectory) + result["tool_observations"] = sum(m.get("role") == "tool" for m in messages) + result["format_errors"] = sum( + m.get("extra", {}).get("interrupt_type") == "FormatError" for m in messages + ) + result["api_calls"] = info.get("model_stats", {}).get("api_calls") + result["agent_exit"] = info.get("exit_status") + return result + + +def run_trial(args, trial: Trial, execution: Path) -> dict: + execution.mkdir(parents=True) + json_write(execution / "mini-swe.yaml", agent_config(args, trial)) + if needs_host_mapping(args): + json_write( + execution / "host-gateway.yaml", + { + "services": { + "main": {"extra_hosts": ["host.docker.internal:host-gateway"]} + } + }, + ) + commands = { + "server": server_command(args, trial), + "harbor": harbor_command(args, trial, execution), + } + json_write(execution / "commands.json", commands) + result = { + "status": "harness_error", + "trial": asdict(trial), + "execution": str(execution), + } + server = harbor = None + ready = False + started = time.monotonic() + with (execution / "server.log").open("w") as server_log, ( + execution / "harbor.log" + ).open("w") as harbor_log: + try: + with socket.socket() as probe: + probe.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + probe.bind((args.host, args.port)) + server = subprocess.Popen( + commands["server"], + env=server_environment(args), + stdin=subprocess.DEVNULL, + stdout=server_log, + stderr=subprocess.STDOUT, + start_new_session=True, + ) + result["startup_seconds"] = wait_ready(args, server) + ready = True + check_container_connection(args, execution) + if args.session_affinity: + request_json( + local_url(args) + f"/v1/sessions/{trial.session_id}/reset", "POST" + ) + trial_started = time.monotonic() + harbor = subprocess.Popen( + commands["harbor"], + stdin=subprocess.DEVNULL, + stdout=harbor_log, + stderr=subprocess.STDOUT, + start_new_session=True, + ) + returncode = harbor.wait() + result["trial_seconds"] = time.monotonic() - trial_started + result.update(read_harbor_result(execution, returncode)) + except KeyboardInterrupt: + result["status"] = "interrupted" + raise + except Exception as error: + result["error"] = f"{type(error).__name__}: {error}" + finally: + stop_process(harbor, signal.SIGINT) + if ready and args.session_affinity: + try: + result["session_cleanup"] = request_json( + local_url(args) + f"/v1/sessions/{trial.session_id}", "DELETE" + ) + except (OSError, ValueError) as error: + result["cleanup_error"] = str(error) + stop_process(server) + server_log.flush() + result["wall_seconds"] = time.monotonic() - started + result["service"] = parse_service_log( + execution / "server.log", + trial.session_id if args.session_affinity else "", + args.max_context, + args.max_output_tokens, + ) + service = result["service"] + result["validation"] = { + "context_within_limit": ( + service["context_violations"] == 0 + if service["completed_turns"] + else None + ), + "observed_prefix_reuse": service["prefix_cache_hits"] > 0, + "cleanup_succeeded": ready + and (not args.session_affinity or "session_cleanup" in result), + } + if result["status"] == "scored" and ( + service["context_violations"] + or not service["completed_turns"] + or "cleanup_error" in result + ): + result["status"] = "harness_error" + result["error"] = ( + "missing turn metrics, context violation, or session cleanup failure; inspect validation and logs" + ) + json_write(execution / "outcome.json", result) + return result + + +def write_results(run_dir: Path, planned: list[Trial]) -> None: + results = [] + for trial in planned: + path = run_dir / "trials" / trial.key / "result.json" + if path.exists(): + results.append(json.loads(path.read_text())) + json_write(run_dir / "results.json", results) + fields = [ + "task", + "attempt", + "status", + "reward", + "api_calls", + "agent_exit", + "tool_observations", + "format_errors", + "trial_seconds", + "completed_turns", + "peak_prompt_tokens", + "reused_tokens", + "prefilled_tokens", + "prefix_cache_hits", + "continuation_hits", + "prefill_ms", + "context_violations", + "execution", + ] + with (run_dir / "results.tsv").open("w", newline="") as stream: + writer = csv.DictWriter(stream, fields, delimiter="\t", extrasaction="ignore") + writer.writeheader() + for result in results: + writer.writerow({**result, **result["trial"], **result["service"]}) + summary = [ + "# Terminal-Bench results", + "", + f"Completed {len(results)} of {len(planned)} planned trials. Each row is one task attempt; this subset is not a full Terminal-Bench score.", + "", + "| Task | Attempt | Status | Reward | Agent exit | Tool observations | Prefill (ms) | Decode (ms) |", + "| --- | ---: | --- | ---: | --- | ---: | ---: | ---: |", + ] + for result in results: + row = result["trial"] + service = result["service"] + cells = [ + Path(row["task"]).name, + row["attempt"], + result["status"], + result.get("reward"), + result.get("agent_exit"), + result.get("tool_observations"), + service["prefill_ms"], + service["decode_ms"], + ] + summary.append( + "| " + + " | ".join( + str(cell).replace("|", "\\|").replace("\n", " ") for cell in cells + ) + + " |" + ) + summary.extend( + [ + "", + "See results.json for token usage, timings, validation, and evidence paths.", + "", + ] + ) + (run_dir / "summary.md").write_text("\n".join(summary)) + + +def run_matrix(args, p: argparse.ArgumentParser) -> int: + manifest = make_manifest(args) + manifest_path = args.run_dir / "manifest.json" + if manifest_path.exists(): + if not args.resume: + p.error( + "run directory already has a manifest; use --resume or a new directory" + ) + if json.loads(manifest_path.read_text()) != manifest: + p.error( + "configuration, sources, or assets changed; use a new run directory" + ) + elif any(path.name != ".lock" for path in args.run_dir.iterdir()): + p.error("run directory is nonempty without a benchmark manifest") + else: + json_write(manifest_path, manifest) + planned = trials(args) + print(f"Results: {args.run_dir}", flush=True) + write_results(args.run_dir, planned) + json_write( + args.run_dir / "plan.json", + [ + { + "trial": asdict(trial), + "agent_config": agent_config(args, trial), + "server": server_command(args, trial), + "harbor": harbor_command( + args, trial, args.run_dir / "trials" / trial.key / "execution-1" + ), + } + for trial in planned + ], + ) + if args.dry_run: + print(f"Planned {len(planned)} trials: {args.run_dir / 'plan.json'}") + return 0 + errors = check_prerequisites(args) + if errors: + print("\n".join(errors), file=sys.stderr) + return 2 + try: + for trial in planned: + root = args.run_dir / "trials" / trial.key + if (root / "result.json").exists(): + continue + attempt = 1 + while (root / f"execution-{attempt}").exists(): + attempt += 1 + print(f"Running {trial.key}", flush=True) + result = run_trial(args, trial, root / f"execution-{attempt}") + json_write(root / "result.json", result) + write_results(args.run_dir, planned) + service = result["service"] + print( + f"{trial.key}: {result['status']}, reward={result.get('reward')}, " + f"agent_exit={result.get('agent_exit')}, tool_observations={result.get('tool_observations')}, " + f"peak_prompt={service['peak_prompt_tokens']} + reserve={args.max_output_tokens} " + f"<= context={args.max_context}\nEvidence: {result['execution']}", + flush=True, + ) + except KeyboardInterrupt: + write_results(args.run_dir, planned) + print( + "Interrupted; completed trials are preserved for --resume.", file=sys.stderr + ) + return 130 + return int( + any( + json.loads( + (args.run_dir / "trials" / trial.key / "result.json").read_text() + )["status"] + != "scored" + for trial in planned + ) + ) + + +def main(argv=None) -> int: + p = parser() + try: + args = p.parse_args(config_arguments(p, sys.argv[1:] if argv is None else argv)) + except (OSError, ValueError) as error: + p.error(str(error)) + try: + validate(args) + except ValueError as error: + p.error(str(error)) + if args.check: + errors = check_prerequisites(args) + print("\n".join(errors) if errors else "Prerequisites ready.") + return int(bool(errors)) + args.run_dir.mkdir(parents=True, exist_ok=True) + with (args.run_dir / ".lock").open("a") as lock: + try: + fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError: + p.error("another benchmark process owns this run directory") + return run_matrix(args, p) + + +if __name__ == "__main__": + signal.signal(signal.SIGTERM, signal.default_int_handler) + raise SystemExit(main()) diff --git a/examples/llm_server/evals/terminal_bench/setup.sh b/examples/llm_server/evals/terminal_bench/setup.sh new file mode 100644 index 00000000000..9f08f7acdbb --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/setup.sh @@ -0,0 +1,91 @@ +#!/usr/bin/env bash +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. +set -euo pipefail + +evals_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd) +repo_dir=$(cd -- "${evals_dir}/../../.." && pwd) +export EXECUTORCH_EVAL_CACHE="${EXECUTORCH_EVAL_CACHE:-${XDG_CACHE_HOME:-${HOME}/.cache}/executorch-evals}" + +harness_dir="${evals_dir}/terminal_bench" +ci=false +if [[ "${1:-}" == --ci ]]; then ci=true; shift; fi +if [[ $# != 0 ]]; then echo "Unknown setup argument: $1" >&2; exit 2; fi + +if ! docker info >/dev/null 2>&1; then + if "${ci}"; then + echo "CI requires a running Docker daemon accessible by this user." >&2 + exit 2 + fi + case "$(uname -s)" in + Darwin) + if ! command -v brew >/dev/null; then + echo "Install Homebrew (https://brew.sh), or start an existing Docker Desktop installation, then rerun setup." >&2 + exit 2 + fi + brew install colima docker docker-compose + if [[ "$(uname -m)" == arm64 ]]; then + colima start --cpu 4 --memory 8 --vm-type vz --vz-rosetta + else + colima start --cpu 4 --memory 8 + fi + ;; + Linux) + # shellcheck disable=SC1091 + source /etc/os-release + if [[ "${ID}" != ubuntu ]]; then + echo "Automatic Docker installation supports Ubuntu. Install and start Docker for ${ID}, then rerun setup." >&2 + exit 2 + fi + privilege=() + if [[ ${EUID} != 0 ]]; then privilege=(sudo); fi + "${privilege[@]}" apt-get update + "${privilege[@]}" apt-get install -y ca-certificates curl + "${privilege[@]}" install -m 0755 -d /etc/apt/keyrings + "${privilege[@]}" curl -fsSL https://download.docker.com/linux/ubuntu/gpg -o /etc/apt/keyrings/docker.asc + "${privilege[@]}" chmod a+r /etc/apt/keyrings/docker.asc + docker_repo="deb [arch=$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/docker.asc] https://download.docker.com/linux/ubuntu ${VERSION_CODENAME} stable" + echo "${docker_repo}" | "${privilege[@]}" tee /etc/apt/sources.list.d/docker.list >/dev/null + "${privilege[@]}" apt-get update + "${privilege[@]}" apt-get install -y docker-ce docker-ce-cli containerd.io docker-buildx-plugin docker-compose-plugin + "${privilege[@]}" systemctl enable --now docker + if [[ ${EUID} != 0 ]]; then + "${privilege[@]}" usermod -aG docker "$(id -un)" + # Continue in the new group without requiring a logout/login cycle. + printf -v retry 'bash %q terminal-bench' "${evals_dir}/setup.sh" + exec sg docker -c "${retry}" + fi + ;; + *) echo "Install Docker on this platform, then rerun setup." >&2; exit 2 ;; + esac +fi +docker info >/dev/null +if ! docker compose version >/dev/null 2>&1 && ! "${ci}" && [[ "$(uname -s)" == Darwin ]]; then + brew install docker-compose + plugin_dir="${DOCKER_CONFIG:-${HOME}/.docker}/cli-plugins" + mkdir -p "${plugin_dir}" + ln -s "$(brew --prefix)/bin/docker-compose" "${plugin_dir}/docker-compose" +fi +docker compose version +command -v git >/dev/null + +export PATH="${EXECUTORCH_EVAL_CACHE}/bin:${PATH}" +if ! command -v uv >/dev/null; then + # Install uv into the evaluation cache without changing shell startup files. + installer=$(mktemp) + trap 'rm -f "${installer}"' EXIT + curl -LsSf https://astral.sh/uv/0.8.22/install.sh -o "${installer}" + UV_UNMANAGED_INSTALL="${EXECUTORCH_EVAL_CACHE}/bin" sh "${installer}" + export PATH="${EXECUTORCH_EVAL_CACHE}/bin:${PATH}" +fi +eval_env="${EXECUTORCH_EVAL_CACHE}/terminal-bench/venv" +if [[ ! -x "${eval_env}/bin/python" ]]; then + uv venv --python 3.12 "${eval_env}" +fi +uv pip install --python "${eval_env}/bin/python" -r "${harness_dir}/requirements.txt" +export PYTHONPATH="${repo_dir}/src${PYTHONPATH:+:${PYTHONPATH}}" +"${eval_env}/bin/python" -m executorch.examples.llm_server.evals.terminal_bench.prepare +echo "Setup complete. Run: bash ${evals_dir}/run.sh terminal-bench --config /path/to/model.toml --suite smoke" diff --git a/examples/llm_server/evals/terminal_bench/suite.py b/examples/llm_server/evals/terminal_bench/suite.py new file mode 100644 index 00000000000..8168ffcb2d8 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/suite.py @@ -0,0 +1,27 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +import json +import os +from pathlib import Path + +SUITES_PATH = Path(__file__).with_name("suites.json") +SUITES = json.loads(SUITES_PATH.read_text()) +CACHE_ROOT = ( + Path( + os.environ.get( + "EXECUTORCH_EVAL_CACHE", + str( + Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache")) + / "executorch-evals" + ), + ) + ) + .expanduser() + .resolve() +) +HARNESS_CACHE = CACHE_ROOT / "terminal-bench" +TASK_ROOT = HARNESS_CACHE / "tasks" / SUITES["dataset"]["revision"] diff --git a/examples/llm_server/evals/terminal_bench/suites.json b/examples/llm_server/evals/terminal_bench/suites.json new file mode 100644 index 00000000000..ef55479f783 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/suites.json @@ -0,0 +1,10 @@ +{ + "dataset": { + "repository": "https://huggingface.co/datasets/harborframework/terminal-bench-2.0", + "revision": "f2e8c75e23add71613117eecc9498f53bcd7e04e" + }, + "suites": { + "smoke": ["fix-git"], + "nightly": ["fix-git", "openssl-selfsigned-cert"] + } +} diff --git a/examples/llm_server/evals/terminal_bench/tests/fixtures/benchmark_process.py b/examples/llm_server/evals/terminal_bench/tests/fixtures/benchmark_process.py new file mode 100644 index 00000000000..409107af779 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/tests/fixtures/benchmark_process.py @@ -0,0 +1,146 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +"""Subprocess doubles for runner tests; these do not evaluate a benchmark task.""" + +import json +import os +import sys +from http.server import BaseHTTPRequestHandler, HTTPServer +from pathlib import Path +from urllib.request import Request, urlopen + + +def event(path, name): + with Path(path).open("a") as stream: + stream.write(name + "\n") + + +def fake_server(port, events): + class Handler(BaseHTTPRequestHandler): + def respond(self, payload): + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps(payload).encode()) + + def do_GET(self): + self.respond({"status": "ok"}) + + def do_POST(self): + if self.path.endswith("/reset"): + event(events, "reset") + self.respond({"reset": True}) + return + body = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + session = self.headers.get("x-session-affinity", "") + later = len(body["messages"]) > 2 + reused = later and session != "" + reason = "exact_prefix" if reused else "new" + print( + f"INFO llm_turn_stats session_id={session} reason={reason} prompt_tokens=100 reused_prompt_tokens={25 if reused else 0} prefilled_prompt_tokens={75 if reused else 100} completion_tokens=2 prefill_ms=10 decode_ms=5 total_ms=15 finish=stop", + flush=True, + ) + event(events, "chat") + self.respond( + {"choices": [{"message": {"role": "assistant", "content": "OK"}}]} + ) + + def do_DELETE(self): + event(events, "delete") + self.respond({"closed": True}) + + event(events, "start") + HTTPServer(("127.0.0.1", int(port)), Handler).serve_forever() + + +def python_server(argv): + from executorch.examples.llm_server.python import server + from executorch.examples.llm_server.python.chat_template import ChatTemplate + from executorch.examples.llm_server.python.tests.conftest import FakeRunner + + class ByteTemplate(ChatTemplate): + def __init__(self, *args, **kwargs): + super().__init__(allow_fallback=True) + + def count_tokens(self, text): + return len(text.encode()) + + server.ChatTemplate = ByteTemplate + server._spawn = lambda args: FakeRunner(["OK"], max_named_sessions=2) + sys.argv = ["server", *argv] + server.main() + + +def fake_harbor(argv): + options = dict(zip(argv[::2], argv[1::2])) + kwargs = dict( + value.split("=", 1) + for flag, value in zip(argv[::2], argv[1::2]) + if flag == "--ak" + ) + env = dict( + value.split("=", 1) + for flag, value in zip(argv[::2], argv[1::2]) + if flag == "--ae" + ) + config = json.loads(Path(kwargs["config_file"]).read_text()) + headers = { + "Content-Type": "application/json", + **config["model"]["model_kwargs"]["extra_headers"], + } + history = [ + {"role": "system", "content": "Help with the task."}, + {"role": "user", "content": "Inspect a.py."}, + ] + for turn in range(2): + request = Request( + env["OPENAI_BASE_URL"] + "/chat/completions", + headers=headers, + data=json.dumps( + { + "model": options["--model"].removeprefix("openai/"), + "messages": history, + "max_tokens": 4, + "temperature": 0, + } + ).encode(), + ) + with urlopen(request, timeout=15) as response: + history.append(json.load(response)["choices"][0]["message"]) + if turn == 0: + history.extend( + [ + {"role": "user", "content": "Command finished successfully."}, + {"role": "assistant", "content": "Now inspect b.py"}, + {"role": "user", "content": "done"}, + ] + ) + if os.environ.get("BENCH_TEST_FAIL"): + return 29 + task = Path(options["--jobs-dir"]) / options["--job-name"] / "fake-task" + (task / "agent").mkdir(parents=True) + (task / "result.json").write_text( + json.dumps( + {"verifier_result": {"rewards": {"reward": 1.0}}, "exception_info": None} + ) + ) + (task / "agent" / "mini-swe-agent.trajectory.json").write_text( + json.dumps( + {"info": {"model_stats": {"api_calls": 2}, "exit_status": "Submitted"}} + ) + ) + return 0 + + +if __name__ == "__main__": + mode, *args = sys.argv[1:] + if mode == "server": + fake_server(*args) + elif mode == "python-server": + python_server(args) + else: + raise SystemExit(fake_harbor(args)) diff --git a/examples/llm_server/evals/terminal_bench/tests/test_setup.py b/examples/llm_server/evals/terminal_bench/tests/test_setup.py new file mode 100644 index 00000000000..2f75b68a549 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/tests/test_setup.py @@ -0,0 +1,156 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest + +from executorch.examples.llm_server.evals.terminal_bench import prepare + +EVALS = Path(__file__).resolve().parents[2] + + +def test_task_cache_pins_revision_and_preserves_local_edits(tmp_path, monkeypatch): + origin = tmp_path / "origin" + origin.mkdir() + task = origin / "sample" + task.mkdir() + (task / "task.toml").write_text('version = "1.0"\n') + (task / "instruction.md").write_text("original task") + subprocess.run(["git", "init", "--quiet", str(origin)], check=True) + subprocess.run(["git", "-C", str(origin), "add", "."], check=True) + subprocess.run( + [ + "git", + "-C", + str(origin), + "-c", + "user.name=Test", + "-c", + "user.email=test@example.com", + "-c", + "core.hooksPath=/dev/null", + "-c", + "commit.gpgsign=false", + "commit", + "--quiet", + "-m", + "task", + ], + check=True, + ) + revision = subprocess.check_output( + ["git", "-C", str(origin), "rev-parse", "HEAD"], text=True + ).strip() + monkeypatch.setattr( + prepare, + "SUITES", + { + "dataset": {"repository": str(origin), "revision": revision}, + "suites": {"smoke": ["sample"]}, + }, + ) + destination = tmp_path / "cache/tasks" + prepare.prepare_tasks(destination) + prepare.prepare_tasks(destination) + assert (destination / "sample/instruction.md").read_text() == "original task" + (destination / "sample/instruction.md").write_text("developer edit") + with pytest.raises(RuntimeError, match="modified"): + prepare.prepare_tasks(destination) + assert (destination / "sample/instruction.md").read_text() == "developer edit" + + +@pytest.mark.parametrize( + "platform,context,fresh_group,agent_host", + [ + ("Darwin", "colima", False, "host.lima.internal"), + ("Darwin", "desktop-linux", False, None), + ("Linux", "default", False, None), + ("Linux", "default", True, None), + ], +) +def test_run_script_uses_checkout_and_preserves_arguments( + tmp_path, platform, context, fresh_group, agent_host +): + cache = tmp_path / "cache with spaces" + binary = cache / "terminal-bench/venv/bin/python" + binary.parent.mkdir(parents=True) + binary.write_text( + f"#!{sys.executable}\nimport json, os, sys\nprint(json.dumps({{'argv':sys.argv[1:], 'path':os.environ['PYTHONPATH'], 'server':os.environ['EXECUTORCH_EVAL_SERVER_PYTHON'], 'agent_host':os.getenv('EXECUTORCH_EVAL_AGENT_HOST'), 'sg':os.getenv('EVAL_TEST_SG')}}))\n" + ) + binary.chmod(0o755) + commands = { + "uname": f"print({platform!r})", + "docker": f"print({context!r}) if sys.argv[1:] == ['context', 'show'] else sys.exit(0 if not {fresh_group!r} or os.getenv('EVAL_TEST_SG') else 1)", + "id": "print('testuser' if sys.argv[1:] == ['-un'] else ('users docker' if len(sys.argv) == 3 or os.getenv('EVAL_TEST_SG') else 'users'))", + "sg": "assert sys.argv[1:3] == ['docker', '-c']; os.environ['EVAL_TEST_SG']='1'; os.execv('/bin/bash', ['bash', '-c', sys.argv[3]])", + } + for name, code in commands.items(): + executable = tmp_path / name + executable.write_text(f"#!{sys.executable}\nimport os, sys\n{code}\n") + executable.chmod(0o755) + env = { + **os.environ, + "PATH": str(tmp_path) + os.pathsep + os.environ["PATH"], + "EXECUTORCH_EVAL_CACHE": str(cache), + "EXECUTORCH_EVAL_SERVER_PYTHON": sys.executable, + } + env.pop("EXECUTORCH_EVAL_AGENT_HOST", None) + env.pop("EVAL_TEST_SG", None) + result = subprocess.run( + [ + "bash", + str(EVALS / "run.sh"), + "terminal-bench", + "--config", + "model with spaces.toml", + "--server-arg=--assistant-header=custom", + ], + env=env, + capture_output=True, + text=True, + check=True, + cwd=tmp_path, + ) + value = json.loads(result.stdout) + assert value["argv"] == [ + "-m", + "executorch.examples.llm_server.evals.terminal_bench", + "--config", + "model with spaces.toml", + "--server-arg=--assistant-header=custom", + ] + assert value["path"].split(os.pathsep)[0] == str(EVALS.parents[2] / "src") + assert value["server"] == sys.executable + assert value["agent_host"] == agent_host + assert bool(value["sg"]) == fresh_group + + +def test_ci_setup_does_not_try_to_install_docker(tmp_path): + docker = tmp_path / "docker" + docker.write_text("#!/bin/sh\nexit 1\n") + docker.chmod(0o755) + result = subprocess.run( + ["bash", str(EVALS / "setup.sh"), "terminal-bench", "--ci"], + env={**os.environ, "PATH": str(tmp_path) + os.pathsep + os.environ["PATH"]}, + capture_output=True, + text=True, + ) + assert result.returncode == 2 + assert "CI requires a running Docker" in result.stderr + + +@pytest.mark.parametrize("script", ["run.sh", "setup.sh"]) +def test_unknown_harness_is_rejected(script): + result = subprocess.run( + ["bash", str(EVALS / script), "not-a-harness"], capture_output=True, text=True + ) + assert result.returncode == 2 + assert "Unsupported evaluation" in result.stderr diff --git a/examples/llm_server/evals/terminal_bench/tests/test_terminal_bench.py b/examples/llm_server/evals/terminal_bench/tests/test_terminal_bench.py new file mode 100644 index 00000000000..9c126a9fb62 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/tests/test_terminal_bench.py @@ -0,0 +1,434 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +import json +import os +import socket +import subprocess +import sys +from pathlib import Path + +import pytest + +from executorch.examples.llm_server.evals.terminal_bench import runner as bench + +PROCESS = Path(__file__).parent / "fixtures" / "benchmark_process.py" +SERVER_COMMAND = bench.server_command + + +@pytest.fixture +def argv(tmp_path): + task = tmp_path / "fix-git" + task.mkdir() + (task / "instruction.md").write_text("Fix the repository.") + (task / "task.toml").write_text('version = "1.0"\n') + for name in ("worker", "model.pte", "tokenizer.json"): + (tmp_path / name).write_text("fixture") + with socket.socket() as port: + port.bind(("127.0.0.1", 0)) + number = port.getsockname()[1] + return [ + "--session-affinity", + "--attempts", + "3", + "--task", + str(task), + "--run-dir", + str(tmp_path / "run"), + "--executorch-root", + str(tmp_path / "executorch"), + "--worker-bin", + str(tmp_path / "worker"), + "--model-path", + str(tmp_path / "model.pte"), + "--tokenizer-path", + str(tmp_path / "tokenizer.json"), + "--hf-tokenizer", + "test", + "--max-context", + "1600", + "--max-output-tokens", + "4", + "--host", + "127.0.0.1", + "--port", + str(number), + "--agent-base-url", + f"http://127.0.0.1:{number}/v1", + ] + + +def arguments(argv): + args = bench.parser().parse_args(argv) + bench.validate(args) + return args + + +@pytest.fixture +def processes(monkeypatch, tmp_path): + events = tmp_path / "events" + actual_harbor = bench.harbor_command + monkeypatch.setattr(bench, "check_prerequisites", lambda args: []) + monkeypatch.setattr( + bench, "check_container_connection", lambda args, execution: None + ) + monkeypatch.setattr( + bench, + "server_command", + lambda args, trial: [ + sys.executable, + str(PROCESS), + "server", + str(args.port), + str(events), + ], + ) + monkeypatch.setattr( + bench, + "harbor_command", + lambda args, trial, execution: [ + sys.executable, + str(PROCESS), + "harbor", + *actual_harbor(args, trial, execution)[2:], + ], + ) + return events + + +def test_trials_and_context_controls(argv): + args = arguments(argv + ["--attempts", "2"]) + planned = bench.trials(args) + assert len(planned) == 2 + assert len({t.session_id for t in planned}) == 2 + for trial in planned: + command = bench.server_command(args, trial) + assert command[command.index("--max-context") + 1] == "1600" + assert "--no-think" in command + assert not any("worker-arg" in arg or "cliff" in arg for arg in command) + assert "max_tokens=4" in bench.harbor_command(args, trial, args.run_dir) + config = bench.agent_config(args, trial) + assert config["model"]["model_kwargs"]["temperature"] == 0 + assert ( + config["model"]["model_kwargs"]["extra_headers"]["x-session-affinity"] + == trial.session_id + ) + + +@pytest.mark.parametrize( + "extra", + [ + ["--max-output-tokens", "1600"], + ["--server-arg=--max-context=8192"], + ["--server-arg=--port=9000"], + ["--agent-base-url", "http://localhost:9000/v1"], + ["--step-limit", "0"], + ], +) +def test_reject_invalid_controls(argv, extra): + with pytest.raises(ValueError): + arguments(argv + extra) + + +def test_parse_metrics_separates_context_from_accumulated_input(tmp_path): + path = tmp_path / "server.log" + path.write_text( + "INFO llm_turn_stats session_id=s reason=prefix_cache prompt_tokens=100 reused_prompt_tokens=40 prefilled_prompt_tokens=60 completion_tokens=4 prefill_ms=1 decode_ms=2 total_ms=3\n" + "INFO llm_turn_stats session_id=s reason=exact_prefix prompt_tokens=150 reused_prompt_tokens=104 prefilled_prompt_tokens=46 completion_tokens=4 prefill_ms=2 decode_ms=3 total_ms=5\n" + "INFO llm_turn_stats session_id=other prompt_tokens=99999\n" + ) + stats = bench.parse_service_log(path, "s", 200, 20) + assert stats["input_tokens"] == 250 + assert stats["peak_prompt_tokens"] == 150 + assert stats["context_violations"] == 0 + assert stats["reused_tokens"] == 144 + assert stats["prefilled_tokens"] == 106 + assert stats["prefix_cache_hits"] == stats["continuation_hits"] == 1 + assert bench.parse_service_log(path, "s", 160, 20)["context_violations"] == 1 + + +def test_process_lifecycle_and_resume(argv, processes): + assert bench.main(argv) == 0 + args = arguments(argv) + rows = json.loads((args.run_dir / "results.json").read_text()) + assert len(rows) == 3 + assert ( + processes.read_text().splitlines() + == ["start", "reset", "chat", "chat", "delete"] * 3 + ) + assert all(row["reward"] == 1 and row["status"] == "scored" for row in rows) + assert all(row["validation"]["cleanup_succeeded"] for row in rows) + assert bench.main(argv + ["--resume"]) == 0 + assert len(processes.read_text().splitlines()) == 15 + with pytest.raises(SystemExit): + bench.main(argv + ["--resume", "--step-limit", "10"]) + + +def test_cleanup_when_harbor_fails(argv, processes, monkeypatch): + monkeypatch.setenv("BENCH_TEST_FAIL", "1") + assert bench.main(argv + ["--attempts", "1"]) == 1 + args = arguments(argv) + row = json.loads((args.run_dir / "results.json").read_text())[0] + assert row["status"] == "harness_error" + assert row["harbor_returncode"] == 29 + assert row["validation"]["cleanup_succeeded"] + assert processes.read_text().splitlines()[-1] == "delete" + + +def test_dry_run_does_not_launch_processes(argv, processes): + assert bench.main(argv + ["--dry-run"]) == 0 + assert not processes.exists() + args = arguments(argv) + assert len(json.loads((args.run_dir / "plan.json").read_text())) == 3 + assert bench.main(argv + ["--resume"]) == 0 + + +def test_startup_failure_is_recorded(argv, processes, monkeypatch): + monkeypatch.setattr( + bench, + "server_command", + lambda args, trial: [sys.executable, "-c", "raise SystemExit(7)"], + ) + assert bench.main(argv + ["--attempts", "1"]) == 1 + args = arguments(argv) + row = json.loads((args.run_dir / "results.json").read_text())[0] + assert "server exited with code 7" in row["error"] + assert row["validation"]["context_within_limit"] is None + + +def test_occupied_port_does_not_launch_or_contact_existing_service(argv, processes): + args = arguments(argv) + with socket.socket() as occupied: + occupied.bind((args.host, args.port)) + occupied.listen() + assert bench.main(argv + ["--attempts", "1"]) == 1 + assert not processes.exists() + + +def test_resume_rejects_changed_task(argv, processes): + assert bench.main(argv + ["--dry-run"]) == 0 + args = arguments(argv) + (args.task[0] / "instruction.md").write_text("Different task") + with pytest.raises(SystemExit): + bench.main(argv + ["--resume"]) + assert not processes.exists() + + +def test_interruption_preserves_evidence_and_can_resume(argv, processes, monkeypatch): + original = bench.read_harbor_result + + def interrupt(*args): + raise KeyboardInterrupt + + monkeypatch.setattr(bench, "read_harbor_result", interrupt) + argv = argv + ["--attempts", "1"] + assert bench.main(argv) == 130 + args = arguments(argv) + trial = bench.trials(args)[0] + root = args.run_dir / "trials" / trial.key + assert ( + json.loads((root / "execution-1" / "outcome.json").read_text())["status"] + == "interrupted" + ) + assert not (root / "result.json").exists() + assert processes.read_text().splitlines()[-1] == "delete" + monkeypatch.setattr(bench, "read_harbor_result", original) + assert bench.main(argv + ["--resume"]) == 0 + assert (root / "execution-2" / "outcome.json").exists() + + +@pytest.mark.parametrize("affinity", [False, True]) +def test_managed_python_server(argv, processes, monkeypatch, affinity): + # Use the real launcher's argv, HTTP endpoints, SessionRuntime, and logs. + monkeypatch.setattr( + bench, + "server_command", + lambda args, trial: [ + sys.executable, + str(PROCESS), + "python-server", + *SERVER_COMMAND(args, trial)[3:], + ], + ) + options = [arg for arg in argv if arg != "--session-affinity"] + ["--attempts", "1"] + if affinity: + options.append("--session-affinity") + assert bench.main(options) == 0 + args = arguments(options) + row = json.loads((args.run_dir / "results.json").read_text())[0] + assert row["validation"]["context_within_limit"] + assert row["validation"]["cleanup_succeeded"] + assert row["service"]["completed_turns"] == 2 + + +def test_config_paths_and_cli_override(argv, tmp_path): + path = tmp_path / "settings.toml" + path.write_text( + 'worker_bin = "worker"\n' + 'model_path = "model.pte"\n' + 'tokenizer_path = "tokenizer.json"\n' + 'hf_tokenizer = "./local-tokenizer"\n' + 'task = ["fix-git"]\n' + "max_context = 1600\n" + "attempts = 2\n" + 'server_arg = ["--assistant-header=custom"]\n' + "thinking = true\n" + "session_affinity = true\n" + 'agent_base_url = "http://host.docker.internal:8000/v1"\n' + ) + p = bench.parser() + args = p.parse_args( + bench.config_arguments(p, ["--config", str(path), "--attempts", "4"]) + ) + bench.validate(args) + assert args.worker_bin == tmp_path / "worker" + assert args.task == [tmp_path / "fix-git"] + assert args.hf_tokenizer == str(tmp_path / "local-tokenizer") + assert args.attempts == 4 + assert args.server_arg == ["--assistant-header=custom"] + assert args.thinking + assert args.session_affinity + + +@pytest.mark.parametrize( + "content", + ["typo = 2", "resume = true", 'thinking = "yes"', 'model_id = ["a", "b"]'], +) +def test_invalid_config_is_rejected(tmp_path, content): + path = tmp_path / "bad.toml" + path.write_text(content) + with pytest.raises(ValueError): + bench.config_arguments(bench.parser(), ["--config", str(path)]) + + +def test_fresh_directories_and_explicit_resume(argv, tmp_path): + index = argv.index("--run-dir") + argv = argv[:index] + argv[index + 2 :] + ["--output-root", str(tmp_path / "runs")] + first, second = arguments(argv), arguments(argv) + assert first.run_dir != second.run_dir + assert first.run_dir.parent == second.run_dir.parent == tmp_path / "runs" + with pytest.raises(ValueError, match="explicit --run-dir"): + arguments(argv + ["--resume"]) + + +def test_summary_records_format_failure_without_claiming_tool_execution(tmp_path): + task = tmp_path / "jobs/trial/task" + (task / "agent").mkdir(parents=True) + (task / "result.json").write_text( + json.dumps({"verifier_result": {"rewards": {"reward": 0}}}) + ) + (task / "agent/mini-swe-agent.trajectory.json").write_text( + json.dumps( + { + "info": { + "exit_status": "RepeatedFormatError", + "model_stats": {"api_calls": 1}, + }, + "messages": [ + { + "role": "user", + "extra": { + "interrupt_type": "FormatError", + "response": "bad call", + }, + } + ], + } + ) + ) + result = bench.read_harbor_result(tmp_path, 0) + assert result["status"] == "scored" and result["reward"] == 0 + assert result["agent_exit"] == "RepeatedFormatError" + assert result["format_errors"] == 1 and result["tool_observations"] == 0 + assert Path(result["trajectory"]).is_file() + + +def test_resume_rejects_asset_content_change_with_same_size_and_timestamp( + argv, processes +): + assert bench.main(argv + ["--dry-run"]) == 0 + args = arguments(argv) + stat = args.model_path.stat() + args.model_path.write_text("changed") + os.utime(args.model_path, ns=(stat.st_atime_ns, stat.st_mtime_ns)) + with pytest.raises(SystemExit): + bench.main(argv + ["--resume"]) + + +def test_suite_overrides_configured_tasks(argv, tmp_path): + config = tmp_path / "model.toml" + config.write_text('task = ["does-not-exist"]\n') + index = argv.index("--task") + options = ( + argv[:index] + + argv[index + 2 :] + + ["--config", str(config), "--suite", "smoke", "--task-root", str(tmp_path)] + ) + p = bench.parser() + args = p.parse_args(bench.config_arguments(p, options)) + bench.validate(args) + assert args.task == [tmp_path / "fix-git"] + assert args.suite == "smoke" + config.write_text('suite = "nightly"\n') + args = p.parse_args(bench.config_arguments(p, argv + ["--config", str(config)])) + bench.validate(args) + assert args.suite is None and args.task == [tmp_path / "fix-git"] + + +def test_linux_container_mapping_matches_harbor_and_probe(argv, tmp_path, monkeypatch): + monkeypatch.setattr(bench.sys, "platform", "linux") + args = arguments( + argv + + ["--agent-base-url", f"http://host.docker.internal:{arguments(argv).port}/v1"] + ) + command = bench.harbor_command(args, bench.trials(args)[0], tmp_path) + assert command[-2:] == [ + "--extra-docker-compose", + str(tmp_path / "host-gateway.yaml"), + ] + calls = [] + + def run(command, **kwargs): + calls.append(command) + return subprocess.CompletedProcess(command, 0, '{"status":"ok"}') + + monkeypatch.setattr(bench.subprocess, "run", run) + bench.check_container_connection(args, tmp_path) + assert "host.docker.internal:host-gateway" in calls[0] + assert calls[-1][:3] == ["docker", "rm", "--force"] + assert calls[-1][-1] == calls[0][calls[0].index("--name") + 1] + + +def test_failed_container_route_cleans_up_without_running_agent( + argv, processes, monkeypatch +): + def fail(args, execution): + raise RuntimeError("container route failed") + + monkeypatch.setattr(bench, "check_container_connection", fail) + assert bench.main(argv + ["--attempts", "1"]) == 1 + args = arguments(argv) + result = json.loads((args.run_dir / "results.json").read_text())[0] + assert result["status"] == "harness_error" + assert "container route failed" in result["error"] + assert result["validation"]["cleanup_succeeded"] + assert processes.read_text().splitlines() == ["start", "delete"] + assert "harness_error" in (args.run_dir / "summary.md").read_text() + + +def test_probe_timeout_removes_its_container(argv, tmp_path, monkeypatch): + calls = [] + + def run(command, **kwargs): + calls.append(command) + if command[1] == "run": + raise subprocess.TimeoutExpired(command, 120) + return subprocess.CompletedProcess(command, 0) + + monkeypatch.setattr(bench.subprocess, "run", run) + with pytest.raises(subprocess.TimeoutExpired): + bench.check_container_connection(arguments(argv), tmp_path) + assert calls[-1][:3] == ["docker", "rm", "--force"]