From a2fdd93fc0faa2d8e53035e93a4044a358b93c20 Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Fri, 2 Oct 2026 21:31:52 -0400 Subject: [PATCH 1/5] [INITIAL] Update [ghstack-poisoned] --- examples/llm_server/README.md | 7 + examples/llm_server/evals/README.md | 40 ++ examples/llm_server/evals/__init__.py | 5 + examples/llm_server/evals/configs/.gitignore | 1 + .../evals/configs/terminal-bench.example.toml | 16 + .../evals/terminal_bench/__init__.py | 5 + .../evals/terminal_bench/requirements.txt | 2 + .../llm_server/evals/terminal_bench/run.sh | 28 ++ .../llm_server/evals/terminal_bench/runner.py | 356 ++++++++++++++++++ .../llm_server/evals/terminal_bench/setup.sh | 88 +++++ 10 files changed, 548 insertions(+) create mode 100644 examples/llm_server/evals/README.md create mode 100644 examples/llm_server/evals/__init__.py create mode 100644 examples/llm_server/evals/configs/.gitignore create mode 100644 examples/llm_server/evals/configs/terminal-bench.example.toml create mode 100644 examples/llm_server/evals/terminal_bench/__init__.py create mode 100644 examples/llm_server/evals/terminal_bench/requirements.txt create mode 100644 examples/llm_server/evals/terminal_bench/run.sh create mode 100644 examples/llm_server/evals/terminal_bench/runner.py create mode 100644 examples/llm_server/evals/terminal_bench/setup.sh diff --git a/examples/llm_server/README.md b/examples/llm_server/README.md index 3d7f7e3d64d..c4a68bf9c4a 100644 --- a/examples/llm_server/README.md +++ b/examples/llm_server/README.md @@ -8,6 +8,7 @@ examples/llm_server/ spec/ # language-neutral OpenAI contract ExecuTorch targets conformance/ # one test suite every language server must pass python/ # Python server implementation (current) + evals/ # task accuracy and performance through the server # cpp/ # future: no-Python single-binary server ``` @@ -106,3 +107,9 @@ Reliability guidance: `tools` were included in the request. - If a request fails with `unsupported_parameter`, remove or disable that OpenAI knob in your pi/client config. + +## Evaluate with Terminal-Bench + +The [evaluation guide](evals/README.md) provides setup and run commands for +Terminal-Bench, configuration templates in `evals/configs/`, and task and +performance metrics on Linux and macOS. diff --git a/examples/llm_server/evals/README.md b/examples/llm_server/evals/README.md new file mode 100644 index 00000000000..b53aa158e52 --- /dev/null +++ b/examples/llm_server/evals/README.md @@ -0,0 +1,40 @@ +# LLM server evaluations + +Each harness lives in its own directory; model configurations live in `configs/`. + +## Terminal-Bench + +Prepare a worker, exported model, and [server environment](../python/README.md), then: + +```bash +bash examples/llm_server/evals/terminal_bench/setup.sh +cp examples/llm_server/evals/configs/terminal-bench.example.toml examples/llm_server/evals/configs/terminal-bench.local.toml +# Edit the model paths in the local TOML. +bash examples/llm_server/evals/terminal_bench/run.sh --config examples/llm_server/evals/configs/terminal-bench.local.toml +``` + +Setup installs Docker Engine on Ubuntu or Colima on macOS (requires Homebrew), +reusing Docker when available. It installs Harbor in a separate Python 3.12 +environment under `~/.cache/executorch-evals`. Harbor installs mini-SWE-agent in +the task container and downloads Terminal-Bench 2.0 tasks on the first run. +Set `EXECUTORCH_EVAL_CACHE` to change the cache location. + +`[server]` entries become server CLI flags. `python` selects the server environment; +optional `module` selects a model-specific launcher. Relative model paths resolve +from the TOML directory. `max_context` must fit the export; the server reserves +`max_output_tokens` within that limit. Local TOML files are ignored by Git. + +`[terminal_bench]` selects tasks, attempts, and agent limits. Use `--task NAME` +to override the task list (repeatable), `--output DIR` for a new results directory, +or `--dry-run` to inspect the commands. The default task is a smoke check, not a +full benchmark. Use a model capable of tool calling for meaningful task scores. + +One server runs for the evaluation, and Harbor runs tasks sequentially with +anonymous requests. The launcher handles Linux host-gateway routing and macOS +Docker Desktop/Colima routing; set `agent_host` under `[terminal_bench]` for a +custom route. The server and its worker stop when the run finishes or is interrupted. + +Results are saved under the evaluation cache: `harbor/` contains Harbor's scores, +agent trajectories, and verifier output; `metrics.json` contains aggregate token +counts and prefill/decode times. Server logs, commands, and configuration are also +saved. Infrastructure errors fail the run; a valid task reward of zero does not. diff --git a/examples/llm_server/evals/__init__.py b/examples/llm_server/evals/__init__.py new file mode 100644 index 00000000000..2e41cd717f6 --- /dev/null +++ b/examples/llm_server/evals/__init__.py @@ -0,0 +1,5 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. diff --git a/examples/llm_server/evals/configs/.gitignore b/examples/llm_server/evals/configs/.gitignore new file mode 100644 index 00000000000..a4ce64c7a7c --- /dev/null +++ b/examples/llm_server/evals/configs/.gitignore @@ -0,0 +1 @@ +*.local.toml diff --git a/examples/llm_server/evals/configs/terminal-bench.example.toml b/examples/llm_server/evals/configs/terminal-bench.example.toml new file mode 100644 index 00000000000..bd202b4ac2c --- /dev/null +++ b/examples/llm_server/evals/configs/terminal-bench.example.toml @@ -0,0 +1,16 @@ +# Copy to terminal-bench.local.toml and set your model paths. +[server] +python = "/path/to/executorch-env/bin/python" +worker_bin = "/path/to/model_worker" +model_path = "/path/to/model.pte" +tokenizer_path = "/path/to/tokenizer.json" +hf_tokenizer = "/path/to/pinned-hf-tokenizer" +model_id = "qwen3" +max_context = 8192 +no_think = true + +[terminal_bench] +tasks = ["fix-git"] +max_output_tokens = 1024 +step_limit = 100 +attempts = 1 diff --git a/examples/llm_server/evals/terminal_bench/__init__.py b/examples/llm_server/evals/terminal_bench/__init__.py new file mode 100644 index 00000000000..2e41cd717f6 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/__init__.py @@ -0,0 +1,5 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. diff --git a/examples/llm_server/evals/terminal_bench/requirements.txt b/examples/llm_server/evals/terminal_bench/requirements.txt new file mode 100644 index 00000000000..9f4ff817b8b --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/requirements.txt @@ -0,0 +1,2 @@ +# Harbor runs separately from the model server and requires Python 3.12+. +harbor==0.22.0 diff --git a/examples/llm_server/evals/terminal_bench/run.sh b/examples/llm_server/evals/terminal_bench/run.sh new file mode 100644 index 00000000000..01b38e330bc --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/run.sh @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. +set -euo pipefail + +evals_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd) +repo_dir=$(cd -- "${evals_dir}/../../.." && pwd) +export EXECUTORCH_EVAL_CACHE="${EXECUTORCH_EVAL_CACHE:-${XDG_CACHE_HOME:-${HOME}/.cache}/executorch-evals}" + +if [[ "$(uname -s)" == Linux ]] && ! docker info >/dev/null 2>&1 && + [[ " $(id -nG) " != *" docker "* ]] && + [[ " $(id -nG "$(id -un)") " == *" docker "* ]]; then + # setup.sh may have just added Docker access while the caller's shell still + # has its original supplementary groups. + printf -v retry '%q ' bash "${evals_dir}/terminal_bench/run.sh" "$@" + exec sg docker -c "${retry}" +fi + +eval_python="${EXECUTORCH_EVAL_CACHE}/terminal-bench/venv/bin/python" +if [[ ! -x "${eval_python}" ]]; then + echo "Run bash ${evals_dir}/terminal_bench/setup.sh first." >&2 + exit 2 +fi +export PYTHONPATH="${repo_dir}/src${PYTHONPATH:+:${PYTHONPATH}}" +exec "${eval_python}" -m "executorch.examples.llm_server.evals.terminal_bench.runner" "$@" diff --git a/examples/llm_server/evals/terminal_bench/runner.py b/examples/llm_server/evals/terminal_bench/runner.py new file mode 100644 index 00000000000..3804eee7c78 --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/runner.py @@ -0,0 +1,356 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +"""Start the LLM server, run Terminal-Bench through Harbor, and save the logs.""" + +import argparse +import json +import os +import signal +import socket +import subprocess +import sys +import time +import tomllib +import urllib.request +from datetime import datetime, timezone +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[4] +CACHE = Path( + os.environ.get("EXECUTORCH_EVAL_CACHE", Path.home() / ".cache/executorch-evals") +) +HTTP = urllib.request.build_opener(urllib.request.ProxyHandler({})) + + +def load_config(path): + config = tomllib.loads(path.read_text()) + server = config["server"] + server.setdefault("python", sys.executable) + server.setdefault("module", "executorch.examples.llm_server.python.server") + server.setdefault("host", "0.0.0.0") + server.setdefault("port", 8000) + for key in ("worker_bin", "model_path", "tokenizer_path", "hf_tokenizer", "python"): + value = server[key] + if key in {"worker_bin", "model_path", "tokenizer_path"} or value.startswith( + ("/", "~", ".") + ): + server[key] = str((path.parent / Path(value).expanduser()).resolve()) + defaults = { + "tasks": ["fix-git"], + "attempts": 1, + "step_limit": 100, + "max_output_tokens": 512, + "agent_host": None, + } + options = config.get("terminal_bench", {}) + if unknown := options.keys() - defaults.keys(): + raise ValueError(f"Unknown Terminal-Bench settings: {sorted(unknown)}") + config["terminal_bench"] = options = defaults | options + if not 0 < options["max_output_tokens"] < server["max_context"]: + raise ValueError( + "max_output_tokens must be positive and smaller than max_context" + ) + if min(options["attempts"], options["step_limit"]) < 1 or not options["tasks"]: + raise ValueError("Choose tasks and positive attempt/step limits") + return config + + +def commands(config, output): + server, options = config["server"], config["terminal_bench"] + serve = [server["python"], "-m", server["module"]] + for key, value in server.items(): + if key in {"python", "module"} or value is False: + continue + flag = "--" + key.replace("_", "-") + if value is True: + serve.append(flag) + else: + serve.extend( + f"{flag}={item}" + for item in (value if isinstance(value, list) else [value]) + ) + agent_host = options["agent_host"] + if not agent_host: + context = subprocess.check_output( + ["docker", "context", "show"], text=True + ).strip() + agent_host = ( + "host.lima.internal" + if sys.platform == "darwin" and context.startswith("colima") + else "host.docker.internal" + ) + url = f"http://{agent_host}:{server['port']}" + harbor = [ + str(CACHE / "terminal-bench/venv/bin/harbor"), + "run", + "--dataset", + "terminal-bench@2.0", + "--agent", + "mini-swe-agent", + "--model", + f"openai/{server['model_id']}", + "--ak", + "version=2.4.6", + "--ak", + f"max_tokens={options['max_output_tokens']}", + "--ak", + f"config_file={output / 'agent.yaml'}", + "--ae", + "MSWEA_API_KEY=local", + "--ae", + "OPENAI_API_KEY=local", + "--ae", + f"OPENAI_BASE_URL={url}/v1", + "--ae", + f"NO_PROXY={agent_host}", + "--ae", + f"no_proxy={agent_host}", + "--n-concurrent", + "1", + "--n-attempts", + str(options["attempts"]), + "--max-retries", + "0", + "--jobs-dir", + str(output), + "--job-name", + "harbor", + ] + for task in options["tasks"]: + harbor.extend(["--include-task-name", task]) + probe = ["docker", "run", "--rm", "--name", f"executorch-eval-{os.getpid()}"] + if sys.platform == "linux" and agent_host == "host.docker.internal": + (output / "host-gateway.yaml").write_text( + json.dumps( + { + "services": { + "main": {"extra_hosts": ["host.docker.internal:host-gateway"]} + } + } + ) + ) + harbor.extend(["--extra-docker-compose", str(output / "host-gateway.yaml")]) + probe.extend(["--add-host", "host.docker.internal:host-gateway"]) + probe.extend( + [ + "curlimages/curl:8.12.1", + "--noproxy", + "*", + "--fail", + "--max-time", + "15", + url + "/health", + ] + ) + return serve, harbor, probe + + +def wait_ready(process, host, port, timeout=180): + host = "127.0.0.1" if host == "0.0.0.0" else host + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if process.poll() is not None: + raise RuntimeError("Server exited; see server.log") + try: + with HTTP.open(f"http://{host}:{port}/health", timeout=2) as response: + if json.load(response).get("status") == "ok": + return + except (OSError, ValueError): + pass + time.sleep(0.25) + raise TimeoutError("Server did not become ready; see server.log") + + +def stop(process, stop_signal=signal.SIGTERM): + if process is None: + return + # The worker can outlive its Python server; terminate the whole process group. + try: + os.killpg(process.pid, stop_signal) + except ProcessLookupError: + pass + try: + process.wait(timeout=20) + except subprocess.TimeoutExpired: + try: + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + process.wait(timeout=10) + + +def metrics(path, context, reserve): + turns = [] + for line in path.read_text(errors="replace").splitlines(): + if "llm_turn_stats " in line: + turns.append( + dict( + item.split("=", 1) + for item in line.split("llm_turn_stats ", 1)[1].split() + if "=" in item + ) + ) + result = { + key: sum((float if key.endswith("_ms") else int)(turn[key]) for turn in turns) + for key in ( + "prompt_tokens", + "completion_tokens", + "prefilled_prompt_tokens", + "reused_prompt_tokens", + "prefill_ms", + "decode_ms", + ) + } + result["requests"] = len(turns) + result["peak_prompt_tokens"] = max( + (int(turn["prompt_tokens"]) for turn in turns), default=0 + ) + result["context_violations"] = sum( + int(turn["prompt_tokens"]) + reserve > context for turn in turns + ) + return result + + +def run(config, output, serve, harbor, probe): + server = config["server"] + process = job = None + env = {**os.environ, "PYTHONUNBUFFERED": "1"} + env["PYTHONPATH"] = str(REPO_ROOT / "src") + os.pathsep + env.get("PYTHONPATH", "") + with (output / "server.log").open("w") as server_log, (output / "harbor.log").open( + "w" + ) as harbor_log: + try: + with socket.socket() as port: + port.bind((server["host"], server["port"])) + process = subprocess.Popen( + serve, + env=env, + stdout=server_log, + stderr=subprocess.STDOUT, + start_new_session=True, + ) + wait_ready(process, server["host"], server["port"]) + try: + with (output / "connection.log").open("w") as log: + subprocess.run( + probe, + stdout=log, + stderr=subprocess.STDOUT, + check=True, + timeout=120, + ) + finally: + subprocess.run( + ["docker", "rm", "--force", probe[probe.index("--name") + 1]], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + timeout=20, + ) + job = subprocess.Popen( + harbor, + stdout=harbor_log, + stderr=subprocess.STDOUT, + start_new_session=True, + ) + if job.wait(): + raise RuntimeError("Harbor failed; see harbor.log") + result = json.loads((output / "harbor/result.json").read_text()) + stats = result["stats"] + if ( + not result["n_total_trials"] + or stats["n_completed_trials"] != result["n_total_trials"] + or stats["n_errored_trials"] + ): + raise RuntimeError( + "Harbor reported incomplete or errored trials; see harbor/result.json" + ) + print(json.dumps(stats, indent=2)) + finally: + stop(job, signal.SIGINT) + stop(process) + server_log.flush() + timing = metrics( + output / "server.log", + server["max_context"], + config["terminal_bench"]["max_output_tokens"], + ) + (output / "metrics.json").write_text(json.dumps(timing, indent=2) + "\n") + if not timing["requests"] or timing["context_violations"]: + raise RuntimeError( + "Missing server metrics or context limit exceeded; see metrics.json" + ) + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--config", type=Path, required=True) + parser.add_argument( + "--task", action="append", help="Task name or glob; repeat to select a subset" + ) + parser.add_argument( + "--output", + type=Path, + default=CACHE + / "terminal-bench/runs" + / datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%S%fZ"), + ) + parser.add_argument("--dry-run", action="store_true") + args = parser.parse_args(argv) + try: + config = load_config(args.config.expanduser().resolve()) + if args.task: + config["terminal_bench"]["tasks"] = args.task + output = args.output.expanduser().resolve() + output.mkdir(parents=True, exist_ok=False) + serve, harbor, probe = commands(config, output) + (output / "agent.yaml").write_text( + json.dumps( + { + "agent": {"step_limit": config["terminal_bench"]["step_limit"]}, + "model": {"model_kwargs": {"temperature": 0}}, + } + ) + ) + revision = subprocess.check_output( + ["git", "-C", str(REPO_ROOT), "rev-parse", "HEAD"], text=True + ).strip() + (output / "run.json").write_text( + json.dumps( + { + "config": config, + "revision": revision, + "server": serve, + "harbor": harbor, + "probe": probe, + }, + indent=2, + ) + + "\n" + ) + (output / "source.diff").write_bytes( + subprocess.check_output(["git", "-C", str(REPO_ROOT), "diff", "HEAD"]) + ) + print(f"Results: {output}", flush=True) + if not args.dry_run: + run(config, output, serve, harbor, probe) + return 0 + except KeyboardInterrupt: + return 130 + except ( + OSError, + ValueError, + KeyError, + RuntimeError, + subprocess.SubprocessError, + ) as error: + print(str(error), file=sys.stderr) + return 1 + + +if __name__ == "__main__": + signal.signal(signal.SIGTERM, signal.default_int_handler) + raise SystemExit(main()) diff --git a/examples/llm_server/evals/terminal_bench/setup.sh b/examples/llm_server/evals/terminal_bench/setup.sh new file mode 100644 index 00000000000..e3b6b986a6e --- /dev/null +++ b/examples/llm_server/evals/terminal_bench/setup.sh @@ -0,0 +1,88 @@ +#!/usr/bin/env bash +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. +set -euo pipefail + +evals_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd) +export EXECUTORCH_EVAL_CACHE="${EXECUTORCH_EVAL_CACHE:-${XDG_CACHE_HOME:-${HOME}/.cache}/executorch-evals}" + +harness_dir="${evals_dir}/terminal_bench" +ci=false +if [[ "${1:-}" == --ci ]]; then ci=true; shift; fi +if [[ $# != 0 ]]; then echo "Unknown setup argument: $1" >&2; exit 2; fi + +if ! docker info >/dev/null 2>&1; then + if "${ci}"; then + echo "CI requires a running Docker daemon accessible by this user." >&2 + exit 2 + fi + case "$(uname -s)" in + Darwin) + if ! command -v brew >/dev/null; then + echo "Install Homebrew (https://brew.sh), or start an existing Docker Desktop installation, then rerun setup." >&2 + exit 2 + fi + brew install colima docker docker-compose + if [[ "$(uname -m)" == arm64 ]]; then + colima start --cpu 4 --memory 8 --vm-type vz --vz-rosetta + else + colima start --cpu 4 --memory 8 + fi + ;; + Linux) + # shellcheck disable=SC1091 + source /etc/os-release + if [[ "${ID}" != ubuntu ]]; then + echo "Automatic Docker installation supports Ubuntu. Install and start Docker for ${ID}, then rerun setup." >&2 + exit 2 + fi + privilege=() + if [[ ${EUID} != 0 ]]; then privilege=(sudo); fi + "${privilege[@]}" apt-get update + "${privilege[@]}" apt-get install -y ca-certificates curl + "${privilege[@]}" install -m 0755 -d /etc/apt/keyrings + "${privilege[@]}" curl -fsSL https://download.docker.com/linux/ubuntu/gpg -o /etc/apt/keyrings/docker.asc + "${privilege[@]}" chmod a+r /etc/apt/keyrings/docker.asc + docker_repo="deb [arch=$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/docker.asc] https://download.docker.com/linux/ubuntu ${VERSION_CODENAME} stable" + echo "${docker_repo}" | "${privilege[@]}" tee /etc/apt/sources.list.d/docker.list >/dev/null + "${privilege[@]}" apt-get update + "${privilege[@]}" apt-get install -y docker-ce docker-ce-cli containerd.io docker-buildx-plugin docker-compose-plugin + "${privilege[@]}" systemctl enable --now docker + if [[ ${EUID} != 0 ]]; then + "${privilege[@]}" usermod -aG docker "$(id -un)" + # Continue in the new group without requiring a logout/login cycle. + printf -v retry 'bash %q' "${harness_dir}/setup.sh" + exec sg docker -c "${retry}" + fi + ;; + *) echo "Install Docker on this platform, then rerun setup." >&2; exit 2 ;; + esac +fi +docker info >/dev/null +if ! docker compose version >/dev/null 2>&1 && ! "${ci}" && [[ "$(uname -s)" == Darwin ]]; then + brew install docker-compose + plugin_dir="${DOCKER_CONFIG:-${HOME}/.docker}/cli-plugins" + mkdir -p "${plugin_dir}" + ln -s "$(brew --prefix)/bin/docker-compose" "${plugin_dir}/docker-compose" +fi +docker compose version +command -v git >/dev/null + +export PATH="${EXECUTORCH_EVAL_CACHE}/bin:${PATH}" +if ! command -v uv >/dev/null; then + # Install uv into the evaluation cache without changing shell startup files. + installer=$(mktemp) + trap 'rm -f "${installer}"' EXIT + curl -LsSf https://astral.sh/uv/0.8.22/install.sh -o "${installer}" + UV_UNMANAGED_INSTALL="${EXECUTORCH_EVAL_CACHE}/bin" sh "${installer}" + export PATH="${EXECUTORCH_EVAL_CACHE}/bin:${PATH}" +fi +eval_env="${EXECUTORCH_EVAL_CACHE}/terminal-bench/venv" +if [[ ! -x "${eval_env}/bin/python" ]]; then + uv venv --python 3.12 "${eval_env}" +fi +uv pip install --python "${eval_env}/bin/python" -r "${harness_dir}/requirements.txt" +echo "Ready. Run bash ${harness_dir}/run.sh --config /path/to/model.toml" From 435e90341d0f62fb1855f0cda1fdc6305ec67703 Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Fri, 2 Oct 2026 21:36:22 -0400 Subject: [PATCH 2/5] [UPDATE] Update [ghstack-poisoned] --- examples/llm_server/evals/README.md | 45 ++++++++++------------------- 1 file changed, 15 insertions(+), 30 deletions(-) diff --git a/examples/llm_server/evals/README.md b/examples/llm_server/evals/README.md index b53aa158e52..d71e8d6c1c8 100644 --- a/examples/llm_server/evals/README.md +++ b/examples/llm_server/evals/README.md @@ -1,40 +1,25 @@ # LLM server evaluations -Each harness lives in its own directory; model configurations live in `configs/`. - ## Terminal-Bench -Prepare a worker, exported model, and [server environment](../python/README.md), then: +Prepare a worker, exported model, and [server environment](../python/README.md). +From the repository root: ```bash -bash examples/llm_server/evals/terminal_bench/setup.sh -cp examples/llm_server/evals/configs/terminal-bench.example.toml examples/llm_server/evals/configs/terminal-bench.local.toml -# Edit the model paths in the local TOML. -bash examples/llm_server/evals/terminal_bench/run.sh --config examples/llm_server/evals/configs/terminal-bench.local.toml +cd examples/llm_server/evals +bash terminal_bench/setup.sh +cp configs/terminal-bench.example.toml configs/terminal-bench.local.toml +# Edit the model paths and context limit in the local TOML. +bash terminal_bench/run.sh --config configs/terminal-bench.local.toml ``` -Setup installs Docker Engine on Ubuntu or Colima on macOS (requires Homebrew), -reusing Docker when available. It installs Harbor in a separate Python 3.12 -environment under `~/.cache/executorch-evals`. Harbor installs mini-SWE-agent in -the task container and downloads Terminal-Bench 2.0 tasks on the first run. -Set `EXECUTORCH_EVAL_CACHE` to change the cache location. - -`[server]` entries become server CLI flags. `python` selects the server environment; -optional `module` selects a model-specific launcher. Relative model paths resolve -from the TOML directory. `max_context` must fit the export; the server reserves -`max_output_tokens` within that limit. Local TOML files are ignored by Git. - -`[terminal_bench]` selects tasks, attempts, and agent limits. Use `--task NAME` -to override the task list (repeatable), `--output DIR` for a new results directory, -or `--dry-run` to inspect the commands. The default task is a smoke check, not a -full benchmark. Use a model capable of tool calling for meaningful task scores. +Setup installs Docker on Ubuntu or Colima on macOS (requires Homebrew), plus Harbor. +Harbor downloads Terminal-Bench 2.0 tasks and runs mini-SWE-agent in containers. -One server runs for the evaluation, and Harbor runs tasks sequentially with -anonymous requests. The launcher handles Linux host-gateway routing and macOS -Docker Desktop/Colima routing; set `agent_host` under `[terminal_bench]` for a -custom route. The server and its worker stop when the run finishes or is interrupted. +The TOML selects the model, tasks, attempts, and token budgets. `max_context` must +fit the exported model. The default `fix-git` task is a smoke check; use a model +capable of tool calling for meaningful scores. -Results are saved under the evaluation cache: `harbor/` contains Harbor's scores, -agent trajectories, and verifier output; `metrics.json` contains aggregate token -counts and prefill/decode times. Server logs, commands, and configuration are also -saved. Infrastructure errors fail the run; a valid task reward of zero does not. +Scores, trajectories, logs, and token/timing metrics are saved under +`~/.cache/executorch-evals/terminal-bench/runs/`. Use `--output DIR` to choose a +results directory or `--dry-run` to inspect commands before running. From ae85875f9557a7964902dc9752b00039262c019f Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Fri, 2 Oct 2026 21:38:19 -0400 Subject: [PATCH 3/5] [UPDATE] Update [ghstack-poisoned] --- examples/llm_server/README.md | 2 +- examples/llm_server/evals/README.md | 4 +- .../llm_server/evals/terminal_bench/run.sh | 11 +--- .../llm_server/evals/terminal_bench/runner.py | 44 +++++-------- .../llm_server/evals/terminal_bench/setup.sh | 65 +++++-------------- 5 files changed, 38 insertions(+), 88 deletions(-) diff --git a/examples/llm_server/README.md b/examples/llm_server/README.md index c4a68bf9c4a..ea16d59d4e2 100644 --- a/examples/llm_server/README.md +++ b/examples/llm_server/README.md @@ -112,4 +112,4 @@ Reliability guidance: The [evaluation guide](evals/README.md) provides setup and run commands for Terminal-Bench, configuration templates in `evals/configs/`, and task and -performance metrics on Linux and macOS. +performance metrics on macOS. diff --git a/examples/llm_server/evals/README.md b/examples/llm_server/evals/README.md index d71e8d6c1c8..87ad736c77e 100644 --- a/examples/llm_server/evals/README.md +++ b/examples/llm_server/evals/README.md @@ -1,6 +1,6 @@ # LLM server evaluations -## Terminal-Bench +## Terminal-Bench (macOS) Prepare a worker, exported model, and [server environment](../python/README.md). From the repository root: @@ -13,7 +13,7 @@ cp configs/terminal-bench.example.toml configs/terminal-bench.local.toml bash terminal_bench/run.sh --config configs/terminal-bench.local.toml ``` -Setup installs Docker on Ubuntu or Colima on macOS (requires Homebrew), plus Harbor. +Setup installs Colima (requires Homebrew) and Harbor, reusing Docker if available. Harbor downloads Terminal-Bench 2.0 tasks and runs mini-SWE-agent in containers. The TOML selects the model, tasks, attempts, and token budgets. `max_context` must diff --git a/examples/llm_server/evals/terminal_bench/run.sh b/examples/llm_server/evals/terminal_bench/run.sh index 01b38e330bc..98659de7b38 100644 --- a/examples/llm_server/evals/terminal_bench/run.sh +++ b/examples/llm_server/evals/terminal_bench/run.sh @@ -8,16 +8,7 @@ set -euo pipefail evals_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd) repo_dir=$(cd -- "${evals_dir}/../../.." && pwd) -export EXECUTORCH_EVAL_CACHE="${EXECUTORCH_EVAL_CACHE:-${XDG_CACHE_HOME:-${HOME}/.cache}/executorch-evals}" - -if [[ "$(uname -s)" == Linux ]] && ! docker info >/dev/null 2>&1 && - [[ " $(id -nG) " != *" docker "* ]] && - [[ " $(id -nG "$(id -un)") " == *" docker "* ]]; then - # setup.sh may have just added Docker access while the caller's shell still - # has its original supplementary groups. - printf -v retry '%q ' bash "${evals_dir}/terminal_bench/run.sh" "$@" - exec sg docker -c "${retry}" -fi +export EXECUTORCH_EVAL_CACHE="${EXECUTORCH_EVAL_CACHE:-${HOME}/.cache/executorch-evals}" eval_python="${EXECUTORCH_EVAL_CACHE}/terminal-bench/venv/bin/python" if [[ ! -x "${eval_python}" ]]; then diff --git a/examples/llm_server/evals/terminal_bench/runner.py b/examples/llm_server/evals/terminal_bench/runner.py index 3804eee7c78..7a56a322b4e 100644 --- a/examples/llm_server/evals/terminal_bench/runner.py +++ b/examples/llm_server/evals/terminal_bench/runner.py @@ -4,7 +4,7 @@ # This source code is licensed under the BSD-style license found in the # LICENSE file in the root directory of this source tree. -"""Start the LLM server, run Terminal-Bench through Harbor, and save the logs.""" +"""Run Terminal-Bench through Harbor against a local LLM server on macOS.""" import argparse import json @@ -80,7 +80,7 @@ def commands(config, output): ).strip() agent_host = ( "host.lima.internal" - if sys.platform == "darwin" and context.startswith("colima") + if context.startswith("colima") else "host.docker.internal" ) url = f"http://{agent_host}:{server['port']}" @@ -122,30 +122,20 @@ def commands(config, output): ] for task in options["tasks"]: harbor.extend(["--include-task-name", task]) - probe = ["docker", "run", "--rm", "--name", f"executorch-eval-{os.getpid()}"] - if sys.platform == "linux" and agent_host == "host.docker.internal": - (output / "host-gateway.yaml").write_text( - json.dumps( - { - "services": { - "main": {"extra_hosts": ["host.docker.internal:host-gateway"]} - } - } - ) - ) - harbor.extend(["--extra-docker-compose", str(output / "host-gateway.yaml")]) - probe.extend(["--add-host", "host.docker.internal:host-gateway"]) - probe.extend( - [ - "curlimages/curl:8.12.1", - "--noproxy", - "*", - "--fail", - "--max-time", - "15", - url + "/health", - ] - ) + probe = [ + "docker", + "run", + "--rm", + "--name", + f"executorch-eval-{os.getpid()}", + "curlimages/curl:8.12.1", + "--noproxy", + "*", + "--fail", + "--max-time", + "15", + url + "/health", + ] return serve, harbor, probe @@ -300,6 +290,8 @@ def main(argv=None): ) parser.add_argument("--dry-run", action="store_true") args = parser.parse_args(argv) + if sys.platform != "darwin": + parser.error("Terminal-Bench evaluations currently support macOS only.") try: config = load_config(args.config.expanduser().resolve()) if args.task: diff --git a/examples/llm_server/evals/terminal_bench/setup.sh b/examples/llm_server/evals/terminal_bench/setup.sh index e3b6b986a6e..adf27321288 100644 --- a/examples/llm_server/evals/terminal_bench/setup.sh +++ b/examples/llm_server/evals/terminal_bench/setup.sh @@ -6,63 +6,30 @@ # LICENSE file in the root directory of this source tree. set -euo pipefail +if [[ "$(uname -s)" != Darwin ]]; then + echo "Terminal-Bench evaluations currently support macOS only." >&2 + exit 2 +fi +if [[ $# != 0 ]]; then echo "Unknown setup argument: $1" >&2; exit 2; fi + evals_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd) -export EXECUTORCH_EVAL_CACHE="${EXECUTORCH_EVAL_CACHE:-${XDG_CACHE_HOME:-${HOME}/.cache}/executorch-evals}" +export EXECUTORCH_EVAL_CACHE="${EXECUTORCH_EVAL_CACHE:-${HOME}/.cache/executorch-evals}" harness_dir="${evals_dir}/terminal_bench" -ci=false -if [[ "${1:-}" == --ci ]]; then ci=true; shift; fi -if [[ $# != 0 ]]; then echo "Unknown setup argument: $1" >&2; exit 2; fi - if ! docker info >/dev/null 2>&1; then - if "${ci}"; then - echo "CI requires a running Docker daemon accessible by this user." >&2 + if ! command -v brew >/dev/null; then + echo "Install Homebrew (https://brew.sh), or start an existing Docker Desktop installation, then rerun setup." >&2 exit 2 fi - case "$(uname -s)" in - Darwin) - if ! command -v brew >/dev/null; then - echo "Install Homebrew (https://brew.sh), or start an existing Docker Desktop installation, then rerun setup." >&2 - exit 2 - fi - brew install colima docker docker-compose - if [[ "$(uname -m)" == arm64 ]]; then - colima start --cpu 4 --memory 8 --vm-type vz --vz-rosetta - else - colima start --cpu 4 --memory 8 - fi - ;; - Linux) - # shellcheck disable=SC1091 - source /etc/os-release - if [[ "${ID}" != ubuntu ]]; then - echo "Automatic Docker installation supports Ubuntu. Install and start Docker for ${ID}, then rerun setup." >&2 - exit 2 - fi - privilege=() - if [[ ${EUID} != 0 ]]; then privilege=(sudo); fi - "${privilege[@]}" apt-get update - "${privilege[@]}" apt-get install -y ca-certificates curl - "${privilege[@]}" install -m 0755 -d /etc/apt/keyrings - "${privilege[@]}" curl -fsSL https://download.docker.com/linux/ubuntu/gpg -o /etc/apt/keyrings/docker.asc - "${privilege[@]}" chmod a+r /etc/apt/keyrings/docker.asc - docker_repo="deb [arch=$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/docker.asc] https://download.docker.com/linux/ubuntu ${VERSION_CODENAME} stable" - echo "${docker_repo}" | "${privilege[@]}" tee /etc/apt/sources.list.d/docker.list >/dev/null - "${privilege[@]}" apt-get update - "${privilege[@]}" apt-get install -y docker-ce docker-ce-cli containerd.io docker-buildx-plugin docker-compose-plugin - "${privilege[@]}" systemctl enable --now docker - if [[ ${EUID} != 0 ]]; then - "${privilege[@]}" usermod -aG docker "$(id -un)" - # Continue in the new group without requiring a logout/login cycle. - printf -v retry 'bash %q' "${harness_dir}/setup.sh" - exec sg docker -c "${retry}" - fi - ;; - *) echo "Install Docker on this platform, then rerun setup." >&2; exit 2 ;; - esac + brew install colima docker docker-compose + if [[ "$(uname -m)" == arm64 ]]; then + colima start --cpu 4 --memory 8 --vm-type vz --vz-rosetta + else + colima start --cpu 4 --memory 8 + fi fi docker info >/dev/null -if ! docker compose version >/dev/null 2>&1 && ! "${ci}" && [[ "$(uname -s)" == Darwin ]]; then +if ! docker compose version >/dev/null 2>&1; then brew install docker-compose plugin_dir="${DOCKER_CONFIG:-${HOME}/.docker}/cli-plugins" mkdir -p "${plugin_dir}" From 75687edd3c489c0b8d25ebbfe8c6f959328deeaa Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Fri, 2 Oct 2026 21:40:28 -0400 Subject: [PATCH 4/5] [UPDATE] Update [ghstack-poisoned] --- examples/llm_server/README.md | 5 ++--- .../evals/{ => terminal_bench}/README.md | 14 ++++++-------- 2 files changed, 8 insertions(+), 11 deletions(-) rename examples/llm_server/evals/{ => terminal_bench}/README.md (67%) diff --git a/examples/llm_server/README.md b/examples/llm_server/README.md index ea16d59d4e2..d3a1dae9560 100644 --- a/examples/llm_server/README.md +++ b/examples/llm_server/README.md @@ -110,6 +110,5 @@ Reliability guidance: ## Evaluate with Terminal-Bench -The [evaluation guide](evals/README.md) provides setup and run commands for -Terminal-Bench, configuration templates in `evals/configs/`, and task and -performance metrics on macOS. +See [Terminal-Bench](evals/terminal_bench/README.md) for setup, configuration, +and evaluation commands on macOS. diff --git a/examples/llm_server/evals/README.md b/examples/llm_server/evals/terminal_bench/README.md similarity index 67% rename from examples/llm_server/evals/README.md rename to examples/llm_server/evals/terminal_bench/README.md index 87ad736c77e..5e2db1e33be 100644 --- a/examples/llm_server/evals/README.md +++ b/examples/llm_server/evals/terminal_bench/README.md @@ -1,16 +1,14 @@ -# LLM server evaluations +# Terminal-Bench (macOS) -## Terminal-Bench (macOS) - -Prepare a worker, exported model, and [server environment](../python/README.md). +Prepare a worker, exported model, and [server environment](../../python/README.md). From the repository root: ```bash -cd examples/llm_server/evals -bash terminal_bench/setup.sh -cp configs/terminal-bench.example.toml configs/terminal-bench.local.toml +cd examples/llm_server/evals/terminal_bench +bash setup.sh +cp ../configs/terminal-bench.example.toml ../configs/terminal-bench.local.toml # Edit the model paths and context limit in the local TOML. -bash terminal_bench/run.sh --config configs/terminal-bench.local.toml +bash run.sh --config ../configs/terminal-bench.local.toml ``` Setup installs Colima (requires Homebrew) and Harbor, reusing Docker if available. From b527ff7c3a81189a784675e4cdc6da8ce9f5aadc Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Fri, 2 Oct 2026 21:57:10 -0400 Subject: [PATCH 5/5] [UPDATE] Update [ghstack-poisoned] --- .../evals/configs/terminal-bench.example.toml | 1 + .../llm_server/evals/terminal_bench/README.md | 8 +- .../llm_server/evals/terminal_bench/runner.py | 92 ++++++++++++++++--- 3 files changed, 86 insertions(+), 15 deletions(-) diff --git a/examples/llm_server/evals/configs/terminal-bench.example.toml b/examples/llm_server/evals/configs/terminal-bench.example.toml index bd202b4ac2c..90446d7e16b 100644 --- a/examples/llm_server/evals/configs/terminal-bench.example.toml +++ b/examples/llm_server/evals/configs/terminal-bench.example.toml @@ -12,5 +12,6 @@ no_think = true [terminal_bench] tasks = ["fix-git"] max_output_tokens = 1024 +temperature = 0.0 step_limit = 100 attempts = 1 diff --git a/examples/llm_server/evals/terminal_bench/README.md b/examples/llm_server/evals/terminal_bench/README.md index 5e2db1e33be..537a664312d 100644 --- a/examples/llm_server/evals/terminal_bench/README.md +++ b/examples/llm_server/evals/terminal_bench/README.md @@ -1,6 +1,7 @@ # Terminal-Bench (macOS) Prepare a worker, exported model, and [server environment](../../python/README.md). +Use a worker built for the server checkout; worker protocols can differ between revisions. From the repository root: ```bash @@ -14,10 +15,11 @@ bash run.sh --config ../configs/terminal-bench.local.toml Setup installs Colima (requires Homebrew) and Harbor, reusing Docker if available. Harbor downloads Terminal-Bench 2.0 tasks and runs mini-SWE-agent in containers. -The TOML selects the model, tasks, attempts, and token budgets. `max_context` must -fit the exported model. The default `fix-git` task is a smoke check; use a model -capable of tool calling for meaningful scores. +The TOML selects the model, tasks, attempts, temperature, and token budgets. +`max_context` must fit the exported model. The default `fix-git` task is a smoke +check; use a model capable of tool calling for meaningful scores. Scores, trajectories, logs, and token/timing metrics are saved under `~/.cache/executorch-evals/terminal-bench/runs/`. Use `--output DIR` to choose a results directory or `--dry-run` to inspect commands before running. +Metrics exclude the generation preflight. diff --git a/examples/llm_server/evals/terminal_bench/runner.py b/examples/llm_server/evals/terminal_bench/runner.py index 7a56a322b4e..e70cdda5886 100644 --- a/examples/llm_server/evals/terminal_bench/runner.py +++ b/examples/llm_server/evals/terminal_bench/runner.py @@ -44,6 +44,7 @@ def load_config(path): "attempts": 1, "step_limit": 100, "max_output_tokens": 512, + "temperature": 0.0, "agent_host": None, } options = config.get("terminal_bench", {}) @@ -56,6 +57,8 @@ def load_config(path): ) if min(options["attempts"], options["step_limit"]) < 1 or not options["tasks"]: raise ValueError("Choose tasks and positive attempt/step limits") + if not 0 <= options["temperature"] <= 2: + raise ValueError("temperature must be between 0 and 2") return config @@ -131,10 +134,23 @@ def commands(config, output): "curlimages/curl:8.12.1", "--noproxy", "*", - "--fail", + "--silent", + "--show-error", + "--fail-with-body", "--max-time", - "15", - url + "/health", + "90", + "--header", + "Content-Type: application/json", + "--data", + json.dumps( + { + "model": server["model_id"], + "messages": [{"role": "user", "content": "Hello."}], + "max_tokens": min(8, options["max_output_tokens"]), + "temperature": options["temperature"], + } + ), + url + "/v1/chat/completions", ] return serve, harbor, probe @@ -173,9 +189,9 @@ def stop(process, stop_signal=signal.SIGTERM): process.wait(timeout=10) -def metrics(path, context, reserve): +def metrics(log, context, reserve): turns = [] - for line in path.read_text(errors="replace").splitlines(): + for line in log.splitlines(): if "llm_turn_stats " in line: turns.append( dict( @@ -205,14 +221,40 @@ def metrics(path, context, reserve): return result +def summarize(harbor_dir): + for path in sorted(harbor_dir.glob("*/result.json")): + trial = json.loads(path.read_text()) + error = trial.get("exception_info") + reward = ((trial.get("verifier_result") or {}).get("rewards") or {}).get( + "reward" + ) + outcome = f"error={error['exception_type']}" if error else f"reward={reward}" + agent_exit = tool_calls = "unknown" + trajectory = path.parent / "agent/mini-swe-agent.trajectory.json" + if trajectory.exists(): + data = json.loads(trajectory.read_text()) + agent_exit = data.get("info", {}).get("exit_status", "unknown") + tool_calls = sum( + len(message.get("tool_calls") or []) + for message in data.get("messages", []) + if message.get("role") == "assistant" + ) + print( + f"{trial['trial_name']}: {outcome}, " + f"agent_exit={agent_exit}, tool_calls={tool_calls}", + flush=True, + ) + + def run(config, output, serve, harbor, probe): server = config["server"] process = job = None + metrics_start = None env = {**os.environ, "PYTHONUNBUFFERED": "1"} env["PYTHONPATH"] = str(REPO_ROOT / "src") + os.pathsep + env.get("PYTHONPATH", "") - with (output / "server.log").open("w") as server_log, (output / "harbor.log").open( - "w" - ) as harbor_log: + with (output / "server.log").open("w+", errors="replace") as server_log, ( + output / "harbor.log" + ).open("w") as harbor_log: try: with socket.socket() as port: port.bind((server["host"], server["port"])) @@ -233,6 +275,24 @@ def run(config, output, serve, harbor, probe): check=True, timeout=120, ) + response = json.loads((output / "connection.log").read_text()) + if ( + response["object"] != "chat.completion" + or response["choices"][0]["message"]["role"] != "assistant" + or response["usage"]["completion_tokens"] < 1 + ): + raise ValueError("Invalid chat completion response") + except ( + OSError, + ValueError, + KeyError, + IndexError, + TypeError, + subprocess.SubprocessError, + ) as error: + raise RuntimeError( + "Generation preflight failed; see connection.log and server.log" + ) from error finally: subprocess.run( ["docker", "rm", "--force", probe[probe.index("--name") + 1]], @@ -240,13 +300,17 @@ def run(config, output, serve, harbor, probe): stderr=subprocess.DEVNULL, timeout=20, ) + metrics_start = server_log.tell() + print("Generation preflight passed.", flush=True) job = subprocess.Popen( harbor, stdout=harbor_log, stderr=subprocess.STDOUT, start_new_session=True, ) - if job.wait(): + returncode = job.wait() + summarize(output / "harbor") + if returncode: raise RuntimeError("Harbor failed; see harbor.log") result = json.loads((output / "harbor/result.json").read_text()) stats = result["stats"] @@ -258,13 +322,13 @@ def run(config, output, serve, harbor, probe): raise RuntimeError( "Harbor reported incomplete or errored trials; see harbor/result.json" ) - print(json.dumps(stats, indent=2)) finally: stop(job, signal.SIGINT) stop(process) server_log.flush() + server_log.seek(metrics_start or 0) timing = metrics( - output / "server.log", + server_log.read() if metrics_start is not None else "", server["max_context"], config["terminal_bench"]["max_output_tokens"], ) @@ -303,7 +367,11 @@ def main(argv=None): json.dumps( { "agent": {"step_limit": config["terminal_bench"]["step_limit"]}, - "model": {"model_kwargs": {"temperature": 0}}, + "model": { + "model_kwargs": { + "temperature": config["terminal_bench"]["temperature"] + } + }, } ) )