Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .github/workflows/_llm_server.yml
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,7 @@ jobs:

python -m pytest -q \
examples/llm_server/python/tests \
examples/llm_server/evals/terminal_bench/tests \
examples/models/muse-glimmer/tests/test_serve.py

cmake -S . -B cmake-out \
Expand Down
92 changes: 92 additions & 0 deletions .github/workflows/llm-server-evals.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
name: LLM Server Evaluations

on:
schedule:
- cron: '17 6 * * *'
workflow_dispatch:
inputs:
suite:
description: Terminal-Bench task subset
type: choice
options: [smoke, nightly]
default: smoke

permissions:
contents: read

jobs:
terminal-bench:
if: ${{ vars.LLM_SERVER_EVAL_TARGETS != '' }}
strategy:
fail-fast: false
matrix:
target: ${{ fromJSON(vars.LLM_SERVER_EVAL_TARGETS || '[{"name":"unconfigured","runner":"ubuntu-22.04"}]') }}
name: Terminal-Bench (${{ matrix.target.name }})
runs-on: ${{ matrix.target.runner }}
timeout-minutes: 180
concurrency:
group: llm-server-evals-${{ matrix.target.name }}
cancel-in-progress: false
env:
EVAL_CONFIG: ${{ matrix.target.config }}
EVAL_PREPARE: ${{ matrix.target.prepare }}
EVAL_SUITE: ${{ inputs.suite || 'nightly' }}
EVAL_TARGET: ${{ matrix.target.name }}
steps:
- name: Initialize artifacts
shell: bash
run: |
set -euo pipefail
EVAL_ARTIFACTS="${RUNNER_TEMP}/llm-server-evals-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}-${EVAL_TARGET}"
mkdir -p "${EVAL_ARTIFACTS}"
echo "EVAL_ARTIFACTS=${EVAL_ARTIFACTS}" >> "${GITHUB_ENV}"
if [[ "${EVAL_CONFIG}" != /* ]]; then
EVAL_CONFIG="${GITHUB_WORKSPACE}/${EVAL_CONFIG}"
fi
echo "EVAL_CONFIG=${EVAL_CONFIG}" >> "${GITHUB_ENV}"
printf '%s\n' "${GITHUB_SHA}" "${RUNNER_OS}" "${RUNNER_ARCH}" > "${EVAL_ARTIFACTS}/ci.txt"
- uses: actions/checkout@v4
with:
submodules: recursive
- uses: actions/setup-python@v5
with:
python-version: '3.12'
- name: Prepare the current checkout and model
shell: bash
run: |
set -euo pipefail
mkdir -p "${EVAL_ARTIFACTS}"
test -f "${EVAL_PREPARE}"
cp "${EVAL_PREPARE}" "${EVAL_ARTIFACTS}/prepare.sh"
bash "${EVAL_PREPARE}" "${GITHUB_WORKSPACE}" "${EVAL_CONFIG}" 2>&1 | tee "${EVAL_ARTIFACTS}/prepare.log"
test -f "${EVAL_CONFIG}"
cp "${EVAL_CONFIG}" "${EVAL_ARTIFACTS}/model.toml"
- name: Set up Terminal-Bench
shell: bash
run: |
set -euo pipefail
bash examples/llm_server/evals/setup.sh terminal-bench --ci 2>&1 | tee "${EVAL_ARTIFACTS}/setup.log"
- name: Evaluate
shell: bash
run: |
set -euo pipefail
bash examples/llm_server/evals/run.sh terminal-bench \
--config "${EVAL_CONFIG}" --suite "${EVAL_SUITE}" \
--executorch-root "${GITHUB_WORKSPACE}" \
--run-dir "${EVAL_ARTIFACTS}/run" 2>&1 | tee "${EVAL_ARTIFACTS}/evaluation.log"
- name: Publish summary
if: always()
shell: bash
run: |
if [[ -f "${EVAL_ARTIFACTS}/run/summary.md" ]]; then
cat "${EVAL_ARTIFACTS}/run/summary.md" >> "${GITHUB_STEP_SUMMARY}"
else
echo 'Evaluation did not produce a summary; inspect preparation and setup logs.' >> "${GITHUB_STEP_SUMMARY}"
fi
- uses: actions/upload-artifact@v4
if: always()
with:
name: terminal-bench-${{ matrix.target.name }}
path: ${{ env.EVAL_ARTIFACTS }}
if-no-files-found: warn
retention-days: 30
7 changes: 7 additions & 0 deletions examples/llm_server/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@ examples/llm_server/
spec/ # language-neutral OpenAI contract ExecuTorch targets
conformance/ # one test suite every language server must pass
python/ # Python server implementation (current)
evals/ # task accuracy and performance through the server
# cpp/ # future: no-Python single-binary server
```

Expand Down Expand Up @@ -106,3 +107,9 @@ Reliability guidance:
`tools` were included in the request.
- If a request fails with `unsupported_parameter`, remove or disable that
OpenAI knob in your pi/client config.

## Evaluate with Terminal-Bench

The [evaluation guide](evals/README.md) provides setup and run commands for
Terminal-Bench, configuration templates in `evals/configs/`, recorded task and
performance metrics, and periodic CI on provisioned Linux GPU and macOS runners.
146 changes: 146 additions & 0 deletions examples/llm_server/evals/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,146 @@
# LLM server evaluations

Run end-to-end evaluations through the ExecuTorch LLM server. Terminal-Bench is
available today; additional harnesses such as tau2 and GuideLLM can be added as
sibling directories with their own dependencies and metrics. TOML configurations
live in `configs/`.

## Setup

From the repository root:

```bash
bash examples/llm_server/evals/setup.sh terminal-bench
```

Setup reuses a working Docker installation, or installs Docker Engine on Ubuntu
and Colima with the Docker CLI on macOS. Installing system packages can require
administrator authentication; macOS installation requires Homebrew. On a
provisioned CI runner, add `--ci` to require existing Docker without changing
system services.

The script creates an isolated Python 3.12 environment with Harbor 0.22.0 and
downloads a pinned Terminal-Bench task subset. Harbor installs mini-SWE-agent
2.4.6 inside each task container when the trial starts. No host installation of
the agent is needed. Setup can be rerun; it preserves an unchanged task cache
and reports modifications instead of overwriting them.

Dependencies and tasks live under `~/.cache/executorch-evals` (or
`$XDG_CACHE_HOME/executorch-evals`). Set `EXECUTORCH_EVAL_CACHE` consistently for
setup and run to use another location. Setup does not replace your ExecuTorch
Python environment. Prepare an exported model, compatible native worker, and
[server environment](../python/README.md)
before running an evaluation.

## Configure once

```bash
cp examples/llm_server/evals/configs/terminal-bench.example.toml examples/llm_server/evals/configs/terminal-bench.local.toml
```

Files named `*.local.toml` are ignored by Git so machine-specific paths stay local.
Edit the worker, model, tokenizer, and server Python paths. Set `max_context` to
a capacity supported by the export and choose `max_output_tokens` independently.
The example uses a 2,048-token context for an integration check; use a qualified
model and appropriate context/output budgets for task-quality evaluation.

## Run

```bash
bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --suite smoke
```

The command checks prerequisites, starts a fresh server for each trial, probes
the server from a Docker container, runs Harbor, and cleans up its processes.
Linux automatically maps `host.docker.internal` to the host gateway for Harbor
task containers. The launcher uses `host.lima.internal` for a Colima context on
macOS and `host.docker.internal` for Docker Desktop. Use `agent_base_url` for a
custom container route; the server must listen on an address containers can reach.

`smoke` runs `fix-git`; `nightly` adds `openssl-selfsigned-cert`. Both are pinned
subsets, not a full Terminal-Bench score. Use `--task /path/to/task` for a custom
task, or `--task-root /path/to/tasks` with a named suite.

```bash
# Inspect dependencies without running a trial.
bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --check
# Record the configuration and commands without launching processes.
bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --dry-run
# Run the same subset used by periodic CI.
bash examples/llm_server/evals/run.sh terminal-bench --config examples/llm_server/evals/configs/terminal-bench.local.toml --suite nightly
```

`--check` checks host prerequisites. The container route is tested after the
server starts during an actual run; a particular task's custom network can still
require additional configuration. See the [Terminal-Bench guide](terminal_bench/README.md)
for budget controls, launcher configuration, and resuming runs.

## Results

The command prints a fresh results directory for every invocation, under
`<evaluation-cache>/terminal-bench/runs` unless `output_root` is configured. It contains
`summary.md`, `results.json`, `results.tsv`, resolved configuration, source and
asset checksums, exact commands, server logs, agent trajectories, and verifier
outputs. Setup dependency versions and the selected task revision are recorded.

Results distinguish task rewards from infrastructure errors and report executed
tool observations, token usage, and prefill/decode timings. Reward zero with no
tool use does not establish a meaningful agent rollout.

## Adding an evaluation

Add a sibling package with its own `__main__.py`, `setup.sh`, `run.sh`,
requirements, documentation, and tests; place its TOML examples in `configs/` and register its entry points
in the two top-level scripts. Keep harness-specific behavior and dependencies inside that package.
Extract shared utilities when a second integration needs them. Server protocol
tests remain beside the server implementation in `../python/tests`.

## Periodic CI on Linux GPU and macOS

The [LLM Server Evaluations workflow](../../../.github/workflows/llm-server-evals.yml)
runs the nightly subset at 06:17 UTC and supports manual dispatch. It becomes
active when the repository variable `LLM_SERVER_EVAL_TARGETS` is configured.
Each matrix entry selects its own runner, model configuration, and preparation
script. For example, after provisioning runners with these labels:

```json
[
{
"name": "linux-gpu",
"runner": ["self-hosted", "Linux", "X64", "executorch-evals-gpu"],
"config": "examples/llm_server/evals/configs/linux-gpu.local.toml",
"prepare": "/opt/executorch-evals/prepare.sh"
},
{
"name": "macos-mlx",
"runner": ["self-hosted", "macOS", "ARM64", "executorch-evals-mlx"],
"config": "examples/llm_server/evals/configs/macos-mlx.local.toml",
"prepare": "/Users/runner/executorch-evals/prepare.sh"
}
]
```

Both runners need working Docker and Compose accessible by the runner user. The
Linux GPU runner also needs its model backend and drivers. The macOS runner
needs its selected model backend (for example, Metal/MLX) and Docker Desktop or Colima with virtualization support;
an ordinary hosted macOS VM is not assumed to meet these requirements.

The preparation script receives the current checkout and configuration paths:

```bash
bash /path/to/prepare.sh "$GITHUB_WORKSPACE" "$GITHUB_WORKSPACE/examples/llm_server/evals/configs/macos-mlx.local.toml"
```

CI resolves relative configuration paths against the checkout before calling
the preparation script. It must build the compatible worker from that checkout, prepare the server
environment, obtain pinned model/tokenizer assets, and write or update the
configuration to those paths. Cached weights may be reused; a worker built from
an older checkout would not evaluate the C++ changes under test. Keep model
revision, quantization, context, output allowance, and task settings fixed when
comparing runs. Report Linux and macOS performance separately.

CI invokes the same setup and run scripts as local development. It uploads
configuration, preparation/setup logs, and evaluation artifacts even on failure.
Infrastructure errors fail the job; task rewards are reported without a quality
threshold until a reliable model baseline is established. Lightweight driver
tests continue to run in the existing LLM server CI workflow.
5 changes: 5 additions & 0 deletions examples/llm_server/evals/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
# Copyright (c) Meta Platforms, Inc. and affiliates.
# All rights reserved.
#
# This source code is licensed under the BSD-style license found in the
# LICENSE file in the root directory of this source tree.
1 change: 1 addition & 0 deletions examples/llm_server/evals/configs/.gitignore
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
*.local.toml
27 changes: 27 additions & 0 deletions examples/llm_server/evals/configs/terminal-bench.example.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
# Copy to terminal-bench.local.toml and adjust paths (relative to this file).
# setup.sh downloads the tasks for the named suites.
suite = "smoke"
worker_bin = "/path/to/model_worker"
model_path = "/path/to/model.pte"
tokenizer_path = "/path/to/tokenizer.json"
hf_tokenizer = "/path/to/pinned-hf-tokenizer"
server_python = "/path/to/executorch-env/bin/python"
model_id = "qwen3"

max_context = 2048
max_output_tokens = 512
step_limit = 100
attempts = 1
# Enable only when the worker advertises named-session support.
session_affinity = false
# Optional model-specific launcher; it must accept the common server flags.
# server_module = "executorch.examples.llm_server.python.server"
# server_arg = ["--assistant-header=<|im_start|>assistant\n"]
thinking = false

host = "0.0.0.0"
port = 8000
# run.sh chooses the Docker/Colima host route. Override for custom networking:
# agent_base_url = "http://host.docker.internal:8000/v1"
# Results default to the evaluation cache. Override if desired:
# output_root = "/path/to/evaluation-results"
18 changes: 18 additions & 0 deletions examples/llm_server/evals/run.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
#!/usr/bin/env bash
# Copyright (c) Meta Platforms, Inc. and affiliates.
# All rights reserved.
#
# This source code is licensed under the BSD-style license found in the
# LICENSE file in the root directory of this source tree.
set -euo pipefail

evals_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)
case "${1:---help}" in
terminal-bench) harness=terminal_bench ;;
--help|-h)
echo "Usage: bash examples/llm_server/evals/run.sh terminal-bench [driver options]"
exit 0 ;;
*) echo "Unsupported evaluation: $1" >&2; exit 2 ;;
esac
shift
exec bash "${evals_dir}/${harness}/run.sh" "$@"
18 changes: 18 additions & 0 deletions examples/llm_server/evals/setup.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
#!/usr/bin/env bash
# Copyright (c) Meta Platforms, Inc. and affiliates.
# All rights reserved.
#
# This source code is licensed under the BSD-style license found in the
# LICENSE file in the root directory of this source tree.
set -euo pipefail

evals_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)
case "${1:---help}" in
terminal-bench) harness=terminal_bench ;;
--help|-h)
echo "Usage: bash examples/llm_server/evals/setup.sh terminal-bench [--ci]"
exit 0 ;;
*) echo "Unsupported evaluation: $1" >&2; exit 2 ;;
esac
shift
exec bash "${evals_dir}/${harness}/setup.sh" "$@"
Loading
Loading