From 82144247d67385179014250ae219504ce5a6a2b9 Mon Sep 17 00:00:00 2001 From: Songhao Jia Date: Fri, 2 Oct 2026 18:45:02 -0700 Subject: [PATCH] Merge cuda-perf.yml into cuda.yml and add 5% perf regression gate (#23385) Summary: cuda-perf.yml is retired. Its benchmark + upload pipeline now runs inside cuda.yml's test-model-cuda-e2e cells, reusing each cell's exported pte/ptd and built runner in place (new helper .ci/scripts/cuda_benchmark_from_e2e.sh, allowlisted to the 7 perf models; whisper-medium backfilled into the accuracy matrix). New jobs: - upload-benchmark-results: aggregates cuda-bench-* artifacts to the unchanged s3://gha-artifacts/executorch-cuda-perf/ prefix (history preserved) + HUD dashboard v3 upload. - check-perf-regression: fails the PR when decode or prefill tok/s drops >5% vs the median of the last 5 successful main runs (same model, quant, GPU, min 3 history points; new .ci/scripts/cuda_check_regression.py). Bypass with the bypass-perf-regression PR label. Cleanup: delete cuda-perf.yml, rename trigger_cuda_perf.sh to trigger_cuda_benchmark.sh (retargeted at cuda.yml dispatch inputs), drop ciflow/cuda-perf from pytorch-probot.yml, update the _ci-run-decision.yml caller comment. Differential Revision: D123135327 --- .ci/scripts/cuda_benchmark_from_e2e.sh | 187 ++++++++++ .ci/scripts/cuda_check_regression.py | 401 +++++++++++++++++++++ .github/pytorch-probot.yml | 1 - .github/scripts/trigger_cuda_benchmark.sh | 81 +++++ .github/scripts/trigger_cuda_perf.sh | 100 ------ .github/workflows/_ci-run-decision.yml | 6 +- .github/workflows/cuda-perf.yml | 407 ---------------------- .github/workflows/cuda.yml | 270 +++++++++++++- 8 files changed, 941 insertions(+), 512 deletions(-) create mode 100755 .ci/scripts/cuda_benchmark_from_e2e.sh create mode 100644 .ci/scripts/cuda_check_regression.py create mode 100755 .github/scripts/trigger_cuda_benchmark.sh delete mode 100755 .github/scripts/trigger_cuda_perf.sh delete mode 100644 .github/workflows/cuda-perf.yml diff --git a/.ci/scripts/cuda_benchmark_from_e2e.sh b/.ci/scripts/cuda_benchmark_from_e2e.sh new file mode 100755 index 00000000000..53520e1918c --- /dev/null +++ b/.ci/scripts/cuda_benchmark_from_e2e.sh @@ -0,0 +1,187 @@ +#!/bin/bash +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +# Run the CUDA perf benchmark for one (model, quant) pair, reusing the +# .pte/.ptd artifacts and the already-built runner left behind by +# test_model_e2e.sh in the same CI job. +# +# Only the (model, quant) pairs in the perf allowlist below are benchmarked. +# Anything else exits 0 after printing BENCHMARK_SKIP, so the caller can tell +# "not covered by perf" apart from a real failure. The optional +# models/quantizations filters (from workflow_dispatch) narrow the benchmark +# stage further without touching the accuracy stage. +# +# Usage: +# cuda_benchmark_from_e2e.sh +# [num_runs] [git_sha] [run_id] [run_url] [models_filter] [quants_filter] +# +# Outputs (in ): +# benchmark_results.json, benchmark_results_v3.json, metadata.json + +set -euo pipefail + +HF_MODEL="${1:?hf_model required, e.g. openai/whisper-small}" +QUANT_NAME="${2:?quant required, e.g. non-quantized}" +MODEL_DIR="${3:?model_dir required}" +RESULTS_DIR="${4:?results_dir required}" +NUM_RUNS="${5:-50}" +GIT_SHA="${6:-unknown}" +WORKFLOW_RUN_ID="${7:-0}" +WORKFLOW_RUN_URL="${8:-}" +MODELS_FILTER="${9:-}" +QUANTS_FILTER="${10:-}" + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +EXECUTORCH_ROOT="$(cd "${SCRIPT_DIR}/../.." && pwd)" + +# (model, quant) pairs covered by perf tracking. Keep in sync with the +# test-model-cuda-e2e matrix in .github/workflows/cuda.yml. +is_bench_pair() { + case "${HF_MODEL} ${QUANT_NAME}" in + "mistralai/Voxtral-Mini-3B-2507 "* | \ + "openai/whisper-small non-quantized" | \ + "openai/whisper-medium non-quantized" | \ + "openai/whisper-large-v3-turbo quantized-"* | \ + "google/gemma-3-4b-it non-quantized" | \ + "google/gemma-3-4b-it quantized-int4-tile-packed" | \ + "nvidia/parakeet-tdt non-quantized" | \ + "nvidia/parakeet-tdt quantized-int4-tile-packed" | \ + "SocialLocalMobile/Qwen3.5-35B-A3B-HQQ-INT4 quantized-int4-tile-packed") + return 0 + ;; + *) + return 1 + ;; + esac +} + +if ! is_bench_pair; then + echo "BENCHMARK_SKIP: ${HF_MODEL} ${QUANT_NAME} is not in the perf allowlist" + exit 0 +fi + +# workflow_dispatch filters: comma-separated HF model IDs / quant names. +# Empty means "benchmark every allowlisted pair". +case ",${MODELS_FILTER}," in +*=",${HF_MODEL},"*) ;; +*) + if [ -n "${MODELS_FILTER}" ]; then + echo "BENCHMARK_SKIP: ${HF_MODEL} not in dispatch models filter" + exit 0 + fi + ;; +esac +case ",${QUANTS_FILTER}," in +*=",${QUANT_NAME},"*) ;; +*) + if [ -n "${QUANTS_FILTER}" ]; then + echo "BENCHMARK_SKIP: ${QUANT_NAME} not in dispatch quantizations filter" + exit 0 + fi + ;; +esac + +cd "${EXECUTORCH_ROOT}" +export LD_LIBRARY_PATH="/opt/conda/lib${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" + +# Runner command mapping, ported from the retired cuda-perf.yml so the +# benchmark prompts and flags (and therefore the numbers) stay identical. +case "${HF_MODEL}" in +mistralai/Voxtral-Mini-3B-2507) + RUNNER="cmake-out/examples/models/voxtral/voxtral_runner" + PREPROCESSOR="${MODEL_DIR}/voxtral_preprocessor.pte" + TOKENIZER="${MODEL_DIR}/tekken.json" + AUDIO="${MODEL_DIR}/poem.wav" + RUNNER_CMD="$RUNNER --model_path ${MODEL_DIR}/model.pte --data_path ${MODEL_DIR}/aoti_cuda_blob.ptd --tokenizer_path $TOKENIZER --audio_path $AUDIO --processor_path $PREPROCESSOR --temperature 0" + MODEL_NAME="voxtral_${QUANT_NAME}" + ;; +openai/whisper-*) + RUNNER="cmake-out/examples/models/whisper/whisper_runner" + PREPROCESSOR="${MODEL_DIR}/whisper_preprocessor.pte" + AUDIO="${MODEL_DIR}/output.wav" + RUNNER_CMD="$RUNNER --model_path ${MODEL_DIR}/model.pte --data_path ${MODEL_DIR}/aoti_cuda_blob.ptd --tokenizer_path ${MODEL_DIR}/ --audio_path $AUDIO --processor_path $PREPROCESSOR --temperature 0" + MODEL_NAME="${HF_MODEL#openai/}_${QUANT_NAME}" + ;; +google/gemma-3-4b-it) + RUNNER="cmake-out/examples/models/gemma3/gemma3_e2e_runner" + IMAGE="docs/source/_static/img/et-logo.png" + RUNNER_CMD="$RUNNER --model_path ${MODEL_DIR}/model.pte --data_path ${MODEL_DIR}/aoti_cuda_blob.ptd --tokenizer_path ${MODEL_DIR}/ --image_path $IMAGE --temperature 0" + MODEL_NAME="gemma3_${QUANT_NAME}" + ;; +nvidia/parakeet-tdt) + RUNNER="cmake-out/examples/models/parakeet/parakeet_runner" + AUDIO="${MODEL_DIR}/test_audio.wav" + TOKENIZER="${MODEL_DIR}/tokenizer.model" + RUNNER_CMD="$RUNNER --model_path ${MODEL_DIR}/model.pte --data_path ${MODEL_DIR}/aoti_cuda_blob.ptd --audio_path $AUDIO --tokenizer_path $TOKENIZER" + MODEL_NAME="parakeet_${QUANT_NAME}" + ;; +SocialLocalMobile/Qwen3.5-35B-A3B-HQQ-INT4) + RUNNER="cmake-out/examples/models/qwen3_5_moe/qwen3_5_moe_runner" + TOKENIZER="${MODEL_DIR}/tokenizer.json" + # A checked-in long prompt (>1000 tokens). A static, meaningful prompt + # avoids the degenerate / repetitive outputs that can result from + # synthetic prompts built by repeating the same sentence. + PROMPT_FILE=".ci/scripts/cuda_perf_prompts/qwen3_5_moe_long_prompt.txt" + RUNNER_CMD="$RUNNER --model_path ${MODEL_DIR}/model.pte --data_path ${MODEL_DIR}/aoti_cuda_blob.ptd --tokenizer_path $TOKENIZER --prompt_file $PROMPT_FILE --max_new_tokens 512 --temperature 0" + MODEL_NAME="qwen3_5_moe_${QUANT_NAME}" + ;; +*) + echo "BENCHMARK_SKIP: no runner mapping for '${HF_MODEL}'" + exit 0 + ;; +esac + +echo "::group::Running benchmark for ${HF_MODEL} (${QUANT_NAME}) with ${NUM_RUNS} runs" + +GPU_NAME=$(nvidia-smi --query-gpu=name --format=csv,noheader | head -1) +echo "Detected GPU: $GPU_NAME" +CUDA_DRIVER_VERSION=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader | head -1) +echo "CUDA Driver Version: $CUDA_DRIVER_VERSION" + +mkdir -p "${RESULTS_DIR}" + +# Run benchmark using cuda_benchmark.py +python .ci/scripts/cuda_benchmark.py \ + --runner_command "$RUNNER_CMD" \ + --model_name "$MODEL_NAME" \ + --num_runs "${NUM_RUNS}" \ + --output_json "${RESULTS_DIR}/benchmark_results.json" \ + --output_v3 "${RESULTS_DIR}/benchmark_results_v3.json" \ + --model "${HF_MODEL}" \ + --quantization "${QUANT_NAME}" \ + --git_sha "${GIT_SHA}" \ + --workflow_run_id "${WORKFLOW_RUN_ID}" \ + --workflow_run_url "${WORKFLOW_RUN_URL}" \ + --gpu_name "$GPU_NAME" \ + --cuda_driver_version "$CUDA_DRIVER_VERSION" + +# Save additional metadata (written via a serializer, never string-built). +BENCH_MODEL="${HF_MODEL}" BENCH_QUANT="${QUANT_NAME}" BENCH_NUM_RUNS="${NUM_RUNS}" \ + BENCH_RUNNER="${RUNNER}" BENCH_SHA="${GIT_SHA}" BENCH_RUN_ID="${WORKFLOW_RUN_ID}" \ + BENCH_RUN_URL="${WORKFLOW_RUN_URL}" BENCH_GPU="${GPU_NAME}" \ + RESULTS_DIR="${RESULTS_DIR}" python3 - <<'PY' +import datetime +import json +import os + +metadata = { + "model": os.environ["BENCH_MODEL"], + "quantization": os.environ["BENCH_QUANT"], + "num_runs": int(os.environ["BENCH_NUM_RUNS"]), + "runner": os.environ["BENCH_RUNNER"], + "timestamp": datetime.datetime.now(datetime.timezone.utc).strftime( + "%Y-%m-%dT%H:%M:%SZ" + ), + "git_sha": os.environ["BENCH_SHA"], + "workflow_run_id": os.environ["BENCH_RUN_ID"], + "workflow_run_url": os.environ["BENCH_RUN_URL"], + "gpu_name": os.environ["BENCH_GPU"], +} +with open(os.path.join(os.environ.get("RESULTS_DIR", "."), "metadata.json"), "w") as f: + json.dump(metadata, f, indent=2) +PY +echo "::endgroup::" diff --git a/.ci/scripts/cuda_check_regression.py b/.ci/scripts/cuda_check_regression.py new file mode 100644 index 00000000000..4c37ce7f35a --- /dev/null +++ b/.ci/scripts/cuda_check_regression.py @@ -0,0 +1,401 @@ +#!/usr/bin/env python3 +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +"""Fail CI when CUDA benchmark throughput regresses vs recent main history. + +Compares the current run's decode throughput (``throughput_mean``) and +prefill throughput (``prefill_throughput_mean``) against the median of the +last few successful ``cuda.yml`` runs on ``main`` for the same +(model, quantization, GPU), using artifacts stored under +``s3:///executorch-cuda-perf///``. + +Any metric dropping more than ``--threshold-pct`` fails the job +(``exit 1`` with ``::error::`` annotations). Keys with fewer than +``--min-history`` baseline points are skipped (cold-start protection). + +A PR carrying ``--bypass-label`` passes unconditionally with a ``::notice::``. +""" + +from __future__ import annotations + +import argparse +import json +import logging +import os +import statistics +import subprocess +import sys +import tempfile +import urllib.request +from typing import Any, Iterable + +logger: logging.Logger = logging.getLogger(__name__) + +KNOWN_QUANTS = ( + "non-quantized", + "quantized-int4-tile-packed", + "quantized-int4-weight-only", +) + +# Artifact dir prefixes, current and legacy (retired cuda-perf.yml) layout. +DIR_PREFIXES = ("cuda-bench-", "results-") + + +def split_artifact_dir(dirname: str) -> tuple[str, str] | None: + """Split '-' into (model_safe, quant). + + Also accepts a nested path: walks up until a segment carries the prefix + (e2e uploads results under a ``bench/`` subdir), so both + ``cuda-bench--`` and ``cuda-bench--/bench`` + resolve. + """ + parts = dirname.split(os.sep) + for part in reversed(parts): + for prefix in DIR_PREFIXES: + if part.startswith(prefix): + rest = part[len(prefix):] + for quant in KNOWN_QUANTS: + if rest.endswith("-" + quant): + return rest[: -(len(quant) + 1)], quant + return None + + +def load_current_results(results_dir: str) -> dict[tuple[str, str, str], dict[str, float]]: + """Walk downloaded artifacts, return {(model, quant, gpu): means}.""" + current: dict[tuple[str, str, str], dict[str, float]] = {} + for root, _dirs, files in os.walk(results_dir): + if "benchmark_results.json" not in files: + continue + with open(os.path.join(root, "benchmark_results.json")) as f: + data = json.load(f) + means = { + "decode": data.get("throughput_mean"), + "prefill": data.get("prefill_throughput_mean"), + } + if means["decode"] is None or means["prefill"] is None: + logger.warning("Skipping %s: missing throughput means", root) + continue + model, quant, gpu = identify_result(root, files, data) + if model is None or quant is None or gpu is None: + logger.warning("Skipping %s: cannot identify (model, quant, gpu)", root) + continue + current[(model, quant, gpu)] = means + return current + + +def identify_result( + root: str, files: list[str], data: dict[str, Any] +) -> tuple[str | None, str | None, str | None]: + """Resolve (model_safe, quant, gpu) preferring metadata.json, then v3, then dirname. + + Each field falls back independently: legacy metadata.json files (from + the retired cuda-perf.yml) carry model/quantization but no gpu_name, + so the v3 runners GPU must still apply even when model/quant are + already known. Slash-normalization matches the metadata path so both + produce the same key. + """ + model: str | None = None + quant: str | None = None + gpu: str | None = None + if "metadata.json" in files: + with open(os.path.join(root, "metadata.json")) as f: + meta = json.load(f) + raw_model = meta.get("model") + if raw_model: + model = str(raw_model).replace("/", "_") + quant = meta.get("quantization") + gpu = meta.get("gpu_name") + if "benchmark_results_v3.json" in files and ( + model is None or quant is None or gpu is None + ): + with open(os.path.join(root, "benchmark_results_v3.json")) as f: + v3 = json.load(f) + if v3: + rec = v3[0] + if model is None or quant is None: + full = rec.get("model", {}).get("name", "") + for known in KNOWN_QUANTS: + if full.endswith("_" + known): + model = (model or full[: -(len(known) + 1)]).replace("/", "_") + quant = quant or known + if gpu is None: + runners = rec.get("runners", []) + gpu = runners[0].get("name") if runners else None + if quant is None: + # v1 model_name ("whisper-small_non-quantized") is org-less, so it + # can only safely contribute the quant, never the model key. + short = data.get("model_name", "") + for known in KNOWN_QUANTS: + if short.endswith("_" + known): + quant = known + if model is None or quant is None: + split = split_artifact_dir(root) + if split is not None: + model = model or split[0] + quant = quant or split[1] + return model, quant, gpu + + +def github_api(url: str, token: str) -> Any: + """GET a GitHub REST URL, return parsed JSON.""" + req = urllib.request.Request( + url, + headers={ + "Accept": "application/vnd.github.v3+json", + "Authorization": f"Bearer {token}", + }, + ) + with urllib.request.urlopen(req, timeout=60) as resp: + return json.load(resp) + + +def list_successful_runs(repo: str, workflow: str, branch: str, token: str) -> list[int]: + """Newest-first run ids of successful workflow runs on a branch.""" + runs: list[int] = [] + page = 1 + while len(runs) < 30 and page <= 3: + url = ( + f"https://api.github.com/repos/{repo}/actions/workflows/{workflow}" + f"/runs?status=success&branch={branch}&per_page=30&page={page}" + ) + payload = github_api(url, token) + batch = payload.get("workflow_runs", []) + if not batch: + break + runs.extend(r["id"] for r in batch) + page += 1 + return runs + + +def has_bypass_label(repo: str, pr_number: str, label: str, token: str) -> bool: + """Whether a PR carries the bypass label.""" + url = f"https://api.github.com/repos/{repo}/issues/{pr_number}/labels?per_page=100" + labels = github_api(url, token) + return any(entry.get("name") == label for entry in labels) + + +def run_aws(args: list[str]) -> subprocess.CompletedProcess[str]: + """Run an aws CLI command, never through a shell.""" + return subprocess.run( + ["aws", *args], capture_output=True, text=True, timeout=300 + ) + + +def list_attempts(bucket: str, prefix: str, run_id: int) -> list[str]: + """Attempt prefixes under s3://bucket/prefix//.""" + result = run_aws(["s3", "ls", f"s3://{bucket}/{prefix}/{run_id}/"]) + if result.returncode != 0: + return [] + attempts = [] + for line in result.stdout.splitlines(): + parts = line.split() + if len(parts) == 2 and parts[0] == "PRE": + attempts.append(parts[1].rstrip("/")) + return attempts + + +def fetch_history_points( + bucket: str, + prefix: str, + run_id: int, + keys: Iterable[tuple[str, str, str]], + tmp_root: str, +) -> dict[tuple[str, str, str], dict[str, list[float]]]: + """Pull one run's benchmark means from S3 into per-key point lists.""" + points: dict[tuple[str, str, str], dict[str, list[float]]] = {} + for attempt in list_attempts(bucket, prefix, run_id): + dest = os.path.join(tmp_root, str(run_id), attempt) + result = run_aws( + [ + "s3", + "sync", + f"s3://{bucket}/{prefix}/{run_id}/{attempt}/", + dest, + "--exclude", + "*", + "--include", + "benchmark_results.json", + "--include", + "benchmark_results_v3.json", + "--include", + "metadata.json", + ] + ) + if result.returncode != 0: + continue + # First attempt with data wins for this run id. + has_data = any( + "benchmark_results.json" in files + for _root, _dirs, files in os.walk(dest) + ) + if has_data: + for key, means in load_current_results(dest).items(): + if key in keys: + entry = points.setdefault( + key, {"decode": [], "prefill": []} + ) + entry["decode"].append(means["decode"]) + entry["prefill"].append(means["prefill"]) + break + return points + + +def collect_baselines(args: argparse.Namespace, keys: set[tuple[str, str, str]]) -> dict: + """Median of the last N successful main runs per key from S3.""" + run_ids = list_successful_runs(args.repo, args.workflow, args.branch, args.token) + # Don't compare against ourselves when re-running the current run id. + run_ids = [r for r in run_ids if str(r) != str(args.exclude_run_id)] + per_key: dict[tuple[str, str, str], dict[str, list[float]]] = { + key: {"decode": [], "prefill": []} for key in keys + } + with tempfile.TemporaryDirectory(prefix="cuda-regression-") as tmp_root: + for run_id in run_ids: + if all(len(v["decode"]) >= args.baseline_window for v in per_key.values()): + break + for key, vals in fetch_history_points( + args.s3_bucket, args.s3_prefix, run_id, keys, tmp_root + ).items(): + slot = per_key[key] + if len(slot["decode"]) < args.baseline_window: + slot["decode"].extend(vals["decode"]) + slot["prefill"].extend(vals["prefill"]) + baselines = {} + for key, vals in per_key.items(): + if len(vals["decode"]) >= args.min_history: + baselines[key] = { + "decode": statistics.median(vals["decode"]), + "prefill": statistics.median(vals["prefill"]), + "n": len(vals["decode"]), + } + return baselines + + +def check_regressions( + current: dict, baselines: dict, threshold_pct: float +) -> list[dict[str, Any]]: + """Compare current means vs baselines, return per-key verdict rows.""" + rows = [] + for key, means in sorted(current.items()): + base = baselines.get(key) + if base is None: + rows.append({"key": key, "status": "SKIP", "reason": "no baseline history"}) + continue + deltas = { + name: (means[name] - base[name]) / base[name] * 100.0 + for name in ("decode", "prefill") + } + failed = any(delta < -threshold_pct for delta in deltas.values()) + rows.append( + { + "key": key, + "status": "FAIL" if failed else "PASS", + "new_decode": means["decode"], + "base_decode": base["decode"], + "delta_decode": deltas["decode"], + "new_prefill": means["prefill"], + "base_prefill": base["prefill"], + "delta_prefill": deltas["prefill"], + "n": base["n"], + } + ) + return rows + + +def render_markdown(rows: list[dict[str, Any]], threshold_pct: float) -> str: + """Render verdict rows as a markdown table.""" + lines = [ + "## CUDA perf regression check", + "", + f"Fail threshold: either metric drops more than {threshold_pct:.1f}% " + "vs the median of recent successful main runs (same model, quant, GPU).", + "", + "| model | quant | gpu | decode new / base (Δ%) | " + "prefill new / base (Δ%) | history | verdict |", + "| --- | --- | --- | --- | --- | --- | --- |", + ] + for row in rows: + model, quant, gpu = row["key"] + if row["status"] == "SKIP": + lines.append( + f"| {model} | {quant} | {gpu} | — | — | 0 | SKIP ({row['reason']}) |" + ) + continue + lines.append( + f"| {model} | {quant} | {gpu} | " + f"{row['new_decode']:.1f} / {row['base_decode']:.1f} " + f"({row['delta_decode']:+.1f}%) | " + f"{row['new_prefill']:.1f} / {row['base_prefill']:.1f} " + f"({row['delta_prefill']:+.1f}%) | " + f"n={row['n']} | {row['status']} |" + ) + return "\n".join(lines) + + +def parse_args(argv: list[str] | None = None) -> argparse.Namespace: + """Parse CLI arguments.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--results-dir", required=True) + parser.add_argument("--repo", required=True, help="owner/repo") + parser.add_argument("--workflow", default="cuda.yml") + parser.add_argument("--branch", default="main") + parser.add_argument("--s3-bucket", default="gha-artifacts") + parser.add_argument("--s3-prefix", default="executorch-cuda-perf") + parser.add_argument("--token", default=os.environ.get("GITHUB_TOKEN", "")) + parser.add_argument("--pr-number", default=os.environ.get("PR_NUMBER", "")) + parser.add_argument("--bypass-label", default="bypass-perf-regression") + parser.add_argument("--exclude-run-id", default=os.environ.get("RUN_ID", "")) + parser.add_argument("--threshold-pct", type=float, default=5.0) + parser.add_argument("--baseline-window", type=int, default=5) + parser.add_argument("--min-history", type=int, default=3) + parser.add_argument("--summary-file", default=os.environ.get("GITHUB_STEP_SUMMARY", "")) + return parser.parse_args(argv) + + +def main(argv: list[str] | None = None) -> int: + """Entry point: bypass check, baseline compare, markdown report.""" + logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") + args = parse_args(argv) + + if args.pr_number and args.token: + try: + if has_bypass_label(args.repo, args.pr_number, args.bypass_label, args.token): + print( + f"::notice::Perf regression check bypassed by " + f"'{args.bypass_label}' label on PR #{args.pr_number}" + ) + return 0 + except Exception as e: + logger.warning("Bypass label lookup failed, continuing with check: %s", e) + + current = load_current_results(args.results_dir) + if not current: + print("::warning::No current benchmark results found; skipping regression check") + return 0 + + baselines = collect_baselines(args, set(current)) + rows = check_regressions(current, baselines, args.threshold_pct) + report = render_markdown(rows, args.threshold_pct) + print(report) + if args.summary_file: + with open(args.summary_file, "a") as f: + f.write(report + "\n") + + failures = [r for r in rows if r["status"] == "FAIL"] + for row in failures: + model, quant, gpu = row["key"] + print( + f"::error::Perf regression in {model} [{quant}] on {gpu}: " + f"decode {row['delta_decode']:+.1f}%, " + f"prefill {row['delta_prefill']:+.1f}% " + f"(threshold -{args.threshold_pct:.1f}%). " + f"Add the '{args.bypass_label}' label to bypass with justification." + ) + return 1 if failures else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/pytorch-probot.yml b/.github/pytorch-probot.yml index 5875441e0dc..d500739a532 100644 --- a/.github/pytorch-probot.yml +++ b/.github/pytorch-probot.yml @@ -5,7 +5,6 @@ ciflow_push_tags: - ciflow/apple - ciflow/coreai - ciflow/cuda -- ciflow/cuda-perf - ciflow/docker - ciflow/metal - ciflow/mlx diff --git a/.github/scripts/trigger_cuda_benchmark.sh b/.github/scripts/trigger_cuda_benchmark.sh new file mode 100755 index 00000000000..4f7330b19fb --- /dev/null +++ b/.github/scripts/trigger_cuda_benchmark.sh @@ -0,0 +1,81 @@ +#!/bin/bash +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +# Quick script to trigger the CUDA workflow (accuracy + benchmark + perf +# regression gate, merged from the retired cuda-perf.yml) via GitHub CLI. +# Benchmarks reuse each accuracy cell's exported .pte/.ptd in place; the +# models/quantizations filters below only narrow the benchmark stage. +# Usage: +# ./trigger_cuda_benchmark.sh # Use defaults (all allowlisted pairs) +# ./trigger_cuda_benchmark.sh "openai/whisper-medium" # Single model +# ./trigger_cuda_benchmark.sh "openai/whisper-small,google/gemma-3-4b-it" "non-quantized,quantized-int4-tile-packed" "100" +# +# NOTE: perf allowlist lives in .ci/scripts/cuda_benchmark_from_e2e.sh and the +# test-model-cuda-e2e matrix in .github/workflows/cuda.yml. Models outside the +# accuracy matrix cannot be benchmarked via this script. + +set -e + +# Models covered by both accuracy e2e and perf tracking (see +# cuda_benchmark_from_e2e.sh is_bench_pair). whisper-medium is the one +# perf-first model backfilled into the accuracy matrix. +ALL_MODELS="mistralai/Voxtral-Mini-3B-2507,openai/whisper-small,openai/whisper-medium,openai/whisper-large-v3-turbo,google/gemma-3-4b-it,nvidia/parakeet-tdt,SocialLocalMobile/Qwen3.5-35B-A3B-HQQ-INT4" +ALL_QUANTIZATIONS="non-quantized,quantized-int4-tile-packed,quantized-int4-weight-only" + +# Check if gh CLI is installed +if ! command -v gh &> /dev/null; then + echo "Error: GitHub CLI (gh) is not installed." + echo "Install it from: https://cli.github.com/" + echo "" + echo "Quick install:" + echo " macOS: brew install gh" + echo " Linux: See https://github.com/cli/cli/blob/trunk/docs/install_linux.md" + exit 1 +fi + +MODELS="${1:-}" +QUANT="${2:-}" +NUM_RUNS="${3:-50}" + +# Display configuration +echo "=========================================" +echo "Triggering cuda workflow (accuracy + benchmark)" +echo "=========================================" +if [ -z "$MODELS" ]; then + echo "Models: (all allowlisted: $ALL_MODELS)" +else + echo "Models: $MODELS" +fi +if [ -z "$QUANT" ]; then + echo "Quantizations: (all allowlisted: $ALL_QUANTIZATIONS)" +else + echo "Quantizations: $QUANT" +fi +echo "Num runs: $NUM_RUNS" +echo "=========================================" + +echo "" + +# Trigger workflow (dispatch inputs only narrow the benchmark stage) +gh workflow run cuda.yml \ + -R pytorch/executorch \ + -f models="$MODELS" \ + -f quantizations="$QUANT" \ + -f num_runs="$NUM_RUNS" + +if [ $? -eq 0 ]; then + echo "✓ Workflow triggered successfully!" + echo "" + echo "View status:" + echo " gh run list --workflow=cuda.yml" + echo "" + echo "Watch the latest run:" + echo " gh run watch \$(gh run list --workflow=cuda.yml --limit 1 --json databaseId --jq '.[0].databaseId')" +else + echo "✗ Failed to trigger workflow" + exit 1 +fi diff --git a/.github/scripts/trigger_cuda_perf.sh b/.github/scripts/trigger_cuda_perf.sh deleted file mode 100755 index 402dd009673..00000000000 --- a/.github/scripts/trigger_cuda_perf.sh +++ /dev/null @@ -1,100 +0,0 @@ -#!/bin/bash -# Copyright (c) Meta Platforms, Inc. and affiliates. -# All rights reserved. -# -# This source code is licensed under the BSD-style license found in the -# LICENSE file in the root directory of this source tree. - -# Quick script to trigger cuda-perf workflow via GitHub CLI -# Usage: -# ./trigger_cuda_perf.sh # Use defaults (random model + quant) -# ./trigger_cuda_perf.sh --all # Run ALL models with ALL quantizations -# ./trigger_cuda_perf.sh "openai/whisper-medium" # Single model -# ./trigger_cuda_perf.sh "openai/whisper-small,google/gemma-3-4b-it" "non-quantized,quantized-int4-tile-packed" "100" - -set -e - -# All available models and quantizations -ALL_MODELS="mistralai/Voxtral-Mini-3B-2507,openai/whisper-small,openai/whisper-medium,openai/whisper-large-v3-turbo,google/gemma-3-4b-it" -ALL_QUANTIZATIONS="non-quantized,quantized-int4-tile-packed,quantized-int4-weight-only" - -# Check if gh CLI is installed -if ! command -v gh &> /dev/null; then - echo "Error: GitHub CLI (gh) is not installed." - echo "Install it from: https://cli.github.com/" - echo "" - echo "Quick install:" - echo " macOS: brew install gh" - echo " Linux: See https://github.com/cli/cli/blob/trunk/docs/install_linux.md" - exit 1 -fi - -# Check for --all flag -RUN_ALL=false -if [ "${1:-}" = "--all" ] || [ "${1:-}" = "-a" ]; then - RUN_ALL=true - shift # Remove the flag from arguments -fi - -# Default parameters -if [ "$RUN_ALL" = true ]; then - MODELS="$ALL_MODELS" - QUANT="$ALL_QUANTIZATIONS" - NUM_RUNS="${1:-50}" - RANDOM_MODEL="false" - echo "=========================================" - echo "Triggering cuda-perf workflow" - echo "Mode: RUN ALL MODELS AND QUANTIZATIONS" - echo "=========================================" - echo "Models: ALL (5 models)" - echo "Quantizations: ALL (3 quantizations)" - echo "Total configs: 15 combinations" - echo "Num runs: $NUM_RUNS" - echo "=========================================" -else - MODELS="${1:-}" - QUANT="${2:-}" - NUM_RUNS="${3:-50}" - RANDOM_MODEL="${4:-false}" - - # Display configuration - echo "=========================================" - echo "Triggering cuda-perf workflow" - echo "=========================================" - if [ -z "$MODELS" ]; then - echo "Models: (random selection)" - else - echo "Models: $MODELS" - fi - if [ -z "$QUANT" ]; then - echo "Quantizations: (random selection)" - else - echo "Quantizations: $QUANT" - fi - echo "Num runs: $NUM_RUNS" - echo "Random model: $RANDOM_MODEL" - echo "=========================================" -fi - -echo "" - -# Trigger workflow -gh workflow run cuda-perf.yml \ - -R pytorch/executorch \ - -f models="$MODELS" \ - -f quantizations="$QUANT" \ - -f num_runs="$NUM_RUNS" \ - -f random_model="$RANDOM_MODEL" - -if [ $? -eq 0 ]; then - echo "✓ Workflow triggered successfully!" - echo "" - echo "View status:" - echo " gh run list --workflow=cuda-perf.yml" - echo "" - echo "Watch the latest run:" - echo " gh run watch \$(gh run list --workflow=cuda-perf.yml --limit 1 --json databaseId --jq '.[0].databaseId')" -else - echo "✗ Failed to trigger workflow" - exit 1 -fi diff --git a/.github/workflows/_ci-run-decision.yml b/.github/workflows/_ci-run-decision.yml index 86c2d25c515..34f3609f093 100644 --- a/.github/workflows/_ci-run-decision.yml +++ b/.github/workflows/_ci-run-decision.yml @@ -22,11 +22,11 @@ name: CI Run Decision # workflows list ``ciflow/trunk/*`` under ``on.push.tags`` and so re-run # from a promotion tag: pull, trunk, apple, windows-msvc, riscv64, # qnn-windows-msvc and viable-strict-gate. Only four of those call this -# workflow. The other seven callers (cuda, cuda-windows, cuda-perf, +# workflow. The other six callers (cuda, cuda-windows, # metal, mlx, rocm, vulkan) do have ciflow tags, but none of them listens # for ``ciflow/trunk``, so a promotion tag never starts them, and they -# hold 32 of the 47 jobs whose ``if:`` gates on ``is-full-run``. Most of -# those 32 also have a changed-files branch and still run on main when +# hold 32 of the 47 jobs whose ``if:`` gates on ``is-full-run``. (cuda-perf +# was merged into cuda, so it is no longer a separate caller. Most of # their own paths are touched; the 12 in mlx.yml do not, so for those the # depth sample below is the only thing that runs them automatically on a # push to main. (``workflow_dispatch`` and ``ciflow/mlx`` can still force diff --git a/.github/workflows/cuda-perf.yml b/.github/workflows/cuda-perf.yml deleted file mode 100644 index d19fc555302..00000000000 --- a/.github/workflows/cuda-perf.yml +++ /dev/null @@ -1,407 +0,0 @@ -name: cuda-perf - -on: - push: - branches: - - main - - release/* - tags: - - ciflow/cuda-perf/* - pull_request: - paths: - - .github/workflows/cuda-perf.yml - - .ci/scripts/cuda_benchmark.py - - .ci/scripts/cuda_perf_prompts/** - - .ci/scripts/export_model_artifact.sh - - .ci/scripts/test_model_e2e.sh - workflow_dispatch: - inputs: - models: - description: Models to be benchmarked (comma-separated HuggingFace model IDs) - required: false - type: string - quantizations: - description: Quantization types (comma-separated) - required: false - type: string - num_runs: - description: Number of benchmark runs per model - required: false - type: string - default: "50" - -concurrency: - group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref_name }}-${{ github.ref_type == 'branch' && github.sha }}-${{ github.event_name == 'workflow_dispatch' }} - cancel-in-progress: true - -permissions: - contents: read - -jobs: - changed-files: - name: Get changed files - uses: ./.github/workflows/_get-changed-files.yml - with: - include-push-diff: true - - run-decision: - name: CI run decision - uses: ./.github/workflows/_ci-run-decision.yml - - set-parameters: - needs: [changed-files, run-decision] - # Path-filtered: mirrors the workflow-level pull_request `paths:` - # filter so push commits that don't touch perf-relevant paths skip - # this whole workflow on non-sampled commits. Sampling preserves - # perf time-series at every 4th commit (vs every commit pre-PR). - if: | - contains(needs.changed-files.outputs.changed-files, '.github/workflows/cuda-perf.yml') || - contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_benchmark.py') || - contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_perf_prompts') || - contains(needs.changed-files.outputs.changed-files, '.ci/scripts/export_model_artifact.sh') || - contains(needs.changed-files.outputs.changed-files, '.ci/scripts/test_model_e2e.sh') || - needs.run-decision.outputs.is-full-run == 'true' - runs-on: ubuntu-22.04 - outputs: - benchmark_configs: ${{ steps.set-parameters.outputs.benchmark_configs }} - steps: - - uses: actions/checkout@v3 - with: - submodules: 'false' - - uses: actions/setup-python@v4 - with: - python-version: '3.10' - - name: Set parameters - id: set-parameters - shell: bash - env: - ALL_MODELS: 'mistralai/Voxtral-Mini-3B-2507,openai/whisper-small,openai/whisper-medium,openai/whisper-large-v3-turbo,google/gemma-3-4b-it,nvidia/parakeet-tdt,SocialLocalMobile/Qwen3.5-35B-A3B-HQQ-INT4' - ALL_QUANTIZATIONS: 'non-quantized,quantized-int4-tile-packed,quantized-int4-weight-only' - NUM_RUNS: ${{ inputs.num_runs || '50' }} - run: | - set -eux - - MODELS="${{ inputs.models }}" - QUANTIZATIONS="${{ inputs.quantizations }}" - - # Use all models/quantizations unless overridden by workflow_dispatch - if [ -z "$MODELS" ]; then - MODELS="$ALL_MODELS" - fi - if [ -z "$QUANTIZATIONS" ]; then - QUANTIZATIONS="$ALL_QUANTIZATIONS" - fi - - # Split models and quantizations into arrays - IFS=',' read -ra MODEL_ARRAY <<< "$MODELS" - IFS=',' read -ra QUANT_ARRAY <<< "$QUANTIZATIONS" - - # Generate benchmark configs (skip invalid model/quant combinations) - CONFIGS='{"include":[' - FIRST=true - for MODEL in "${MODEL_ARRAY[@]}"; do - for QUANT in "${QUANT_ARRAY[@]}"; do - # Qwen3.5 MoE only supports quantized-int4-tile-packed - if [[ "$MODEL" == *"Qwen3.5-35B-A3B"* ]] && [ "$QUANT" != "quantized-int4-tile-packed" ]; then - continue - fi - if [ "$FIRST" = true ]; then - FIRST=false - else - CONFIGS+=',' - fi - # Sanitize model name for use in artifact paths - MODEL_SAFE=$(echo "$MODEL" | sed 's/\//_/g') - CONFIGS+="{\"model\":\"$MODEL\",\"quant\":\"$QUANT\",\"model_safe\":\"$MODEL_SAFE\",\"num_runs\":\"$NUM_RUNS\"}" - done - done - CONFIGS+=']}' - - echo "benchmark_configs=$CONFIGS" >> $GITHUB_OUTPUT - echo "Generated benchmark configs:" - echo "$CONFIGS" | python -m json.tool - - benchmark-cuda: - name: benchmark-cuda - needs: - - changed-files - - run-decision - - set-parameters - # Same gate as set-parameters, repeated so the job reads on its own; - # the needs on set-parameters already skips it when the gate is closed. - if: | - contains(needs.changed-files.outputs.changed-files, '.github/workflows/cuda-perf.yml') || - contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_benchmark.py') || - contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_perf_prompts') || - contains(needs.changed-files.outputs.changed-files, '.ci/scripts/export_model_artifact.sh') || - contains(needs.changed-files.outputs.changed-files, '.ci/scripts/test_model_e2e.sh') || - needs.run-decision.outputs.is-full-run == 'true' - uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main - permissions: - id-token: write - contents: read - secrets: inherit - strategy: - matrix: ${{ fromJson(needs.set-parameters.outputs.benchmark_configs) }} - fail-fast: false - with: - timeout: 120 - secrets-env: EXECUTORCH_HF_TOKEN - runner: ${{ contains(matrix.model, 'Qwen3.5-35B-A3B') && 'mt-l-x86iavx512-11-125-a100' || 'mt-l-x86aavx2-11-41-a10g' }} - gpu-arch-type: cuda - gpu-arch-version: "13.0" - submodules: recursive - upload-artifact: results-${{ matrix.model_safe }}-${{ matrix.quant }} - ref: ${{ github.event_name == 'pull_request' && github.event.pull_request.head.sha || github.sha }} - script: | - set -eux - - # OSDC mounts HF_HOME read-only at /mnt/hf_cache; redirect to a writable dir - # (RUNNER_TEMP, or /tmp when RUNNER_TEMP isn't writable inside the container). - export HF_HOME="${RUNNER_TEMP:-/tmp}/hf_cache" - mkdir -p "${HF_HOME}" 2>/dev/null || export HF_HOME=/tmp/hf_cache - mkdir -p "${HF_HOME}" - - echo "::group::Setup ExecuTorch" - # OSDC runners can't reach the public PyPI CDN that download.pytorch.org's - # transitive deps resolve to. Pre-install torch's pure-python deps from the - # in-cluster pypi-cache and drop the default cpu extra-index so the cuda - # torch wheel is the only candidate. - export PIP_EXTRA_INDEX_URL= - # fsspec is pinned to satisfy datasets' fsspec[http]<=2025.3.0 so the later - # examples install doesn't try to downgrade it from the public CDN. - pip install filelock typing-extensions "setuptools<82" sympy networkx jinja2 "fsspec[http]<=2025.3.0" numpy pillow - # Disable MKL to avoid duplicate target error when conda has multiple MKL installations - export USE_MKL=OFF - ./install_executorch.sh - echo "::endgroup::" - - echo "::group::Setup Huggingface" - pip install -U "huggingface_hub[cli]>=1.2.1,<2.0" accelerate "optimum~=2.0.0" "transformers==5.0.0rc1" - export HF_TOKEN="$(printf '%s' "$SECRET_EXECUTORCH_HF_TOKEN" | tr -d '\r\n')" - OPTIMUM_ET_VERSION=$(cat .ci/docker/ci_commit_pins/optimum-executorch.txt) - pip install --no-deps git+https://github.com/huggingface/optimum-executorch.git@${OPTIMUM_ET_VERSION} - echo "::endgroup::" - - echo "::group::Export model and build runner" - # Export and benchmark in the same job so the model never leaves the runner - # (no GitHub Actions artifact upload/download). - RUN_EXPORT=1 bash .ci/scripts/test_model_e2e.sh cuda "${{ matrix.model }}" "${{ matrix.quant }}" model_artifacts - echo "::endgroup::" - - echo "::group::Running benchmark for ${{ matrix.model }} (${{ matrix.quant }}) with ${{ matrix.num_runs }} runs" - export LD_LIBRARY_PATH=/opt/conda/lib:$LD_LIBRARY_PATH - - # Get GPU name using nvidia-smi - GPU_NAME=$(nvidia-smi --query-gpu=name --format=csv,noheader | head -1) - echo "Detected GPU: $GPU_NAME" - - # Get CUDA driver version - CUDA_DRIVER_VERSION=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader | head -1) - echo "CUDA Driver Version: $CUDA_DRIVER_VERSION" - - # Create results directory (separate from model artifacts) - RESULTS_DIR="benchmark_results" - mkdir -p "$RESULTS_DIR" - - # Determine model name and runner command based on model - case "${{ matrix.model }}" in - mistralai/Voxtral-Mini-3B-2507) - RUNNER="cmake-out/examples/models/voxtral/voxtral_runner" - PREPROCESSOR="model_artifacts/voxtral_preprocessor.pte" - TOKENIZER="model_artifacts/tekken.json" - AUDIO="model_artifacts/poem.wav" - RUNNER_CMD="$RUNNER --model_path model_artifacts/model.pte --data_path model_artifacts/aoti_cuda_blob.ptd --tokenizer_path $TOKENIZER --audio_path $AUDIO --processor_path $PREPROCESSOR --temperature 0" - MODEL_NAME="voxtral_${{ matrix.quant }}" - ;; - openai/whisper-*) - RUNNER="cmake-out/examples/models/whisper/whisper_runner" - PREPROCESSOR="model_artifacts/whisper_preprocessor.pte" - AUDIO="model_artifacts/output.wav" - RUNNER_CMD="$RUNNER --model_path model_artifacts/model.pte --data_path model_artifacts/aoti_cuda_blob.ptd --tokenizer_path model_artifacts/ --audio_path $AUDIO --processor_path $PREPROCESSOR --temperature 0" - MODEL_NAME=$(echo "${{ matrix.model }}" | sed 's/openai\///')_${{ matrix.quant }} - ;; - google/gemma-3-4b-it) - RUNNER="cmake-out/examples/models/gemma3/gemma3_e2e_runner" - IMAGE="docs/source/_static/img/et-logo.png" - RUNNER_CMD="$RUNNER --model_path model_artifacts/model.pte --data_path model_artifacts/aoti_cuda_blob.ptd --tokenizer_path model_artifacts/ --image_path $IMAGE --temperature 0" - MODEL_NAME="gemma3_${{ matrix.quant }}" - ;; - nvidia/parakeet-tdt) - RUNNER="cmake-out/examples/models/parakeet/parakeet_runner" - AUDIO="model_artifacts/test_audio.wav" - TOKENIZER="model_artifacts/tokenizer.model" - RUNNER_CMD="$RUNNER --model_path model_artifacts/model.pte --data_path model_artifacts/aoti_cuda_blob.ptd --audio_path $AUDIO --tokenizer_path $TOKENIZER" - MODEL_NAME="parakeet_${{ matrix.quant }}" - ;; - SocialLocalMobile/Qwen3.5-35B-A3B-HQQ-INT4) - RUNNER="cmake-out/examples/models/qwen3_5_moe/qwen3_5_moe_runner" - TOKENIZER="model_artifacts/tokenizer.json" - # Use a checked-in long prompt (>1000 tokens) for benchmarking. A - # static, meaningful prompt avoids the degenerate / repetitive - # outputs that can result from synthetic prompts built by - # repeating the same sentence. - PROMPT_FILE=".ci/scripts/cuda_perf_prompts/qwen3_5_moe_long_prompt.txt" - RUNNER_CMD="$RUNNER --model_path model_artifacts/model.pte --data_path model_artifacts/aoti_cuda_blob.ptd --tokenizer_path $TOKENIZER --prompt_file $PROMPT_FILE --max_new_tokens 512 --temperature 0" - MODEL_NAME="qwen3_5_moe_${{ matrix.quant }}" - ;; - *) - echo "Error: Unsupported model '${{ matrix.model }}'" - exit 1 - ;; - esac - - # Run benchmark using cuda_benchmark.py - python .ci/scripts/cuda_benchmark.py \ - --runner_command "$RUNNER_CMD" \ - --model_name "$MODEL_NAME" \ - --num_runs "${{ matrix.num_runs }}" \ - --output_json "$RESULTS_DIR/benchmark_results.json" \ - --output_v3 "$RESULTS_DIR/benchmark_results_v3.json" \ - --model "${{ matrix.model }}" \ - --quantization "${{ matrix.quant }}" \ - --git_sha "${{ github.sha }}" \ - --workflow_run_id "${{ github.run_id }}" \ - --workflow_run_url "https://github.com/${{ github.repository }}/actions/runs/${{ github.run_id }}" \ - --gpu_name "$GPU_NAME" \ - --cuda_driver_version "$CUDA_DRIVER_VERSION" - - # Save additional metadata - cat > "$RESULTS_DIR/metadata.json" <5% regression check (bypass with the `bypass-perf-regression` +# PR label). The retired cuda-perf.yml was merged into this workflow. +# # Intentionally skipped CUDA version 13.2 check due to ci image unsupported. # # Note: ExecuTorch automatically detects the system CUDA version using nvcc and @@ -26,10 +32,28 @@ on: - .ci/scripts/test-cuda-build.sh - .ci/scripts/export_model_artifact.sh - .ci/scripts/test_model_e2e.sh + - .ci/scripts/cuda_benchmark.py + - .ci/scripts/cuda_benchmark_from_e2e.sh + - .ci/scripts/cuda_check_regression.py + - .ci/scripts/cuda_perf_prompts/** - examples/models/muse-glimmer/** - extension/pybindings/** - runtime/__init__.py workflow_dispatch: + inputs: + models: + description: Models to be benchmarked (comma-separated HuggingFace model IDs, empty = all allowlisted) + required: false + type: string + quantizations: + description: Quantizations to benchmark (comma-separated, empty = all allowlisted) + required: false + type: string + num_runs: + description: Number of benchmark runs per model + required: false + type: string + default: "50" concurrency: group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref_name }}-${{ github.ref_type == 'branch' && github.sha }}-${{ github.event_name == 'workflow_dispatch' }}-${{ github.event_name == 'schedule' }} @@ -539,6 +563,10 @@ jobs: contains(needs.changed-files.outputs.changed-files, '.ci/scripts/test-cuda-build.sh') || contains(needs.changed-files.outputs.changed-files, '.ci/scripts/export_model_artifact.sh') || contains(needs.changed-files.outputs.changed-files, '.ci/scripts/test_model_e2e.sh') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_benchmark.py') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_benchmark_from_e2e.sh') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_check_regression.py') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_perf_prompts') || needs.run-decision.outputs.is-full-run == 'true' ) uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main @@ -558,6 +586,8 @@ jobs: name: "diar_streaming_sortformer_4spk-v2" - repo: "openai" name: "whisper-small" + - repo: "openai" + name: "whisper-medium" - repo: "openai" name: "whisper-large-v3-turbo" - repo: "google" @@ -638,6 +668,7 @@ jobs: quant: "quantized-int4-weight-only" # Whisper: cover all quant types and both sizes with minimal combos # whisper-small: non-quantized only + # whisper-medium: non-quantized only (perf tracking model) # whisper-large-v3-turbo: int4-tile-packed + int4-weight-only - model: repo: "openai" @@ -647,6 +678,14 @@ jobs: repo: "openai" name: "whisper-small" quant: "quantized-int4-weight-only" + - model: + repo: "openai" + name: "whisper-medium" + quant: "quantized-int4-tile-packed" + - model: + repo: "openai" + name: "whisper-medium" + quant: "quantized-int4-weight-only" - model: repo: "openai" name: "whisper-large-v3-turbo" @@ -677,6 +716,7 @@ jobs: gpu-arch-type: cuda gpu-arch-version: "13.0" submodules: recursive + upload-artifact: cuda-bench-${{ matrix.model.repo }}_${{ matrix.model.name }}-${{ matrix.quant }} ref: ${{ github.event_name == 'pull_request' && github.event.pull_request.head.sha || github.sha }} script: | set -eux @@ -717,6 +757,34 @@ jobs: "${{ matrix.quant }}" \ "${MODEL_DIR}" + # Accuracy passed. Benchmark the same exported artifacts in place + # (helper skips non-allowlisted pairs, e.g. sortformer, dinov2). + BENCH_RESULTS_DIR="${RUNNER_ARTIFACT_DIR}/bench" + mkdir -p "${BENCH_RESULTS_DIR}" + if bash .ci/scripts/cuda_benchmark_from_e2e.sh \ + "${{ matrix.model.repo }}/${{ matrix.model.name }}" \ + "${{ matrix.quant }}" \ + "${MODEL_DIR}" \ + "${BENCH_RESULTS_DIR}" \ + "${{ inputs.num_runs || '50' }}" \ + "${{ github.sha }}" \ + "${{ github.run_id }}" \ + https://github.com/${{ github.repository }}/actions/runs/${{ github.run_id }}" \ + "${{ inputs.models }}" \ + "${{ inputs.quantizations }}"; then + if [ -f "${BENCH_RESULTS_DIR}/benchmark_results.json" ]; then + echo "Benchmark results will be uploaded via the job artifact." + else + echo "Benchmark skipped for this (model, quant) pair." + # Keep the artifact non-empty so the aggregation jobs have a + # stable download layout; skipped cells are ignored downstream. + echo '{"skipped": true}' > "${BENCH_RESULTS_DIR}/benchmark_skipped.json" + fi + else + echo "::error::Benchmark failed for ${{ matrix.model.repo }}/${{ matrix.model.name }} (${{ matrix.quant }})" + exit 1 + fi + if [ -n "${{ matrix.pybind_model }}" ]; then echo "::group::Run CUDA model with pybind" conda install -y -c conda-forge 'libstdcxx-ng>=12' @@ -736,6 +804,206 @@ jobs: echo "::endgroup::" fi + upload-benchmark-results: + name: upload-benchmark-results + # Same gate as test-model-cuda-e2e: only aggregate when e2e itself ran. + # `always()` keeps uploading successful cells when some cells fail. + needs: [changed-files, run-decision, test-model-cuda-e2e] + if: | + always() && + needs.test-model-cuda-e2e.result != 'skipped' && + needs.test-model-cuda-e2e.result != 'cancelled' && + ( + contains(needs.changed-files.outputs.changed-files, 'backends/cuda') || + contains(needs.changed-files.outputs.changed-files, 'backends/aoti') || + contains(needs.changed-files.outputs.changed-files, '.github/workflows/cuda.yml') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_benchmark.py') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_benchmark_from_e2e.sh') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_check_regression.py') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_perf_prompts') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/test-cuda-build.sh') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/export_model_artifact.sh') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/test_model_e2e.sh') || + needs.run-decision.outputs.is-full-run == 'true' + ) + runs-on: ubuntu-22.04 + environment: upload-benchmark-results + permissions: + id-token: write + contents: read + steps: + - uses: actions/checkout@v3 + with: + submodules: false + + - name: Setup Python + uses: actions/setup-python@v4 + with: + python-version: '3.10' + + - name: Download all benchmark results + uses: actions/download-artifact@v4 + with: + pattern: cuda-bench-* + path: all_results/ + + - name: Process and display results + shell: bash + run: | + set -eux + echo "::group::Benchmark Results Summary" + + for RESULT_JSON in all_results/cuda-bench-*/bench/benchmark_results.json; do + [ -e "$RESULT_JSON" ] || continue + RESULT_DIR="$(dirname "$(dirname "$RESULT_JSON")")" + BENCH_DIR="$(dirname "$RESULT_JSON")" + echo "" + echo "================================" + echo "Results from: $(basename "$RESULT_DIR")" + echo "================================" + + # Display benchmark results (mean performance) + cat "$RESULT_JSON" | python -m json.tool + + # Display metadata + if [ -f "$BENCH_DIR/metadata.json" ]; then + echo "" + echo "--- Metadata ---" + cat "$BENCH_DIR/metadata.json" | python -m json.tool + fi + echo "" + done + + for SKIP_JSON in all_results/cuda-bench-*/bench/benchmark_skipped.json; do + [ -e "$SKIP_JSON" ] || continue + echo "Skipped: $(basename "$(dirname "$(dirname "$SKIP_JSON")")") (not in perf allowlist)" + done + + echo "::endgroup::" + + - name: Authenticate with AWS + uses: aws-actions/configure-aws-credentials@v4 + with: + role-to-assume: arn:aws:iam::308535385114:role/gha_workflow_upload-benchmark-results + role-duration-seconds: 18000 + aws-region: us-east-1 + + - name: Upload to S3 + shell: bash + env: + S3_BUCKET: gha-artifacts + S3_PREFIX: executorch-cuda-perf/${{ github.run_id }}/${{ github.run_attempt }} + run: | + set -eux + pip install awscli + + echo "Uploading benchmark results to S3..." + aws s3 sync all_results/ "s3://${S3_BUCKET}/${S3_PREFIX}/" \ + --exclude "*" \ + --include "*.json" \ + --include "*.log" + + echo "Results uploaded to: s3://${S3_BUCKET}/${S3_PREFIX}/" + + - name: Prepare v3 results for dashboard upload + shell: bash + run: | + set -eux + echo "::group::Prepare v3 results" + + mkdir -p benchmark-results/v3 + + # Collect all v3 results into a single directory + for V3_JSON in all_results/cuda-bench-*/bench/benchmark_results_v3.json; do + [ -e "$V3_JSON" ] || continue + # Generate unique filename based on directory name + FILENAME=$(basename "$(dirname "$(dirname "$V3_JSON")")") + cp "$V3_JSON" "benchmark-results/v3/${FILENAME}.json" + echo "✓ Copied $FILENAME v3 results" + done + + echo "V3 results prepared:" + ls -lah benchmark-results/v3/ + echo "::endgroup::" + + - name: Upload benchmark results to dashboard + uses: pytorch/test-infra/.github/actions/upload-benchmark-results@main + with: + benchmark-results-dir: benchmark-results/v3 + dry-run: false + schema-version: v3 + github-token: ${{ secrets.GITHUB_TOKEN }} + + check-perf-regression: + name: check-perf-regression + # PR-blocking gate: fails when decode or prefill throughput drops more + # than 5% vs the median of recent successful main runs (same model, + # quant, GPU). Bypass by adding the `bypass-perf-regression` label. + needs: [changed-files, run-decision, upload-benchmark-results] + if: | + always() && + needs.upload-benchmark-results.result == 'success' && + ( + contains(needs.changed-files.outputs.changed-files, 'backends/cuda') || + contains(needs.changed-files.outputs.changed-files, 'backends/aoti') || + contains(needs.changed-files.outputs.changed-files, '.github/workflows/cuda.yml') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_benchmark.py') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_benchmark_from_e2e.sh') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_check_regression.py') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/cuda_perf_prompts') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/test-cuda-build.sh') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/export_model_artifact.sh') || + contains(needs.changed-files.outputs.changed-files, '.ci/scripts/test_model_e2e.sh') || + needs.run-decision.outputs.is-full-run == 'true' + ) + runs-on: ubuntu-22.04 + permissions: + id-token: write + contents: read + pull-requests: read + steps: + - uses: actions/checkout@v3 + with: + submodules: false + + - name: Setup Python + uses: actions/setup-python@v4 + with: + python-version: '3.10' + + - name: Download all benchmark results + uses: actions/download-artifact@v4 + with: + pattern: cuda-bench-* + path: all_results/ + + - name: Authenticate with AWS + uses: aws-actions/configure-aws-credentials@v4 + with: + role-to-assume: arn:aws:iam::308535385114:role/gha_workflow_upload-benchmark-results + role-duration-seconds: 18000 + aws-region: us-east-1 + + - name: Check for perf regressions + shell: bash + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + PR_NUMBER: ${{ github.event.pull_request.number }} + RUN_ID: ${{ github.run_id }} + run: | + set -eux + pip install awscli + python .ci/scripts/cuda_check_regression.py \ + --results-dir all_results \ + --repo "${{ github.repository }}" \ + --workflow cuda.yml \ + --branch main \ + --s3-bucket gha-artifacts \ + --s3-prefix executorch-cuda-perf \ + --threshold-pct 5.0 \ + --baseline-window 5 \ + --min-history 3 + test-muse-glimmer-cuda-e2e: name: test-muse-glimmer-cuda-e2e-${{ matrix.variant }}-${{ matrix.mode }} needs: [changed-files, run-decision]