diff --git a/.ci/scripts/install_nvidia_driver_windows.ps1 b/.ci/scripts/install_nvidia_driver_windows.ps1 new file mode 100644 index 00000000000..5b18ceac284 --- /dev/null +++ b/.ci/scripts/install_nvidia_driver_windows.ps1 @@ -0,0 +1,28 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +# Installs the NVIDIA display driver on a Windows GPU runner. The runner images carry the +# GPU but no driver, so without this nvcuda.dll is missing and every CUDA call fails with +# cudaErrorInsufficientDriver. The same driver, from the same place, that PyTorch's Windows +# CUDA smoke tests install first (.ci/pytorch/windows/internal/driver_update.bat in +# pytorch/pytorch), so both projects test against one driver. +$ErrorActionPreference = "Stop" +$version = "580.88" +$installer = Join-Path $env:RUNNER_TEMP "$version-data-center-tesla-desktop-win10-win11-64bit-dch-international.exe" +$url = "https://ossci-windows.s3.amazonaws.com/$(Split-Path $installer -Leaf)" + +$nvcuda = Join-Path $env:SystemRoot "System32\nvcuda.dll" +if (Test-Path $nvcuda) { + Write-Host "NVIDIA driver already present: $((Get-Item $nvcuda).VersionInfo.FileVersion)" + exit 0 +} +curl.exe --retry 3 -fsSL $url --output $installer +if ($LASTEXITCODE -ne 0) { throw "downloading the NVIDIA driver from $url failed" } +$process = Start-Process -FilePath $installer -ArgumentList "-s", "-noreboot" -Wait -PassThru +Remove-Item $installer -ErrorAction SilentlyContinue +if ($process.ExitCode -ne 0) { throw "the NVIDIA driver installer exited with $($process.ExitCode)" } +if (-not (Test-Path $nvcuda)) { throw "the NVIDIA driver installed but $nvcuda is missing" } +Write-Host "NVIDIA driver $version installed: $((Get-Item $nvcuda).VersionInfo.FileVersion)" diff --git a/.ci/scripts/wheel/install_cuda_redist.py b/.ci/scripts/wheel/install_cuda_redist.py new file mode 100644 index 00000000000..ce4121cc4ed --- /dev/null +++ b/.ci/scripts/wheel/install_cuda_redist.py @@ -0,0 +1,127 @@ +#!/usr/bin/env python +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +"""Assemble part of a CUDA toolkit from NVIDIA's redistributable component archives. + +The Windows CUDA wheel end-to-end jobs cover several CUDA trains while the CI images carry +one toolkit, and NVIDIA publishes no full Windows installer for every release. The redist +archives exist for every release on both platforms, need no installer or admin rights, and +are checksummed in a JSON index. Merging the components a job needs into one directory +gives the layout nvcc, CMake's FindCUDAToolkit and torch's cross-compile step expect. + + install_cuda_redist.py --train 13.4 --platform linux-x86_64 --dest /opt/cuda-13.4 + install_cuda_redist.py --train 13.4 --platform windows-x86_64 --dest DIR --components cuda_cudart +""" + +import argparse +import hashlib +import json +import shutil +import sys +import tarfile +import tempfile +import urllib.request +import zipfile +from pathlib import Path + +_REDIST = "https://developer.download.nvidia.com/compute/cuda/redist" + +# The patch release each train resolves to. One entry per train in filter_cuda_matrix.py's +# SUPPORTED_CUDA_VERSIONS, which the end-to-end jobs take their matrix from; a train missing +# here fails the job at argument parsing rather than testing a different release. +_RELEASES = {"13.0": "13.0.2", "13.2": "13.2.2", "13.4": "13.4.1"} + +# What a build of the CUDA backend needs. "a|b" names one component a release publishes +# under either name (CCCL was renamed between releases). +_BUILD_COMPONENTS = [ + "cuda_nvcc", + "cuda_cudart", + "cuda_crt", + "cccl|cuda_cccl", + "libnvvm", + "cuda_cuobjdump", + "cuda_nvrtc", + "cuda_profiler_api", + "cuda_nvtx", + # rand.cu includes curand_kernel.h. + "libcurand", +] + + +def _fetch(url: str, destination: Path) -> None: + with urllib.request.urlopen(url, timeout=300) as response, open( + destination, "wb" + ) as handle: + shutil.copyfileobj(response, handle) + + +def _extract(archive: Path, into: Path) -> Path: + if archive.suffix == ".zip": + with zipfile.ZipFile(archive) as bundle: + bundle.extractall(into) + else: + with tarfile.open(archive) as bundle: + if hasattr(tarfile, "fully_trusted_filter"): + bundle.extractall(into, filter="fully_trusted") + else: + bundle.extractall(into) + # Each archive holds one top-level directory named after itself. + (top,) = [entry for entry in into.iterdir() if entry.is_dir()] + return top + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + parser.add_argument("--train", required=True, choices=sorted(_RELEASES)) + parser.add_argument( + "--platform", required=True, choices=["linux-x86_64", "windows-x86_64"] + ) + parser.add_argument("--dest", required=True, type=Path) + parser.add_argument("--components", nargs="+", default=_BUILD_COMPONENTS) + arguments = parser.parse_args() + + release = _RELEASES[arguments.train] + with urllib.request.urlopen( + f"{_REDIST}/redistrib_{release}.json", timeout=120 + ) as response: + index = json.load(response) + arguments.dest.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory() as work: + work = Path(work) + for choice in arguments.components: + names = [ + name + for name in choice.split("|") + if arguments.platform in index.get(name, {}) + ] + if not names: + sys.exit( + f"CUDA {release} publishes none of {choice} for {arguments.platform}" + ) + entry = index[names[0]][arguments.platform] + archive = work / Path(entry["relative_path"]).name + _fetch(f"{_REDIST}/{entry['relative_path']}", archive) + digest = hashlib.sha256(archive.read_bytes()).hexdigest() + if digest != entry["sha256"]: + sys.exit(f"checksum mismatch for {archive.name}") + staging = work / "staging" + top = _extract(archive, staging) + shutil.copytree(top, arguments.dest, symlinks=True, dirs_exist_ok=True) + shutil.rmtree(staging) + archive.unlink() + print( + f"merged {names[0]} {index[names[0]]['version']} into {arguments.dest}" + ) + # The Linux archives keep libraries in lib/, while nvcc's link step and CMake's + # FindCUDAToolkit look in lib64/, which is where an installed toolkit has them. + lib, lib64 = arguments.dest / "lib", arguments.dest / "lib64" + if arguments.platform == "linux-x86_64" and lib.is_dir() and not lib64.exists(): + lib64.symlink_to("lib", target_is_directory=True) + + +if __name__ == "__main__": + main() diff --git a/.ci/scripts/wheel/pre_build_script.sh b/.ci/scripts/wheel/pre_build_script.sh index ae184cff309..0629d3afaa8 100755 --- a/.ci/scripts/wheel/pre_build_script.sh +++ b/.ci/scripts/wheel/pre_build_script.sh @@ -116,6 +116,23 @@ else export CMAKE_ARGS="${CMAKE_ARGS:-} -DEXECUTORCH_BUILD_CUDA=ON" echo "CMAKE_ARGS=${CMAKE_ARGS}" >> "${GITHUB_ENV}" echo "row '${CU_VERSION:-${DESIRED_CUDA:-}}' is a CUDA row, requiring the CUDA build" + # The Linux rows take the row's GPU list from envvar_cuda_linux.sh. Windows has no env-var + # script slot, so the same list is resolved here; without it the build compiles device + # code only for whatever GPU the builder has. The reusable workflow writes its own list + # into BUILD_ENV_FILE, which the build step sources after this hook and which would + # otherwise win, so the row's list is appended there as well as to GITHUB_ENV. + # Also needed for the ninja the Windows CUDA build generates with. + if [[ $UNAME_S == *"MINGW"* || $UNAME_S == *"MSYS"* ]]; then + source "${GITHUB_WORKSPACE}/${REPOSITORY}/.ci/scripts/wheel/cuda_arch_list.sh" + TORCH_CUDA_ARCH_LIST="$(EXECUTORCH_BUILD_CUDA=1 executorch_cuda_arch_list)" + export TORCH_CUDA_ARCH_LIST + echo "TORCH_CUDA_ARCH_LIST=${TORCH_CUDA_ARCH_LIST}" >> "${GITHUB_ENV}" + if [[ -n "${BUILD_ENV_FILE:-}" ]]; then + echo "export TORCH_CUDA_ARCH_LIST='${TORCH_CUDA_ARCH_LIST}'" >> "${BUILD_ENV_FILE}" + fi + echo "building device code for: ${TORCH_CUDA_ARCH_LIST}" + pip install --quiet ninja + fi fi # On Windows, enable symlinks and re-checkout the current revision to create @@ -125,14 +142,6 @@ if [[ $UNAME_S == *"MINGW"* || $UNAME_S == *"MSYS"* ]]; then git config core.symlinks true git checkout -f HEAD - # Windows wheels are CPU-only (build-wheels-windows.yml sets - # with-cuda: disabled), but the Windows CI image ships a CUDA toolkit on - # PATH, which makes setup.py auto-enable EXECUTORCH_BUILD_CUDA. That bakes a - # CUDA _C into the CPU wheel, which then fails its DLL load in the - # smoke test ("DLL load failed while importing _C"). Force a - # CPU-only build. - export CMAKE_ARGS="${CMAKE_ARGS:-} -DEXECUTORCH_BUILD_CUDA=OFF" - echo "CMAKE_ARGS=${CMAKE_ARGS}" >> "${GITHUB_ENV}" fi # Manually install build requirements because `python setup.py bdist_wheel` does @@ -202,7 +211,8 @@ if [[ "${EXECUTORCH_BUILD_VULKAN:-0}" != "0" \ echo "glslc installed: $(command -v glslc)" fi else - # This is the Vulkan equivalent of the Windows CUDA force-off above (#20527). + # Off unless asked for: a builder with the Vulkan SDK would otherwise turn it on + # for a wheel that does not ship it, as a CUDA toolkit once did for Windows (#20527). export CMAKE_ARGS="${CMAKE_ARGS:-} -DEXECUTORCH_BUILD_VULKAN=OFF" echo "CMAKE_ARGS=${CMAKE_ARGS}" >> "${GITHUB_ENV}" fi diff --git a/.ci/scripts/wheel/test_cpp_sdk.py b/.ci/scripts/wheel/test_cpp_sdk.py index 176b1aedb5d..e29adff7aa8 100644 --- a/.ci/scripts/wheel/test_cpp_sdk.py +++ b/.ci/scripts/wheel/test_cpp_sdk.py @@ -1620,11 +1620,13 @@ def test_pre_3_28_route_builds_a_consumer_through_variables(work_dir: Path) -> N "target_link_libraries(consumer PRIVATE ${EXECUTORCH_LIBRARIES})\n" "set_target_properties(consumer PROPERTIES CXX_STANDARD ${EXECUTORCH_CXX_STANDARD})\n" # No imported targets on this route, so no TARGET_RUNTIME_DLLS either. A Windows - # consumer copies the DLLs from the directory the package reports. + # consumer copies the DLLs from the directory the package reports, plus any the + # package lists separately because they ship elsewhere. "if(WIN32)\n" ' file(GLOB _dlls "${EXECUTORCH_RUNTIME_LIBRARY_DIR}/*.dll")\n' " add_custom_command(TARGET consumer POST_BUILD COMMAND ${CMAKE_COMMAND} -E " - "copy_if_different ${_dlls} $)\n" + "copy_if_different ${_dlls} ${EXECUTORCH_RUNTIME_DLLS_EXTRA} " + "$)\n" "endif()\n" ) build_dir = work_dir / "pre-328-build" diff --git a/.ci/scripts/wheel/test_cuda_windows.py b/.ci/scripts/wheel/test_cuda_windows.py new file mode 100644 index 00000000000..6d8a2db4911 --- /dev/null +++ b/.ci/scripts/wheel/test_cuda_windows.py @@ -0,0 +1,619 @@ +#!/usr/bin/env python +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +"""Smoke test for a Windows CUDA wheel row. + +The Windows CUDA wheel ships the CUDA delegate for C++ applications only. A CUDA program +cannot be lowered on Windows, since lowering compiles the model with a toolchain that only +the Linux side has, so a Windows user exports on Linux (or WSL) and runs the program with +this wheel. The Python extension therefore carries no CUDA dependency at all, and the +checks here hold both halves of that: + + the CUDA DLLs ship, with import libraries a C++ consumer links + the Python extension depends on no CUDA DLL, so importing the package needs no CUDA + the delegate registers in a C++ application linking executorch::backend_cuda + the device code covers the GPUs the row claims + a program exported on Linux for a Windows target runs through the C++ SDK and matches eager + +The last check needs such a program. Run this file with --export DIR on a Linux machine +with CUDA and the Windows cross toolchain to write one, then point +EXECUTORCH_CUDA_WINDOWS_ARTIFACTS at that directory here. CI does exactly that in the +Windows CUDA wheel workflow's end-to-end jobs, one per CUDA train. The wheel build's own +smoke test has no exported program, so there the check prints SKIP and only the checks +that need no GPU and no program count. +""" + +import json +import os +import platform +import re +import shutil +import subprocess +import sys +import tempfile +from pathlib import Path + +_EXPORT_SCRIPT = """ +import json +import sys +from pathlib import Path + +import torch +from executorch.backends.cuda.cuda_backend import CudaBackend +from executorch.backends.cuda.cuda_partitioner import CudaPartitioner +from executorch.exir import EdgeCompileConfig, to_edge_transform_and_lower +from executorch.exir.backend.compile_spec_schema import CompileSpec + + +class Net(torch.nn.Module): + # Weights, so the program has an external data file, and several operator kinds. + def __init__(self): + super().__init__() + self.fc1 = torch.nn.Linear(16, 32) + self.fc2 = torch.nn.Linear(32, 8) + + def forward(self, x): + return self.fc2(torch.nn.functional.gelu(self.fc1(x))) + x[:, :8] + + +destination = Path(sys.argv[1]) +destination.mkdir(parents=True, exist_ok=True) +torch.manual_seed(0) +model = Net().eval() +example = (torch.randn(4, 16),) +with torch.no_grad(): + expected = model(*example) + +compile_specs = [ + CudaBackend.generate_method_name_compile_spec("forward"), + CompileSpec("platform", b"windows"), +] +program = to_edge_transform_and_lower( + torch.export.export(model, example), + partitioner=[CudaPartitioner(compile_specs)], + compile_config=EdgeCompileConfig(_check_ir_validity=False), +).to_executorch() +assert b"CudaBackend" in program.buffer, "the program carries no CUDA delegate" +with open(destination / "model.pte", "wb") as handle: + program.write_to_file(handle) +program.write_tensor_data_to_file(str(destination)) +assert (destination / "aoti_cuda_blob.ptd").is_file(), "no aoti_cuda_blob.ptd was written" +(destination / "reference.json").write_text( + json.dumps( + { + "shape": list(example[0].shape), + "input": example[0].flatten().tolist(), + "expected": expected.flatten().tolist(), + # The CUDA train the program was compiled with, which the Windows side + # checks against the wheel's, since every CUDA 13 train loads the same + # cudart64_13.dll and a mismatch would not fail on its own. + "cuda": torch.version.cuda, + } + ) +) +print(f"exported a Windows CUDA program to {destination}") +""" + +_REGISTRY_SOURCE = r""" +#include +#include + +#include + +int main() { + executorch::runtime::runtime_init(); + const size_t count = executorch::runtime::get_num_registered_backends(); + for (size_t i = 0; i < count; ++i) { + const auto name = executorch::runtime::get_backend_name(i); + if (name.ok()) { + std::printf("BACKEND %s\n", *name); + } + } + return 0; +} +""" + +_RUNNER_SOURCE = r""" +#include +#include + +#include +#include +#include +#include +#include + +using namespace executorch::extension; + +std::vector read_floats(const char* path) { + std::ifstream file(path); + std::vector values; + float value = 0.0f; + while (file >> value) { + values.push_back(value); + } + return values; +} + +int main(int argc, char** argv) { + if (argc < 7) { + std::printf("usage: runner \n"); + return 2; + } + Module module(argv[1], argv[2]); + auto input_data = read_floats(argv[5]); + const auto expected = read_floats(argv[6]); + auto input = make_tensor_ptr( + {std::atoi(argv[3]), std::atoi(argv[4])}, std::move(input_data)); + const auto result = module.forward(input); + if (!result.ok()) { + std::printf("forward failed: 0x%x\n", static_cast(result.error())); + return 1; + } + const auto output = result->at(0).toTensor(); + if (static_cast(output.numel()) != expected.size()) { + std::printf("output has %zu values, expected %zu\n", + static_cast(output.numel()), expected.size()); + return 1; + } + const float* actual = output.const_data_ptr(); + double worst = 0.0; + for (size_t i = 0; i < expected.size(); ++i) { + const double diff = std::fabs(static_cast(actual[i]) - expected[i]); + if (!std::isfinite(diff)) { + std::printf("output value %zu is not comparable\n", i); + return 1; + } + worst = diff > worst ? diff : worst; + } + if (worst > 1e-3) { + std::printf("output differs from eager PyTorch by %g\n", worst); + return 1; + } + std::printf("ok maxdiff=%g\n", worst); + return 0; +} +""" + + +def _package_dir() -> Path: + import executorch + + return Path(executorch.__path__[0]) + + +def test_cuda_libraries_are_shipped() -> None: + """The row is named for CUDA, so the delegate, its helpers and their link inputs ship.""" + package = _package_dir() + expected = [ + package / "lib" / "executorch_backend_cuda.dll", + package / "lib" / "executorch_backend_cuda.lib", + package / "lib" / "executorch_extension_cuda.dll", + package / "lib" / "executorch_extension_cuda.lib", + package / "backends" / "cuda" / "aoti_cuda_shims.dll", + # What a compiled program links when lowering for Windows, and what the package + # config names as the shim layer's import library. + package / "data" / "lib" / "aoti_cuda_shims.lib", + ] + missing = [ + str(path.relative_to(package)) for path in expected if not path.is_file() + ] + assert not missing, ( + f"this is a CUDA row but {missing} are not in the wheel, so a C++ application could " + "not link or load the CUDA delegate" + ) + print(f"✓ the CUDA DLLs and their import libraries ship ({len(expected)} files)") + + +def test_python_extension_carries_no_cuda() -> None: + """The Python extension must depend on no CUDA DLL. + + CUDA programs are lowered on Linux, so the Windows extension has nothing to do with the + delegate, and depending on it would make importing the package fail on a machine + without the CUDA runtime. + """ + import test_shared_libraries + + package = _package_dir() + extensions = sorted((package / "extension" / "pybindings").glob("_C.*.pyd")) + assert len(extensions) == 1, f"expected one _C extension, found {extensions}" + dependents = { + name.lower() for name in test_shared_libraries._pe_dependents(extensions[0]) + } + cuda = sorted( + name + for name in dependents + if "cuda" in name or name.startswith(("cudart", "cublas", "nvrtc")) + ) + assert not cuda, ( + f"{extensions[0].name} depends on {cuda}, so the Python side of the Windows wheel " + "carries CUDA it cannot use and fails to import without the CUDA runtime" + ) + from executorch.extension.pybindings.portable_lib import ( + _get_registered_backend_names, + ) + + registered = _get_registered_backend_names() + assert "CudaBackend" not in registered, ( + f"CudaBackend is registered in the Python extension ({registered}), which the " + "Windows wheel does not ship it for" + ) + print( + f"✓ {extensions[0].name} depends on no CUDA DLL and registers no CUDA backend" + ) + + +def test_cuda_runtime_is_linked_statically() -> None: + """The wheel's CUDA DLLs import no CUDA runtime DLL, only reaching the driver. + + On Windows CMake links the CUDA runtime library (cudart.lib) into each DLL, and that + library loads the driver, nvcuda.dll, which the display driver installs. So no shipped + DLL may import cudart64_*.dll. + + A compiled model is different: the library AOTInductor builds into the .pte imports + cudart64_13.dll, which comes from the CUDA Toolkit's bin directory on PATH. The + nvidia-cuda-runtime package on PyPI also carries it for win_amd64, but a C++ program + does not search site-packages for DLLs, so declaring it would not make it loadable; the + wheel declares no NVIDIA package for Windows and the guide says where the DLL comes from. + """ + import test_shared_libraries + + package = _package_dir() + for library in ( + package / "lib" / "executorch_backend_cuda.dll", + package / "lib" / "executorch_extension_cuda.dll", + package / "backends" / "cuda" / "aoti_cuda_shims.dll", + ): + dependents = { + name.lower() for name in test_shared_libraries._pe_dependents(library) + } + cudart = sorted(name for name in dependents if name.startswith("cudart")) + assert not cudart, ( + f"{library.name} imports {cudart}, which only the CUDA toolkit provides on " + "Windows, so the wheel would fail to load without it" + ) + shims = package / "backends" / "cuda" / "aoti_cuda_shims.dll" + assert b"nvcuda.dll" in shims.read_bytes(), ( + f"{shims.name} carries no reference to the CUDA driver, so the CUDA runtime it " + "should contain is missing" + ) + import importlib.metadata + + declared = [ + requirement + for requirement in importlib.metadata.requires("executorch") or [] + if "nvidia" in requirement.lower() and "linux" not in requirement.lower() + ] + assert ( + not declared + ), f"the wheel declares {declared} for Windows, where its DLLs need only the driver" + print("✓ no shipped CUDA DLL imports a CUDA runtime DLL, only the driver") + + +def _row_architectures() -> list: + """The GPU architectures this row claims, from the list the build is meant to use. + + Read from cuda_arch_list.sh, the same source the Linux check reads and the pre-build hook + resolves, keyed by the wheel's own +cuXYZ version. Not from TORCH_CUDA_ARCH_LIST: that is + what the build consumed, so a check that trusted it would pass a build that was handed the + wrong list, which is exactly what happened once (the reusable workflow's list replaced + the row's, adding sm_75 and dropping sm_89). + """ + import importlib.metadata + + local = importlib.metadata.version("executorch").partition("+")[2] + assert re.fullmatch( + r"cu\d+", local + ), f"this is run as a CUDA row but the installed version carries no +cuXYZ tag: {local!r}" + text = (Path(__file__).parent / "cuda_arch_list.sh").read_text() + + def assigned(name: str) -> str: + found = re.search(rf'^{name}="([^"]*)"', text, re.M) + assert found, f"cuda_arch_list.sh assigns no {name}, so this row has no list" + value = found.group(1) + reference = re.fullmatch(r"\$\{(\w+)\}", value) + return assigned(reference.group(1)) if reference else value + + return [ + "sm_" + value.replace(".", "") + for value in assigned(f"_cuda_arch_x86_64_{local}").split() + ] + + +def _device_code(cuobjdump: str, library: Path, kind: str) -> set: + """The sm_XY architectures cuobjdump lists for a library, as ELF ("elf") or PTX ("ptx").""" + listed = subprocess.run( + [cuobjdump, f"--list-{kind}", str(library)], + capture_output=True, + text=True, + check=False, + ).stdout + return { + token for token in listed.replace(".", " ").split() if token.startswith("sm_") + } + + +def test_device_code_covers_the_row() -> None: + """Every DLL with device code carries exactly the row's GPUs, and the newest as PTX too. + + Both directions, per library: a missing architecture is a GPU the row claims and cannot + run on, and an extra one means the build did not use the row's list. The newest + architecture must also ship in its portable form, so a GPU newer than any in the row can + compile it on load. --list-elf cannot see that form; --list-ptx can. + """ + expected = set(_row_architectures()) + cuobjdump = shutil.which("cuobjdump") + assert cuobjdump, "cuobjdump from the CUDA toolkit is required to check device code" + package = _package_dir() + with_device_code = {} + for library in sorted(package.rglob("*.dll")): + found = _device_code(cuobjdump, library, "elf") + if found: + with_device_code[library] = found + assert ( + with_device_code + ), f"no shipped DLL carries GPU device code, while the row claims {sorted(expected)}" + for library, found in with_device_code.items(): + missing, extra = sorted(expected - found), sorted(found - expected) + assert not missing and not extra, ( + f"{library.name} carries device code for {sorted(found)} but the row claims " + f"{sorted(expected)}: missing {missing}, unexpected {extra}" + ) + newest = max(expected, key=lambda arch: int(arch.removeprefix("sm_"))) + portable = [ + library.name + for library in with_device_code + if newest in _device_code(cuobjdump, library, "ptx") + ] + assert portable, ( + f"no shipped DLL carries portable device code for {newest}, the newest architecture " + "in the row, so a newer GPU would find no code it can run" + ) + print( + f"✓ device code covers exactly the row {sorted(expected)} in " + f"{', '.join(library.name for library in with_device_code)}, with {newest} also as PTX" + ) + + +def _build(work_dir: Path, name: str, source: str, components) -> Path: + import test_cpp_sdk + + source_dir = work_dir / name + source_dir.mkdir(parents=True, exist_ok=True) + (source_dir / "consumer.cpp").write_text(source) + (source_dir / "CMakeLists.txt").write_text(test_cpp_sdk._consumer_cmake(components)) + config = _package_dir() / "share" / "cmake" + build_dir = work_dir / f"{name}-build" + for command in ( + [ + test_cpp_sdk._tool("cmake"), + "-S", + str(source_dir), + "-B", + str(build_dir), + f"-DCMAKE_PREFIX_PATH={config}", + ], + test_cpp_sdk._cmake_build(test_cpp_sdk._tool("cmake"), build_dir), + ): + result = subprocess.run(command, capture_output=True, text=True, check=False) + assert result.returncode == 0, ( + f"a C++ application linking {components} could not be built against the " + f"installed wheel:\n{result.stdout[-2500:]}\n{result.stderr[-2500:]}" + ) + return test_cpp_sdk._executable(build_dir, "consumer") + + +def test_the_delegate_registers_in_a_cpp_application(work_dir: Path) -> None: + """A C++ application linking executorch::backend_cuda sees CudaBackend registered. + + The delegate registers from a static initializer, so this proves the anchor keeps the DLL + in the import table, that the package config copies the delegate's stream helper and + shim layer beside the program, and that all of them load. Needs no GPU. + """ + import test_cpp_sdk + + consumer = _build( + work_dir, + "cuda-registry", + _REGISTRY_SOURCE, + ["runtime", "kernels_optimized", "backend_cuda"], + ) + beside = {path.name for path in consumer.parent.glob("*.dll")} + for needed in ( + "executorch_backend_cuda.dll", + "executorch_extension_cuda.dll", + "aoti_cuda_shims.dll", + ): + assert needed in beside, ( + f"$ did not copy {needed} beside the application, so it " + f"cannot load the delegate. Copied: {sorted(beside)}" + ) + result = subprocess.run( + [str(consumer)], + capture_output=True, + text=True, + check=False, + env=test_cpp_sdk._loader_clean_environment(), + timeout=test_cpp_sdk._RUN_TIMEOUT, + ) + assert result.returncode == 0, ( + "a C++ application linking the CUDA delegate failed to start, so a DLL it needs is " + f"missing:\n{result.stdout[-1000:]}\n{result.stderr[-1000:]}" + ) + backends = re.findall(r"^BACKEND (\S+)", result.stdout, re.M) + assert "CudaBackend" in backends, ( + f"the application linked executorch::backend_cuda but CudaBackend is not registered: " + f"{backends}" + ) + print(f"✓ a C++ application linking executorch::backend_cuda registers {backends}") + + +def _cuda_environment() -> str: + """The driver and runtime this machine offers, for reading a failure that happens there. + + On Windows every CUDA 13 runtime library, the toolkit's cudart64_13.dll included, loads + the runtime that ships with the display driver (nvcudart_hybrid64.dll), so whether a + program runs depends on the driver more than on the toolkit. CI runners do not put + nvidia-smi on PATH, so this asks the runtime directly. + """ + import ctypes + + lines = [] + nvcuda = ( + Path(os.environ.get("SystemRoot", r"C:\Windows")) / "System32" / "nvcuda.dll" + ) + lines.append(f"nvcuda.dll present: {nvcuda.is_file()}") + smi = shutil.which("nvidia-smi") or str(nvcuda.parent / "nvidia-smi.exe") + if Path(smi).is_file(): + query = subprocess.run( + [ + smi, + "--query-gpu=name,driver_version,compute_cap", + "--format=csv,noheader", + ], + capture_output=True, + text=True, + check=False, + timeout=60, + ) + lines.append(f"nvidia-smi: {(query.stdout or query.stderr).strip()}") + runtime = next( + ( + candidate + for directory in os.environ.get("PATH", "").split(os.pathsep) + if directory + for candidate in sorted(Path(directory).glob("cudart64_*.dll")) + ), + None, + ) + if runtime is None: + lines.append("no cudart64_*.dll on PATH to ask") + return "\n".join(lines) + try: + library = ctypes.WinDLL(str(runtime)) + except OSError as error: + lines.append(f"{runtime} does not load: {error}") + return "\n".join(lines) + library.cudaGetErrorName.restype = ctypes.c_char_p + driver, version, count = ctypes.c_int(0), ctypes.c_int(0), ctypes.c_int(0) + library.cudaDriverGetVersion(ctypes.byref(driver)) + library.cudaRuntimeGetVersion(ctypes.byref(version)) + status = library.cudaGetDeviceCount(ctypes.byref(count)) + name = library.cudaGetErrorName(status) + lines.append( + f"{runtime.name}: driver supports CUDA {driver.value}, runtime {version.value}, " + f"cudaGetDeviceCount -> {name.decode() if name else status} ({count.value} devices)" + ) + return "\n".join(lines) + + +def test_a_program_exported_on_linux_runs(work_dir: Path) -> None: + """A CUDA program lowered on Linux for Windows runs through the C++ SDK and matches eager. + + This is the only route a Windows user has, and the only check here that computes. It needs + artifacts written by `test_cuda_windows.py --export DIR` on Linux and a CUDA device. + """ + import test_cpp_sdk + + artifacts = os.environ.get("EXECUTORCH_CUDA_WINDOWS_ARTIFACTS", "") + if not artifacts: + print( + "SKIP: EXECUTORCH_CUDA_WINDOWS_ARTIFACTS is not set, so no Linux-exported program " + "is available to run. Write one with `test_cuda_windows.py --export DIR` on Linux." + ) + return + artifacts = Path(artifacts) + reference = json.loads((artifacts / "reference.json").read_text()) + import importlib.metadata + + wheel_train = importlib.metadata.version("executorch").partition("+")[2] + exported_train = "cu" + str(reference.get("cuda", "")).replace(".", "") + assert exported_train == wheel_train, ( + f"the program was exported with CUDA {reference.get('cuda')} but this wheel is the " + f"{wheel_train} build, so this would not test the pairing the row claims" + ) + input_file = work_dir / "cuda_input.data" + expected_file = work_dir / "cuda_expected.data" + input_file.write_text(" ".join(repr(v) for v in reference["input"])) + expected_file.write_text(" ".join(repr(v) for v in reference["expected"])) + runner = _build( + work_dir, + "cuda-runner", + _RUNNER_SOURCE, + ["runtime", "kernels_optimized", "backend_cuda"], + ) + environment = _cuda_environment() + print(environment) + result = subprocess.run( + [ + str(runner), + str(artifacts / "model.pte"), + str(artifacts / "aoti_cuda_blob.ptd"), + str(reference["shape"][0]), + str(reference["shape"][1]), + str(input_file), + str(expected_file), + ], + capture_output=True, + text=True, + check=False, + env=test_cpp_sdk._loader_clean_environment(), + timeout=test_cpp_sdk._RUN_TIMEOUT, + ) + assert result.returncode == 0, ( + "a CUDA program exported on Linux did not run correctly through the Windows wheel's " + f"C++ SDK:\n{result.stdout[-2000:]}\n{result.stderr[-2000:]}\n" + f"CUDA on this machine:\n{environment}" + ) + print( + f"✓ a Linux-exported CUDA program runs on Windows and matches eager ({result.stdout.strip().splitlines()[-1]})" + ) + + +def export(destination: str) -> None: + """Write a Windows-targeted CUDA program for the execution check. Run on Linux.""" + assert platform.system() == "Linux", "lowering for Windows runs on Linux" + subprocess.run([sys.executable, "-c", _EXPORT_SCRIPT, destination], check=True) + + +if __name__ == "__main__": + if len(sys.argv) == 3 and sys.argv[1] == "--export": + export(sys.argv[2]) + sys.exit(0) + + assert platform.system() == "Windows", "this is the Windows CUDA row's smoke test" + import test_clean_install + import test_cpp_sdk + import test_shared_libraries + + with tempfile.TemporaryDirectory() as work_dir: + test_clean_install.run_tests(Path(work_dir)) + + test_cuda_libraries_are_shipped() + test_python_extension_carries_no_cuda() + test_cuda_runtime_is_linked_statically() + test_device_code_covers_the_row() + with tempfile.TemporaryDirectory() as work_dir: + test_the_delegate_registers_in_a_cpp_application(Path(work_dir)) + with tempfile.TemporaryDirectory() as work_dir: + test_a_program_exported_on_linux_runs(Path(work_dir)) + + # Everything a CPU Windows row checks applies here too, with one exception: the Arm + # Cortex-M Python module (test_base.test_cmsis_nn_install) is not built when the CUDA + # build uses Ninja, which cannot generate that module's two identically named sources. + with tempfile.TemporaryDirectory() as work_dir: + test_shared_libraries.run_tests(Path(work_dir)) + with tempfile.TemporaryDirectory() as work_dir: + test_cpp_sdk.run_tests(Path(work_dir)) + # The same model run the CPU row ends with, through the Python bindings. + import test_windows + from executorch.examples.models import Backend, Model + from test_base import ModelTest + + test_windows.run_tests( + model_tests=[ModelTest(model=Model.Mv3, backend=Backend.Xnnpack)] + ) diff --git a/.ci/scripts/wheel/test_cuda_windows_e2e.ps1 b/.ci/scripts/wheel/test_cuda_windows_e2e.ps1 new file mode 100644 index 00000000000..c8b61976a8c --- /dev/null +++ b/.ci/scripts/wheel/test_cuda_windows_e2e.ps1 @@ -0,0 +1,91 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +# Builds the Windows CUDA wheel for one CUDA train, installs it, and runs its smoke test +# against a program a Linux job exported for that train. This is the route a Windows user +# has: lowering for CUDA cannot run on Windows, so the program comes from Linux and only +# the wheel's C++ SDK runs it. +param( + [Parameter(Mandatory = $true)][string]$Train, + [Parameter(Mandatory = $true)][string]$Artifacts +) +$ErrorActionPreference = "Stop" + +# The runner's powershell.exe is Windows PowerShell 5.1, where a native command that +# fails does not stop the script ($PSNativeCommandUseErrorActionPreference is 7.3+ only). +# Without this a failing test printed its traceback and the job still passed. +function Invoke-Native { + param([Parameter(Mandatory = $true)][scriptblock]$Command) + & $Command + if ($LASTEXITCODE -ne 0) { + throw "exit code ${LASTEXITCODE}: $Command" + } +} + +if (-not (Test-Path (Join-Path $Artifacts "model.pte"))) { + throw "no model.pte in $Artifacts; the Linux export job did not hand over a program" +} + +# The GPU runners come without a driver; the same step PyTorch's Windows CUDA tests run. +& "$PSScriptRoot\..\install_nvidia_driver_windows.ps1" + +# The runner image carries some toolkits; any other train is assembled from NVIDIA's +# redistributable archives so each row compiles with its own nvcc. +$installed = "C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA\v$Train" +if (Test-Path (Join-Path $installed "bin\nvcc.exe")) { + $cudaHome = $installed +} else { + $cudaHome = Join-Path $env:RUNNER_TEMP "cuda-$Train" + Invoke-Native { python .ci/scripts/wheel/install_cuda_redist.py --train $Train --platform windows-x86_64 --dest $cudaHome } +} +$env:CUDA_HOME = $cudaHome +$env:CUDA_PATH = $cudaHome +$env:PATH = "$cudaHome\bin\x64;$cudaHome\bin;$env:PATH" +Remove-Item Env:CUDACXX -ErrorAction SilentlyContinue +Invoke-Native { nvcc --version } + +Invoke-Native { conda create --yes --quiet -n et python=3.12 } +conda activate et +& "C:\Program Files (x86)\Microsoft Visual Studio\2022\BuildTools\Common7\Tools\Launch-VsDevShell.ps1" -Arch amd64 + +# The Python side of this wheel carries no CUDA, and install_requirements installs CPU +# torch on Windows, which is what the wheel build and its import checks need. It also +# installs the torchao nightly the wheel declares, which PyPI does not carry. +Invoke-Native { python install_requirements.py } +Invoke-Native { pip install ninja } + +# The release row's identity: its +cuXYZ tag and its GPU list, read from the same script +# the release build and the smoke test read. +$cu = "cu" + $Train.Replace(".", "") +$env:BUILD_VERSION = "$(Get-Content version.txt)+$cu" +$archScript = Get-Content .ci/scripts/wheel/cuda_arch_list.sh +$name = "_cuda_arch_x86_64_$cu" +do { + $line = $archScript | Where-Object { $_ -match "^$name=`"([^`"]*)`"" } | Select-Object -First 1 + if (-not $line) { throw "cuda_arch_list.sh assigns no $name" } + $value = [regex]::Match($line, "^$name=`"([^`"]*)`"").Groups[1].Value + $reference = [regex]::Match($value, '^\$\{(\w+)\}$') + if ($reference.Success) { $name = $reference.Groups[1].Value } +} while ($reference.Success) +$env:TORCH_CUDA_ARCH_LIST = $value +$env:CMAKE_ARGS = "-DEXECUTORCH_BUILD_CUDA=ON -DEXECUTORCH_BUILD_VULKAN=OFF" +$env:DISTUTILS_USE_SDK = "1" +Invoke-Native { python setup.py bdist_wheel } +$wheel = Get-ChildItem dist\*.whl | Select-Object -First 1 +# With its declared dependencies, as a user installs it: the smoke test's clean-install +# check fails on any module the wheel needs and does not declare. +Invoke-Native { pip install $wheel.FullName } + +$env:EXECUTORCH_CUDA_WINDOWS_ARTIFACTS = (Resolve-Path $Artifacts).Path +$env:PYTHONIOENCODING = "utf-8" +# From outside the checkout, so the checks inspect the installed wheel, not the sources. +$scripts = (Resolve-Path .ci/scripts/wheel).Path +Push-Location $env:RUNNER_TEMP +try { + Invoke-Native { python (Join-Path $scripts "test_cuda_windows.py") } +} finally { + Pop-Location +} diff --git a/.ci/scripts/wheel/test_shared_libraries.py b/.ci/scripts/wheel/test_shared_libraries.py index 4764e3813a4..8d059ca14d5 100644 --- a/.ci/scripts/wheel/test_shared_libraries.py +++ b/.ci/scripts/wheel/test_shared_libraries.py @@ -214,6 +214,9 @@ "executorch.dll", "executorch::runtime::register_backend", ), + # The delegate's entry points come from cuda_platform, an archive, so its export list does not + # name them either. + "CUDA delegate": ("executorch.dll", "executorch::runtime::register_backend"), } _PE_REPORTS: dict = {} @@ -1793,9 +1796,13 @@ def _assert_shipped_libraries_relocate_with_dyld() -> None: if sys.argv[3] == "torch": import torch # noqa: F401 -# executorch/lib, which a C++ program copies beside itself and the package's entry points -# register. Whether they do is test_package_entry_points_load_their_libraries. -os.add_dll_directory(sys.argv[2]) +# The directories the package ships DLLs in, the ones the Linux libraries record relative +# search paths for: executorch/lib, which a C++ program copies beside itself and the +# package's entry points register (test_package_entry_points_load_their_libraries checks +# that they do), and backends/cuda, where the CUDA delegate's shim layer ships. +for directory in [sys.argv[2], *sys.argv[4:]]: + if os.path.isdir(directory): + os.add_dll_directory(directory) ctypes.WinDLL(sys.argv[1]) """ @@ -1829,6 +1836,7 @@ def _assert_shipped_libraries_load_on_windows(root: Path, package_dir: Path) -> str(target), str(lib_dir), torch_needed, + str(root / "backends" / "cuda"), ], capture_output=True, text=True, @@ -2756,6 +2764,9 @@ def test_extension_contains_no_component() -> None: marker in name for marker in ("kernels_quantized", "kernels_torchao", "extension_cuda") ) + # Not on Windows, where a CUDA program cannot be lowered, so the delegate ships for + # C++ applications and the extension deliberately does not link it. + and not (_WINDOWS and "backend_cuda" in name) } unused = sorted(expected - needed) assert not unused, ( diff --git a/.github/workflows/build-wheels-cuda-windows.yml b/.github/workflows/build-wheels-cuda-windows.yml new file mode 100644 index 00000000000..c798b35e94e --- /dev/null +++ b/.github/workflows/build-wheels-cuda-windows.yml @@ -0,0 +1,216 @@ +# From https://github.com/pytorch/test-infra/wiki/Using-Nova-Reusable-Build-Workflows +name: Build Windows CUDA Wheels + +on: + pull_request: + paths: + - .ci/**/* + - .github/scripts/filter_cuda_matrix.py + - .github/workflows/build-wheels-cuda-windows.yml + - '**/CMakeLists.txt' + - backends/cuda/**/* + - extension/cuda/**/* + - install_utils.py + - pyproject.toml + - setup.py + - docs/source/using-executorch-cpp.md + - tools/cmake/**/* + # The wheel ships these as its C++ SDK. Whole trees rather than the exact directories + # setup.py copies from, so adding one there cannot silently drop it from this list. + - devtools/etdump/**.h + - extension/**.h + - runtime/**.h + push: + branches: + - nightly + - release/* + tags: + # NOTE: Binary build pipelines should only get triggered on release candidate builds + # Release candidate tags look like: v1.11.0-rc1 + - v[0-9]+.[0-9]+.[0-9]+-rc[0-9]+ + - ciflow/binaries/* + workflow_dispatch: + +permissions: + id-token: write + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref_name }}-${{ github.ref_type == 'branch' && github.sha }}-${{ github.event_name == 'workflow_dispatch' }} + cancel-in-progress: true + +jobs: + generate-matrix: + uses: pytorch/test-infra/.github/workflows/generate_binary_build_matrix.yml@main + with: + package-type: wheel + os: windows + test-infra-repository: pytorch/test-infra + test-infra-ref: main + with-cuda: enable + with-cpu: disable + with-rocm: disable + python-versions: '["3.10", "3.11", "3.12", "3.13", "3.14"]' + + # The same filter the Linux CUDA rows use: only the trains this project supports, and one row + # on a pull request. It fails rather than emitting an empty matrix, because a workflow with no + # build job reads as a pass. + filter-matrix: + needs: generate-matrix + runs-on: ubuntu-latest + outputs: + matrix: ${{ steps.filter.outputs.matrix }} + steps: + - uses: actions/setup-python@v6 + with: + python-version: '3.12' + - uses: actions/checkout@v4 + - name: Filter the matrix + id: filter + run: | + set -eou pipefail + MATRIX_BLOB=${{ toJSON(needs.generate-matrix.outputs.matrix) }} + LIMIT_PR=${{ (github.event_name == 'pull_request' && !contains(github.event.pull_request.labels.*.name, 'ciflow/binaries/all')) && 'true' || 'false' }} + MATRIX_BLOB="$(python3 .github/scripts/filter_cuda_matrix.py \ + --matrix "${MATRIX_BLOB}" --limit-pr-builds "${LIMIT_PR}")" + echo "${MATRIX_BLOB}" + echo "matrix=${MATRIX_BLOB}" >> "${GITHUB_OUTPUT}" + + build: + needs: filter-matrix + strategy: + fail-fast: false + matrix: + include: + - repository: pytorch/executorch + pre-script: .ci\\scripts\\wheel\\pre_build_script.sh + env-script: .ci\\scripts\\wheel\\vc_env_helper.bat + post-script: .ci\\scripts\\wheel\\post_build_script.sh + smoke-test-script: .ci/scripts/wheel/test_cuda_windows.py + package-name: executorch + name: ${{ matrix.repository }} + uses: pytorch/test-infra/.github/workflows/build_wheels_windows.yml@main + with: + repository: ${{ matrix.repository }} + ref: "" + test-infra-repository: pytorch/test-infra + test-infra-ref: main + build-matrix: ${{ needs.filter-matrix.outputs.matrix }} + pre-script: ${{ matrix.pre-script }} + env-script: ${{ matrix.env-script }} + post-script: ${{ matrix.post-script }} + package-name: ${{ matrix.package-name }} + smoke-test-script: ${{ matrix.smoke-test-script }} + trigger-event: ${{ github.event_name }} + wheel-build-params: "--verbose" + # Submodules are initialized in pre_build_script.sh with OpenSSL to avoid + # schannel SSL errors on Windows when cloning from non-GitHub hosts. + submodules: false + + docker-image: + name: Resolve CI docker image + uses: ./.github/workflows/_docker-image.yml + + # The trains the end-to-end jobs cover: exactly the ones published as wheels, read from the + # list filter_cuda_matrix.py builds the release rows from, so adding or dropping a train there + # changes what is tested here too. + cuda-trains: + runs-on: ubuntu-latest + outputs: + versions: ${{ steps.list.outputs.versions }} + steps: + - uses: actions/checkout@v4 + - id: list + run: | + set -eou pipefail + VERSIONS="$(python3 -c 'import json, sys; sys.path.insert(0, ".github/scripts"); import filter_cuda_matrix as f; print(json.dumps([v[2:-1] + "." + v[-1] for v in f.SUPPORTED_CUDA_VERSIONS]))')" + echo "${VERSIONS}" + echo "versions=${VERSIONS}" >> "${GITHUB_OUTPUT}" + + # The route a Windows user has, per CUDA train: lowering for CUDA cannot run on Windows, + # so a Linux job exports a program for a Windows target with that train's torch, and a + # Windows GPU job builds the wheel with that train's nvcc and runs the program through + # the wheel's C++ SDK. The build job above has no GPU and no program, so this is what + # executes CUDA code. Modeled on cuda-windows.yml, which does the same for models. + export-cuda-windows-wheel-program: + needs: [docker-image, cuda-trains] + if: github.event.pull_request.head.repo.full_name == github.repository || github.event_name != 'pull_request' + strategy: + fail-fast: false + matrix: + cuda-version: ${{ fromJSON(needs.cuda-trains.outputs.versions) }} + name: export-cuda-windows-wheel-program-${{ matrix.cuda-version }} + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main + permissions: + id-token: write + contents: read + with: + timeout: 90 + runner: mt-l-x86aavx2-11-41-a10g + gpu-arch-type: cuda + gpu-arch-version: ${{ matrix.cuda-version }} + # The image carries the MinGW cross compiler and the CUDA 13.0 Windows runtime. + docker-image: ${{ needs.docker-image.outputs.docker-registry }}/ci-image:executorch-ubuntu-22.04-cuda-windows-${{ needs.docker-image.outputs.ci-docker-hash }} + submodules: recursive + upload-artifact: cuda-windows-wheel-program-${{ matrix.cuda-version }} + ref: ${{ github.event_name == 'pull_request' && github.event.pull_request.head.sha || github.sha }} + script: | + set -eux + TRAIN="${{ matrix.cuda-version }}" + # The pybindings need GLIBCXX_3.4.30, which the image's conda libstdc++ lacks. + mv /opt/conda/lib/libstdc++.so.6 /opt/conda/lib/libstdc++.so.6.bak || true + ln -sf /usr/lib/x86_64-linux-gnu/libstdc++.so.6 /opt/conda/lib/libstdc++.so.6 + if [ "${TRAIN}" != "13.0" ]; then + # Each other train gets its own Linux nvcc, which decides the torch that + # install_executorch picks, and its own Windows runtime to link against. + # Under the job's temp directory: the image runs as ci-user, which cannot + # write to /opt. + TOOLKITS="${RUNNER_TEMP:-/tmp}/cuda-toolkits" + mkdir -p "${TOOLKITS}" 2>/dev/null || TOOLKITS=/tmp/cuda-toolkits + mkdir -p "${TOOLKITS}" + python .ci/scripts/wheel/install_cuda_redist.py --train "${TRAIN}" \ + --platform linux-x86_64 --dest "${TOOLKITS}/cuda-${TRAIN}" + export CUDA_HOME="${TOOLKITS}/cuda-${TRAIN}" + export PATH="${CUDA_HOME}/bin:${PATH}" + export CUDACXX="${CUDA_HOME}/bin/nvcc" + python .ci/scripts/wheel/install_cuda_redist.py --train "${TRAIN}" \ + --platform windows-x86_64 --dest "${TOOLKITS}/cuda-windows-${TRAIN}" --components cuda_cudart + export WINDOWS_CUDA_HOME="${TOOLKITS}/cuda-windows-${TRAIN}" + fi + x86_64-w64-mingw32-g++ --version + nvcc --version + ls -la "${WINDOWS_CUDA_HOME}/bin/x64" + export USE_MKL=OFF + PYTHON_EXECUTABLE=python ./install_executorch.sh + python -c "import torch; v = torch.version.cuda; assert v == '${TRAIN}', v" + python .ci/scripts/wheel/test_cuda_windows.py --export "${RUNNER_ARTIFACT_DIR}" + + test-cuda-windows-wheel-e2e: + needs: [cuda-trains, export-cuda-windows-wheel-program] + # Not gated on the export job's result: that is one result for the whole matrix, so a + # single train failing to export skipped every train here. Each train downloads its own + # program, and a train whose export failed fails here on the missing artifact. + if: ${{ !cancelled() && needs.cuda-trains.result == 'success' }} + strategy: + fail-fast: false + matrix: + cuda-version: ${{ fromJSON(needs.cuda-trains.outputs.versions) }} + name: test-cuda-windows-wheel-e2e-${{ matrix.cuda-version }} + uses: pytorch/test-infra/.github/workflows/windows_job.yml@main + with: + timeout: 240 + runner: windows.g5.4xlarge.nvidia.gpu + gpu-arch-type: cuda + gpu-arch-version: ${{ matrix.cuda-version }} + download-artifact: cuda-windows-wheel-program-${{ matrix.cuda-version }} + ref: ${{ github.event_name == 'pull_request' && github.event.pull_request.head.sha || github.sha }} + script: | + git config --global http.sslBackend openssl + git submodule update --init --recursive + conda init powershell + powershell -Command "& { + Set-PSDebug -Trace 1 + \$ErrorActionPreference = 'Stop' + \$PSNativeCommandUseErrorActionPreference = \$true + .ci/scripts/wheel/test_cuda_windows_e2e.ps1 -Train '${{ matrix.cuda-version }}' -Artifacts \$env:RUNNER_ARTIFACT_DIR + }" diff --git a/CMakeLists.txt b/CMakeLists.txt index b70d8243cd9..287dd00c193 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -927,6 +927,11 @@ endif() # vendored torch-tensorrt delegate). Built before backends/aoti and # backends/cuda, which link it. if(EXECUTORCH_BUILD_CUDA OR EXECUTORCH_BUILD_ROCM) + if(EXECUTORCH_BUILD_CUDA) + # extension/cuda links CUDA::cudart before backends/cuda finds the toolkit, + # and a CPU torch does not define it. + find_package(CUDAToolkit REQUIRED) + endif() add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/extension/cuda) install( DIRECTORY extension/cuda/ @@ -1366,8 +1371,12 @@ if(EXECUTORCH_BUILD_PYBIND) # Build common AOTI functionality if needed by CUDA, ROCm, or Metal backends. if(EXECUTORCH_BUILD_CUDA OR EXECUTORCH_BUILD_ROCM) - # CUDA uses SlimTensor-based shims - list(APPEND _dep_libs aoti_cuda_backend) + # CUDA uses SlimTensor-based shims. Not on Windows: a CUDA program cannot be + # lowered there, only run, so the delegate ships for C++ applications and + # the Python extension stays free of any CUDA dependency. + if(NOT WIN32) + list(APPEND _dep_libs aoti_cuda_backend) + endif() elseif(EXECUTORCH_BUILD_METAL) # Metal still uses ETensor-based shims (for now) list(APPEND _dep_libs aoti_common) diff --git a/backends/cuda/CMakeLists.txt b/backends/cuda/CMakeLists.txt index 187fceaab03..f86db5ab3e7 100644 --- a/backends/cuda/CMakeLists.txt +++ b/backends/cuda/CMakeLists.txt @@ -170,6 +170,15 @@ endif() if(NOT EXECUTORCH_BUILD_ROCM AND CMAKE_CUDA_COMPILER) enable_language(CUDA) + # nvcc compiles host code with cl.exe on Windows, and the CCCL that ships with + # CUDA 13.2 and newer stops with an #error under cl.exe's traditional + # preprocessor. The standard one is what CCCL asks for, and older toolkits + # accept it too. + if(WIN32) + add_compile_options( + "$<$:-Xcompiler=/Zc:preprocessor>" + ) + endif() endif() # Centralize Windows/MSVC checks used throughout this file. @@ -335,7 +344,21 @@ endif() # appends to any INSTALL_RPATH the caller set rather than replacing it, so a # source install keeps its own paths. if(EXECUTORCH_BUILD_SHARED) + if(_cuda_is_windows_msvc) + # In the shared layout the delegate is its own DLL and imports the stream + # guard functions (get/set/peek/clearCurrentCUDAStream) from this one. They + # carry no AOTI_SHIM_EXPORT, so the annotations alone would not export them; + # export everything, as every other shipped DLL does. + set_target_properties( + aoti_cuda_shims PROPERTIES WINDOWS_EXPORT_ALL_SYMBOLS ON + ) + endif() executorch_target_shipped_runtime_path(aoti_cuda_shims) + if(_cuda_is_msvc_toolchain) + # extension_cuda is linked PRIVATE on MSVC and so brings no runtime; resolve + # it from the shared library rather than the static core cuda_platform uses. + executorch_target_link_shared_runtime(aoti_cuda_shims) + endif() endif() install( diff --git a/docs/source/using-executorch-cpp.md b/docs/source/using-executorch-cpp.md index e175789225a..4c719a834d7 100644 --- a/docs/source/using-executorch-cpp.md +++ b/docs/source/using-executorch-cpp.md @@ -197,8 +197,8 @@ These are the components the package provides: | `etdump` | Profiling, to record what ran and how long it took. | Linux, macOS | | `kernels_quantized` | The quantized operator kernels | Linux, macOS, Windows | | `kernels_torchao` | The TorchAO low-bit quantized kernels | Linux and macOS, aarch64 only | -| `backend_cuda` | The CUDA delegate | Linux | -| `extension_cuda` | The CUDA stream extension | Linux | +| `backend_cuda` | The CUDA delegate | Linux, Windows (runs programs exported on Linux) | +| `extension_cuda` | The CUDA stream extension | Linux, Windows | | `backend_openvino` | The OpenVINO delegate | Linux | | `backend_coreml` | The Core ML delegate, for Apple GPU and Neural Engine execution | macOS | | `backend_mlx` | The MLX delegate, for Apple GPU execution | macOS, Apple Silicon | @@ -404,8 +404,9 @@ as Ninja, or build with `cmake --build build --config Release` under Visual Stud against the release C++ library, and a Debug program uses the debug one, whose types are laid out differently, so mixing the two would corrupt memory. The runtime headers therefore refuse a Debug build at compile time, with an error asking for Release. On CMake older than 3.28, copy the DLLs from -`${EXECUTORCH_RUNTIME_LIBRARY_DIR}` instead, and apply `${EXECUTORCH_COMPILE_DEFINITIONS}`, which -carries that check. +`${EXECUTORCH_RUNTIME_LIBRARY_DIR}` instead, together with any listed in +`${EXECUTORCH_RUNTIME_DLLS_EXTRA}`, and apply `${EXECUTORCH_COMPILE_DEFINITIONS}`, which carries +that check. The `etdump` component is not offered on Windows, because the Windows wheel is built without the event tracer. @@ -500,6 +501,34 @@ outside an ARM package. Note that `torch.cuda.get_arch_list()` is not the right PyTorch builds for a wider set at the bottom than these packages do, so a GPU can appear in that list and still not be supported. +#### CUDA on Windows + +The Windows CUDA package runs CUDA programs; it cannot lower them. Lowering compiles the model with +a toolchain only the Linux side has, so export on Linux or in WSL, adding one compile spec that +names the Windows target, then copy `model.pte` and `aoti_cuda_blob.ptd` to the Windows machine: + +```python +from executorch.backends.cuda.cuda_backend import CudaBackend +from executorch.backends.cuda.cuda_partitioner import CudaPartitioner +from executorch.exir.backend.compile_spec_schema import CompileSpec + +compile_specs = [ + CudaBackend.generate_method_name_compile_spec("forward"), + CompileSpec("platform", b"windows"), +] +partitioner = [CudaPartitioner(compile_specs)] +``` + +That export also needs the MinGW cross compiler and the Windows CUDA runtime on the Linux side; +`.ci/docker/common/install_cuda_windows_cross_compile.sh` installs both. + +On Windows, link the same components as on Linux and copy the DLLs beside your program as described +in the Windows section above. `$` also brings the delegate's stream helper and +its AOTI shim layer. Those need only the NVIDIA driver, but the model itself does not: the library +AOTInductor compiles into `model.pte` imports `cudart64_13.dll` (for the CUDA 13 packages), which +comes from the CUDA Toolkit, whose installer puts its `bin` directory on `PATH`. The Python module in +this package carries no CUDA; it is the C++ SDK that runs the program. + ### Building from source diff --git a/pyproject.toml b/pyproject.toml index b632a81628a..b682008e3e0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -2,6 +2,7 @@ requires = [ "cmake>=3.26,<4.0.0; sys_platform != 'win32'", # For building binary targets in the wheel. 4.0.0 breaks third-party CMake build so temporarily pin the version. "cmake>=3.27,<4.0.0; sys_platform == 'win32'", # The shared Windows build uses $, new in 3.27. + "ninja; sys_platform == 'win32'", # The Windows CUDA build's generator. "packaging>=24.2", # Lower bound required by setuptools "patchelf; sys_platform == 'linux'", # Writes the runtime search paths that let the shipped libraries find each other. "pip>=23", # For building the pip package. diff --git a/setup.py b/setup.py index f6eb6deda74..6d75d2f3489 100644 --- a/setup.py +++ b/setup.py @@ -886,6 +886,34 @@ def _cuda_train() -> str: return detected +def _windows_cuda_toolkit(cmake_configuration_args: List[str]) -> str: + """The CUDA toolkit root a Windows CUDA build compiles with, or "" when CUDA is off. + + Decided with the same conditions the CUDA gate below uses, because the answer picks the + generator for the whole configure: a toolkit here with CUDA turned off would still build the + CPU wheel with Ninja instead of the Visual Studio ClangCL toolset. The compiler is the one + install_utils reports, so the toolkit that compiles is the train packaging declares. + """ + arguments = cmake_configuration_args + [item for item in _cmake_args() if item] + if ( + _is_minimal_build() + or _row_is_cpu_only() + or not install_utils.is_cmake_option_on( + arguments, "EXECUTORCH_BUILD_CUDA", default=True + ) + ): + return "" + explicit = install_utils.is_cmake_option_on( + arguments, "EXECUTORCH_BUILD_CUDA", default=False + ) + if not explicit and not install_utils.is_cuda_available(): + return "" + nvcc = shutil.which(install_utils._selected_nvcc()[0]) + if not nvcc: + return "" + return Path(nvcc).resolve().parent.parent.as_posix() + + def _cuda_libraries_built(cmake_cache_dir: Optional[str]) -> bool: """Whether this build produced the CUDA libraries, read from the CMake cache. @@ -1369,6 +1397,18 @@ def _windows_import_libraries() -> List["BuiltFile"]: None, ["EXECUTORCH_BUILD_XNNPACK"], ), + ( + "backends/cuda/", + "executorch_backend_cuda", + None, + ["EXECUTORCH_BUILD_CUDA"], + ), + ( + "extension/cuda/", + "executorch_extension_cuda", + None, + ["EXECUTORCH_BUILD_CUDA"], + ), ] return [ BuiltFile( @@ -2637,9 +2677,35 @@ def run(self): # noqa C901 f"-DCMAKE_BUILD_TYPE={cmake_build_type}", ] - # Use ClangCL on Windows. + # Use ClangCL on Windows. A CUDA build uses Ninja with clang-cl instead when ninja + # is available: the CUDA toolkit's Visual Studio integration fails compiler + # identification under ClangCL on some toolkit and Visual Studio pairs (MSB4023 in + # its targets file), and switching the toolset to cl.exe instead fails on sources + # cl.exe cannot compile. The multi-config generator keeps the per-configuration + # output directories packaging reads. nvcc still compiles device code with cl.exe as + # its host compiler. Without ninja, which a --no-build-isolation install does not + # provide, the build keeps the Visual Studio generator it always used. if _is_windows(): - cmake_configuration_args += ["-T ClangCL"] + windows_cuda_home = _windows_cuda_toolkit(cmake_configuration_args) + if windows_cuda_home and shutil.which("ninja"): + cmake_configuration_args += [ + "-GNinja Multi-Config", + "-DCMAKE_C_COMPILER=clang-cl", + "-DCMAKE_CXX_COMPILER=clang-cl", + f"-DCUDAToolkit_ROOT={windows_cuda_home}", + # Ninja does not generate the Arm Cortex-M Python module's two + # identically named sources as distinct rules. + "-DEXECUTORCH_BUILD_CMSIS_NN_PYBINDS=OFF", + ] + # CMake reads CUDACXX only when CMAKE_CUDA_COMPILER is unset, so naming + # the compiler would drop options a user put there, such as the common + # -allow-unsupported-compiler for a newer Visual Studio. + if not os.environ.get("CUDACXX"): + cmake_configuration_args += [ + f"-DCMAKE_CUDA_COMPILER={windows_cuda_home}/bin/nvcc.exe" + ] + else: + cmake_configuration_args += ["-T ClangCL"] # Allow adding extra cmake args through the environment. Used by some # tests and demos to expand the set of targets included in the pip diff --git a/third-party/CMakeLists.txt b/third-party/CMakeLists.txt index cb67aeda6ab..7a504c74cd0 100644 --- a/third-party/CMakeLists.txt +++ b/third-party/CMakeLists.txt @@ -60,6 +60,15 @@ set(_executorch_host_osx_args "$<${_executorch_apple_cross}:-DCMAKE_OSX_SYSROOT=>" "-DCMAKE_OSX_DEPLOYMENT_TARGET:STRING=$" ) +# The host's executable suffix, for the byproducts and the imported locations +# alike. Not CMAKE_EXECUTABLE_SUFFIX, which is the target's: a cross toolchain +# such as Zephyr's sets .elf, and a byproduct named after it is not the file the +# host build writes, so Ninja would have no rule for the tool. +if(CMAKE_HOST_WIN32) + set(_executorch_host_executable_suffix ".exe") +else() + set(_executorch_host_executable_suffix "") +endif() # Allow reusing a prebuilt host flatc instead of building it from source. This is # required when cross-compiling with the Ninja generator on Windows: the WIN32 @@ -91,24 +100,21 @@ ExternalProject_Add( # flatc runs on the build machine, so it must target the host, not the app's # iOS platform. See _executorch_host_osx_args above. ${_executorch_host_osx_args} - BUILD_BYPRODUCTS /bin/flatc + # With the suffix, or Ninja on Windows has no rule producing the flatc.exe + # that the generated schema headers depend on. + BUILD_BYPRODUCTS /bin/flatc${_executorch_host_executable_suffix} ${_executorch_external_project_additional_args} ${_flatbuffers_ep_additional_args} ) ExternalProject_Get_Property(flatbuffers_ep INSTALL_DIR) add_executable(flatc IMPORTED GLOBAL) add_dependencies(flatc flatbuffers_ep) -if(CMAKE_HOST_WIN32) - # flatbuffers does not use CMAKE_BUILD_TYPE. Internally, the build forces - # Release config, but from CMake's perspective the build type is always Debug. - set_target_properties( - flatc PROPERTIES IMPORTED_LOCATION ${INSTALL_DIR}/bin/flatc.exe - ) -else() - set_target_properties( - flatc PROPERTIES IMPORTED_LOCATION ${INSTALL_DIR}/bin/flatc - ) -endif() +# flatbuffers does not use CMAKE_BUILD_TYPE. Internally, the build forces Release +# config, but from CMake's perspective the build type is always Debug. +set_target_properties( + flatc PROPERTIES IMPORTED_LOCATION + ${INSTALL_DIR}/bin/flatc${_executorch_host_executable_suffix} +) endif() # TODO: re-enable once flatbuffers is added as a subdirectory. @@ -158,22 +164,18 @@ ExternalProject_Add( -DCMAKE_TOOLCHAIN_FILE= ${_executorch_host_osx_args} ${_flatcc_extra_cmake_args} - BUILD_BYPRODUCTS /bin/flatcc + BUILD_BYPRODUCTS /bin/flatcc${_executorch_host_executable_suffix} ${_executorch_external_project_additional_args} ${_flatbuffers_ep_additional_args} ) ExternalProject_Get_Property(flatcc_ep INSTALL_DIR) add_executable(flatcc_cli IMPORTED GLOBAL) add_dependencies(flatcc_cli flatcc_ep) -if(CMAKE_HOST_WIN32) - set_target_properties( - flatcc_cli PROPERTIES IMPORTED_LOCATION ${INSTALL_DIR}/bin/flatcc.exe - ) -else() - set_target_properties( - flatcc_cli PROPERTIES IMPORTED_LOCATION ${INSTALL_DIR}/bin/flatcc - ) -endif() +set_target_properties( + flatcc_cli + PROPERTIES IMPORTED_LOCATION + ${INSTALL_DIR}/bin/flatcc${_executorch_host_executable_suffix} +) endif() set(FLATCC_RTONLY diff --git a/tools/cmake/executorch-wheel-config.cmake b/tools/cmake/executorch-wheel-config.cmake index 3f6ffc9268a..1da6b7d118f 100644 --- a/tools/cmake/executorch-wheel-config.cmake +++ b/tools/cmake/executorch-wheel-config.cmake @@ -38,6 +38,10 @@ # that installs its own binary elsewhere adds this to its INSTALL_RPATH, because # CMake removes the entry it recorded while building. # +# EXECUTORCH_RUNTIME_DLLS_EXTRA -- Windows only: DLLs a consumer on the +# variables route copies beside its program in addition to those in +# EXECUTORCH_RUNTIME_LIBRARY_DIR. Set when the CUDA delegate ships. +# # EXECUTORCH_LIBRARIES -- Libraries to link against: the prebuilt runtime and # the components the wheel shipped, except the ones documented below as opt in. # Not the Python extension, which carries unresolved interpreter symbols that @@ -77,9 +81,10 @@ # MLX_METALLIB_PATH, see below. # executorch::kernels_torchao The TorchAO kernels. Linux and macOS on # aarch64 only. -# executorch::backend_cuda The CUDA delegate. Linux only. -# executorch::extension_cuda The CUDA stream and device helpers. Linux -# only. +# executorch::backend_cuda The CUDA delegate. Linux and Windows; Windows +# runs only, since lowering needs Linux. +# executorch::extension_cuda The CUDA stream and device helpers. Linux and +# Windows. # executorch::backend_openvino The OpenVINO delegate. Linux only. Opens the # OpenVINO runtime by name, which a C++ program # installs and points OPENVINO_LIB_PATH at. @@ -317,6 +322,16 @@ if(_executorch_runtime_library) EXECUTORCH_RUNTIME_LIBRARY_DIR "${_executorch_runtime_library}" DIRECTORY ) endif() +# The CUDA delegate's AOTI shim layer ships in backends/cuda rather than lib/. +# Linux reaches it through a relative search path; a Windows consumer on the +# variables route copies it beside its program along with the rest. +if(WIN32 AND EXISTS + "${_executorch_package_root}/backends/cuda/aoti_cuda_shims.dll" +) + set(EXECUTORCH_RUNTIME_DLLS_EXTRA + "${_executorch_package_root}/backends/cuda/aoti_cuda_shims.dll" + ) +endif() if(_executorch_runtime_library AND NOT _executorch_targets_supported) # The imported targets are skipped, but the libraries themselves are present @@ -771,6 +786,43 @@ _executorch_define_component(backend_openvino executorch_backend_openvino) # while configuring. _executorch_define_component(backend_cuda executorch_backend_cuda) _executorch_define_component(extension_cuda executorch_extension_cuda) +# On Windows the delegate loads two more DLLs at run time: the stream helper, +# and the AOTI shim layer, which the compiled model it loads imports by name. A +# DLL is found only beside the program, so both join the delegate's runtime DLL +# set, which is what $ copies. The shim layer is internal +# and not a component; its import library is the lowering stub the wheel already +# ships. +if(WIN32 AND TARGET executorch::backend_cuda) + set(_executorch_cuda_shims + "${_executorch_package_root}/backends/cuda/aoti_cuda_shims.dll" + ) + if(EXISTS "${_executorch_cuda_shims}" AND NOT TARGET + executorch::_aoti_cuda_shims + ) + add_library(executorch::_aoti_cuda_shims SHARED IMPORTED) + set_target_properties( + executorch::_aoti_cuda_shims + PROPERTIES IMPORTED_LOCATION "${_executorch_cuda_shims}" + IMPORTED_IMPLIB + "${_executorch_package_root}/data/lib/aoti_cuda_shims.lib" + ) + endif() + if(TARGET executorch::_aoti_cuda_shims) + set_property( + TARGET executorch::backend_cuda + APPEND + PROPERTY INTERFACE_LINK_LIBRARIES executorch::_aoti_cuda_shims + ) + endif() + if(TARGET executorch::extension_cuda) + set_property( + TARGET executorch::backend_cuda + APPEND + PROPERTY INTERFACE_LINK_LIBRARIES executorch::extension_cuda + ) + endif() + unset(_executorch_cuda_shims) +endif() # The Qualcomm delegate, present only in a wheel whose build found the QNN SDK, # which today means Linux x86_64. Like the OpenVINO delegate it carries no # undefined vendor symbols, so it links without the SDK present and resolves the @@ -819,10 +871,8 @@ elseif(_executorch_runtime_library) # Tested on the located library rather than set(EXT_SUFFIX "") set(_C_LIBRARY "") else() - # Reported rather than fatal. The arm above only fires when a runtime library - # was located, and a Windows wheel ships none: lib/ holds the CMake package - # and nothing else. So a Windows consumer whose interpreter is not callable as - # python3 reached this branch and find_package aborted its configure even + # Reported rather than fatal. A consumer whose interpreter is not callable as + # python3 reaches this branch, and aborting here would end its configure even # under QUIET, which an optional-dependency probe must never do. Leaving the # extension unset lets the caller see executorch_FOUND=0 and carry on. message( diff --git a/tools/cmake/preset/pybind.cmake b/tools/cmake/preset/pybind.cmake index d3a28344a3f..af11f82e122 100644 --- a/tools/cmake/preset/pybind.cmake +++ b/tools/cmake/preset/pybind.cmake @@ -139,11 +139,7 @@ elseif(CMAKE_SYSTEM_NAME STREQUAL "Windows" OR CMAKE_SYSTEM_NAME STREQUAL ) # One shared runtime, as on Linux and macOS. The event tracer stays off here, # so the profiler library ships only because the Python extension links it. - # Not with CUDA: the Windows CUDA delegate has not been built as a DLL, so a - # CUDA build keeps the static layout it had. - if(NOT EXECUTORCH_BUILD_CUDA) - set_overridable_option(EXECUTORCH_BUILD_SHARED ON) - endif() + set_overridable_option(EXECUTORCH_BUILD_SHARED ON) else() message( FATAL_ERROR "Unsupported CMAKE_SYSTEM_NAME for pybind: ${CMAKE_SYSTEM_NAME}"