From 51d716e640a0f2adc4d02c2f05912008dd78652c Mon Sep 17 00:00:00 2001 From: Anthony Shoumikhin Date: Fri, 2 Oct 2026 16:36:35 -0700 Subject: [PATCH] Stop publishing CUDA 13.0 wheels ExecuTorch publishes a wheel for each CUDA version the PyTorch build matrix offers. PyTorch is moving off CUDA 13.0. It now keeps 13.0 on its nightly builds only as a temporary hold for other projects, and its release-candidate builds do not carry it. If ExecuTorch keeps publishing cu130, nightlies carry a train that no release gets, and the train goes away again when PyTorch ends the hold. Nobody notices when it goes, because the filter skips a CUDA version the generator no longer offers and the run stays green. That is how the cu130 wheel stopped without a failure for four days (Sep 29 to Oct 2). This change publishes CUDA 13.2 and 13.4 only. The single row built for a pull request moves from cu130 to cu132, and the install docs drop the cu130 row. A machine on CUDA 13.0 can use the cu132 wheel, because CUDA minor versions are compatible. A machine on an older CUDA can still build from source. What was tested: - The filter unit tests pass (26 of 26). With the original filter, the two tests that pin the published list fail. - Replayed the shared matrix generator for Linux x86_64 and aarch64, on the nightly and test channels and for a pull request. Nightly goes from 15 rows to 10 (cu132 and cu134, five Pythons each). The test channel stays at 10. The pull request row is cu132. - black and flake8 give the same results as on main. --- .ci/scripts/tests/test_filter_cuda_matrix.py | 2 +- .github/scripts/filter_cuda_matrix.py | 14 +++++++++----- docs/source/getting-started.md | 8 ++++---- docs/source/using-executorch-cpp.md | 6 +++--- 4 files changed, 17 insertions(+), 13 deletions(-) diff --git a/.ci/scripts/tests/test_filter_cuda_matrix.py b/.ci/scripts/tests/test_filter_cuda_matrix.py index 35a36cfe9df..d78c30a5a33 100644 --- a/.ci/scripts/tests/test_filter_cuda_matrix.py +++ b/.ci/scripts/tests/test_filter_cuda_matrix.py @@ -284,7 +284,7 @@ class TestPublishedSets(unittest.TestCase): """ def test_published_cuda_versions(self): - self.assertEqual(FILTER.SUPPORTED_CUDA_VERSIONS, ["cu130", "cu132", "cu134"]) + self.assertEqual(FILTER.SUPPORTED_CUDA_VERSIONS, ["cu132", "cu134"]) def test_published_cuda_versions_are_documented(self): # The install table on the getting started page is the only place a user is told diff --git a/.github/scripts/filter_cuda_matrix.py b/.github/scripts/filter_cuda_matrix.py index bdce1c3feab..15c23e4a781 100644 --- a/.github/scripts/filter_cuda_matrix.py +++ b/.github/scripts/filter_cuda_matrix.py @@ -46,8 +46,8 @@ # ExecuTorch wheel for the same CUDA version, and a missing version means that consumer has # nothing to depend on: # -# cu130 the floor, the generator's stable choice, and the default for accelerator consumers -# cu132 a current TensorRT build target +# cu132 the floor, the generator's stable choice, the default for accelerator consumers, +# and a current TensorRT build target # cu134 the newest, which consumers building against the latest CUDA need # # Skip wholly absent trains so an upstream removal cannot block the remaining releases. @@ -59,11 +59,15 @@ # builds. A machine on CUDA 12.6 can still build from source, where the pinned torch comes # from a channel that carries 12.6. # +# cu130 is not published. PyTorch keeps it on its nightly builds only as a temporary hold for +# other projects, and its release-candidate builds do not carry it, so a release would have +# no cu130 train even while nightlies do. +# # cu132 is included because omitting it would leave a published consumer row with no # ExecuTorch wheel to pair with. It is executable on a device one minor behind, since CUDA # minor versions are compatible, so a cu132 wheel has been run end to end on a CUDA 13.0 # device. The packaging properties are checked on every row regardless. -SUPPORTED_CUDA_VERSIONS: List[str] = ["cu130", "cu132", "cu134"] +SUPPORTED_CUDA_VERSIONS: List[str] = ["cu132", "cu134"] # Python versions to publish, stated rather than derived for the same reason the CUDA # versions are. Deriving them from the rows that survived the filter made the release @@ -73,7 +77,7 @@ SUPPORTED_PYTHON_VERSIONS: List[str] = ["3.10", "3.11", "3.12", "3.13", "3.14"] # The single row built for a pull request. A full matrix on every push would cost hours for -# little signal, and cu130 is the version with a machine on hand that can run a model on it. +# little signal, and cu132 is the version with a machine on hand that can run a model on it. # # The python is not a free choice. When a pull request is limited, the shared generator replaces # the offered python list with its first entry, so that entry is the only python any row can @@ -81,7 +85,7 @@ # the pull request silently built whichever python the generator had left, so the constant # described a row that was never built. PR_PYTHON_VERSION: str = SUPPORTED_PYTHON_VERSIONS[0] -PR_CUDA_VERSION: str = "cu130" +PR_CUDA_VERSION: str = "cu132" # Jetson devices are their own row: a JetPack image, one Python version, and one CUDA # version. Kept empty on purpose today, so no Jetson row is emitted. diff --git a/docs/source/getting-started.md b/docs/source/getting-started.md index 07a5fda2778..1407d6727ab 100644 --- a/docs/source/getting-started.md +++ b/docs/source/getting-started.md @@ -34,19 +34,19 @@ pip install executorch torch \ | Machine you export on | Variant | | --- | --- | | CPU only | `cpu` | -| NVIDIA GPU, CUDA 13.0 | `cu130` | | NVIDIA GPU, CUDA 13.2 | `cu132` | | NVIDIA GPU, CUDA 13.4 | `cu134` | The CUDA packages are built for Linux, on x86_64 and ARM64. Use the `cpu` variant on macOS and on Windows. If your CUDA version is not in the table, choose the closest lower one with the same major version, because CUDA works -across minor versions but not across major ones. There is no package for CUDA -12, so on CUDA 12 use the `cpu` variant or build from source. +across minor versions but not across major ones. There is no package for CUDA 12, +so on CUDA 12 use the `cpu` variant or build from source. On CUDA 13.0 or 13.1 no +lower variant exists, so use `cu132`. To get a change that has landed on the `main` branch but is not in a release yet, use a nightly build. These are rebuilt every day. Put `nightly/` in front -of the variant name, for example `nightly/cu130`, and add `--pre` to the +of the variant name, for example `nightly/cu132`, and add `--pre` to the command, otherwise pip skips development versions. Nightly builds cover the same variants. CUDA 13.4 is the newest, and until the next release it is in nightly builds only. diff --git a/docs/source/using-executorch-cpp.md b/docs/source/using-executorch-cpp.md index 931c136990b..517a79f1655 100644 --- a/docs/source/using-executorch-cpp.md +++ b/docs/source/using-executorch-cpp.md @@ -370,12 +370,12 @@ last of the two settings decides the tag for every entry in the link. ### Running on a GPU with the CUDA package -The CUDA build is a separate package. Releases cover CUDA 13.0, 13.2 and 13.4, so pick the index -matching the CUDA version you have (`cu130`, `cu132` or `cu134`). For CUDA 13.0: +The CUDA build is a separate package. Releases cover CUDA 13.2 and 13.4, so pick the index +matching the CUDA version you have (`cu132` or `cu134`). For CUDA 13.2: ``` pip install executorch torch \ - --index-url https://download.pytorch.org/whl/cu130 \ + --index-url https://download.pytorch.org/whl/cu132 \ --extra-index-url https://pypi.org/simple ```