From 74646aa12293349df07afb348dbaca104ad74a0d Mon Sep 17 00:00:00 2001 From: Anthony Shoumikhin Date: Thu, 1 Oct 2026 03:18:18 -0700 Subject: [PATCH] Declare the torch version in release wheels The ExecuTorch wheel only works with a PyTorch (`torch`) version close to the one it was built with. But since 1.5, nothing tells pip which version that is. So pip can install any torch next to it, and the problem only shows up later, at runtime: ``` pip install executorch==1.5.1 torch==2.12.0 # installs fine # then running a model fails: RuntimeError: tensor does not have a device # 1.5.1 was built on torch 2.14 ``` Releases 0.2 to 1.4 declared torch, because each release branch added the requirement by hand. For 1.5 it was not added. This change makes a release declare its torch version automatically, so pip picks a torch that works: ``` Requires-Dist: torch<2.15,>=2.14.0a0 ``` How it works: - The version comes from the torch the wheel is built against, so nobody has to type it in. - The upper limit (`<2.15`) matters as much as the lower one. Without it, a later torch can break a release that already shipped, as `executorch==1.3.1` with torch 2.14.1 does today. - The `a0` in the lower limit lets a torch built from source (version `2.14.0a0+git...`) count as 2.14. A build on a nightly torch snapshot (version `2.14.0.dev...`) uses that snapshot as the lower limit, because it sorts below `a0`. - It does not choose between the CPU and CUDA builds of torch. The package index you install from still does that. Only releases declare it. Nightly builds, local builds, and the small export-only wheel still declare no torch. The release workflows mark a build as a release by setting `BUILD_VERSION` to a plain version like `1.6.0+cu132`. The wheel tests now also check that a release declares torch and that the installed torch fits the range. Nightly wheels skip this, so on `main` the new unit tests cover the rule instead. Test plan: - New unit tests. Each fails if a part of the rule is removed, including the line in `setup.py` that adds it. - Built the package metadata from the real `setup.py` for release, nightly, local, and export-only builds, and checked the torch line in each. - Checked what pip does with the new range. A fresh install gets torch 2.14.1. An installed 2.14.1 CUDA build or source build is kept, and 2.13 or 2.15 is replaced. - The new wheel test fails on the published 1.5.1 wheel, which shipped without the requirement. --- .../tests/test_release_torch_requirement.py | 109 ++++++++++++++++++ .ci/scripts/wheel/cuda_arch_list.sh | 2 +- .ci/scripts/wheel/test_clean_install.py | 48 +++++++- .ci/scripts/wheel/test_cuda_linux.py | 2 + .ci/scripts/wheel/test_shared_libraries.py | 6 +- docs/source/getting-started.md | 10 +- install_utils.py | 32 +++++ setup.py | 17 ++- 8 files changed, 213 insertions(+), 13 deletions(-) create mode 100644 .ci/scripts/tests/test_release_torch_requirement.py diff --git a/.ci/scripts/tests/test_release_torch_requirement.py b/.ci/scripts/tests/test_release_torch_requirement.py new file mode 100644 index 00000000000..34576e64951 --- /dev/null +++ b/.ci/scripts/tests/test_release_torch_requirement.py @@ -0,0 +1,109 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +# Tests for the torch requirement a release wheel declares. +# +# 1.5.0 and 1.5.1 shipped with no torch requirement at all, and only a release build +# exercises it, so main's own wheel jobs, which are nightlies, never would. + +import ast +import importlib.util +import unittest +from pathlib import Path +from unittest import mock + +from packaging.requirements import Requirement + +ROOT = Path(__file__).resolve().parents[3] + + +def _load_module(name, path): + spec = importlib.util.spec_from_file_location(name, path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +INSTALL_UTILS = _load_module("install_utils", ROOT / "install_utils.py") + + +class TestReleaseTorchRequirement(unittest.TestCase): + def requirement(self, build_version, torch_version="2.14.1+cu132"): + with mock.patch.object( + INSTALL_UTILS.importlib.metadata, "version", return_value=torch_version + ) as version: + requirement = INSTALL_UTILS.release_torch_requirement(build_version) + if requirement is not None: + version.assert_called_once_with("torch") + return requirement + + def test_release_and_candidate_builds_declare_the_built_minor(self): + # Linux and Windows carry the build variant after a plus sign, macOS carries none. + for build_version in ("1.6.0+cpu", "1.6.0+cu132", "1.6.0"): + with self.subTest(build_version=build_version): + self.assertEqual( + self.requirement(build_version), "torch>=2.14.0a0,<2.15" + ) + + def test_nightly_and_local_builds_declare_nothing(self): + for build_version in ("1.6.0.dev20261001+cpu", "1.6.0.dev20261001", "", None): + with self.subTest(build_version=build_version): + self.assertIsNone(self.requirement(build_version)) + + def test_the_range_follows_the_installed_torch(self): + self.assertEqual( + self.requirement("1.7.0+cpu", "2.15.0"), "torch>=2.15.0a0,<2.16" + ) + + def test_the_range_admits_the_torch_the_wheel_was_built_on(self): + # A nightly snapshot sorts below a0, which is how a CUDA row built on one would have + # declared a range excluding its own torch. + for torch_version in ( + "2.14.0", + "2.14.1+cu130", + "2.14.0a0+git0123abc", + "2.14.0.dev20260810+cu134", + ): + with self.subTest(torch_version=torch_version): + requirement = Requirement(self.requirement("1.6.0+cpu", torch_version)) + self.assertTrue( + requirement.specifier.contains(torch_version, prereleases=True) + ) + self.assertFalse(requirement.specifier.contains("2.13.1")) + self.assertFalse( + requirement.specifier.contains( + "2.15.0.dev20261001", prereleases=True + ) + ) + + +class TestSetupDeclaresIt(unittest.TestCase): + def test_only_the_full_wheel_declares_it(self): + # setup.py is read rather than imported, because importing it runs setup(). + module = ast.parse((ROOT / "setup.py").read_text()) + called = [ + { + call.func.id + for call in ast.walk(node.value) + if isinstance(call, ast.Call) and isinstance(call.func, ast.Name) + } + for node in ast.walk(module) + if isinstance(node, ast.Assign) + and any( + isinstance(target, ast.Subscript) + and getattr(target.slice, "value", None) == "install_requires" + for target in node.targets + ) + ] + full = [names for names in called if "_base_dependencies" in names] + minimal = [names for names in called if "_minimal_dependencies" in names] + self.assertEqual((len(full), len(minimal)), (1, 1), called) + self.assertIn("_torch_dependencies", full[0]) + self.assertNotIn("_torch_dependencies", minimal[0]) + + +if __name__ == "__main__": + unittest.main() diff --git a/.ci/scripts/wheel/cuda_arch_list.sh b/.ci/scripts/wheel/cuda_arch_list.sh index 9e74af6c644..8eacc509483 100644 --- a/.ci/scripts/wheel/cuda_arch_list.sh +++ b/.ci/scripts/wheel/cuda_arch_list.sh @@ -53,7 +53,7 @@ _cuda_arch_aarch64_cu134="${_cuda_arch_aarch64_cu130}" # rather than the ones a generic wheel resolves. # # 8.7 is that exception. This is the only row whose CUDA major matches what that module's software -# release ships, and the wheel declares no PyTorch, so the user supplies the build that carries +# release ships, and the wheel does not pick a PyTorch build, so the user supplies the one that carries # their architecture. Omitting it does not protect them from a bad pairing, it only removes the # device code they need. # diff --git a/.ci/scripts/wheel/test_clean_install.py b/.ci/scripts/wheel/test_clean_install.py index 65943277324..87e86c5ccd6 100644 --- a/.ci/scripts/wheel/test_clean_install.py +++ b/.ci/scripts/wheel/test_clean_install.py @@ -21,6 +21,7 @@ import json import os +import re import subprocess import sys from pathlib import Path @@ -45,9 +46,9 @@ # Distributions that are legitimately present without being declared. # -# torch, because the wheel deliberately does not declare it: a consumer brings the build -# matching their platform and accelerator. The rest are what torch itself requires, so they are -# guaranteed alongside it. +# torch, because only a release wheel declares it and this check also runs on nightlies, where a +# consumer brings the build matching their platform and accelerator. The rest are what torch itself +# requires, so they are guaranteed alongside it. ASSUMED_PRESENT: Set[str] = {"torch", "executorch"} @@ -208,12 +209,53 @@ def run_tests(work_dir: Path) -> None: ) _check_top_level_names() + test_release_declares_torch() print( f"All {len(REQUIRED_IMPORTS)} modules import with only declared dependencies, and the " f"metadata names only the package." ) +def test_release_declares_torch() -> None: + """A release wheel must declare the torch it was built against, and torch must satisfy it. + + The native code links torch's C++ library, so a release that declares no torch lets pip pair + it with any torch at all. 1.5.0 and 1.5.1 shipped that way. + + Release means what it means to setup.py: BUILD_VERSION is a plain version. The installed + version cannot tell, because a local build without BUILD_VERSION is also a plain version + followed by its git hash. + """ + import importlib.metadata as metadata + + from packaging.requirements import Requirement + + build_version = os.environ.get("BUILD_VERSION", "").strip() + if not re.fullmatch(r"\d+(\.\d+)*", build_version.split("+", 1)[0]): + print( + f"BUILD_VERSION {build_version!r} is not a release, so no torch is declared" + ) + return + + version = metadata.version("executorch") + declared = [ + requirement + for requirement in map(Requirement, metadata.requires("executorch") or []) + if requirement.name == "torch" + ] + assert declared, ( + f"executorch {version} is a release but declares no torch requirement, so pip will " + "pair it with a torch its native code was not built against" + ) + installed = metadata.version("torch") + assert declared[0].specifier.contains( + installed, prereleases=True + ), f"executorch {version} requires {declared[0]}, but torch {installed} is installed" + print( + f"✓ release executorch {version} requires {declared[0]}, torch is {installed}" + ) + + # Reads its arguments from stdin, blocks the named modules, then imports. A blocked module # raises ModuleNotFoundError exactly as it would be absent, so the traceback shows the import # chain that wanted it. diff --git a/.ci/scripts/wheel/test_cuda_linux.py b/.ci/scripts/wheel/test_cuda_linux.py index 914fafaa3f3..1bbfd079d15 100644 --- a/.ci/scripts/wheel/test_cuda_linux.py +++ b/.ci/scripts/wheel/test_cuda_linux.py @@ -34,6 +34,7 @@ from typing import Optional, Set import test_base +import test_clean_install import test_cpp_sdk import test_shared_libraries from examples.models import Backend, Model @@ -585,6 +586,7 @@ def test_a_model_runs_through_the_delegate() -> None: test_cuda_libraries_are_shipped() test_cuda_runtime_is_declared() + test_clean_install.test_release_declares_torch() test_cuda_libraries_resolve_relatively() test_device_code_covers_the_row() test_portable_device_code_is_present() diff --git a/.ci/scripts/wheel/test_shared_libraries.py b/.ci/scripts/wheel/test_shared_libraries.py index e23aabb4f6a..44c57c80792 100644 --- a/.ci/scripts/wheel/test_shared_libraries.py +++ b/.ci/scripts/wheel/test_shared_libraries.py @@ -2028,9 +2028,9 @@ def _names_a_build_directory(entry: str) -> bool: ) -# Absolute directories a shipped library may name. PyTorch's own is allowed because the wheel -# neither declares nor bundles PyTorch, so an absolute path is the only way to reach it. The maths -# library arch directories are allowed because a real installation spells them below a prefix, as +# Absolute directories a shipped library may name. PyTorch's own is allowed because the wheel does +# not bundle PyTorch, so an absolute path is the only way to reach it. The maths library arch +# directories are allowed because a real installation spells them below a prefix, as # /opt/intel/mkl/lib/intel64, which the environment genuinely provides. # # Matched as a suffix. A substring test exempted any path merely CONTAINING one of these, so a diff --git a/docs/source/getting-started.md b/docs/source/getting-started.md index 07a5fda2778..bef5d894005 100644 --- a/docs/source/getting-started.md +++ b/docs/source/getting-started.md @@ -17,10 +17,12 @@ The following are required to install the ExecuTorch host libraries, needed to e ## Installation To use ExecuTorch, you will need to install both the Python package and the appropriate platform-specific runtime libraries. Pip is the recommended way to install the ExecuTorch python package. Consider installing it within a virtual environment, such as one provided by [conda](https://docs.conda.io/projects/conda/en/latest/user-guide/getting-started.html#creating-environments) or [venv](https://packaging.python.org/en/latest/guides/installing-using-pip-and-virtual-environments/#create-and-use-virtual-environments). -Install PyTorch in the same command. The ExecuTorch package does not declare it -as a dependency, because the build you need depends on your hardware, so pip -cannot choose one for you. Installing ExecuTorch on its own gives an environment -where exporting a model stops with `No module named 'torch'`. +Install PyTorch in the same command. A release of ExecuTorch declares the +PyTorch versions it works with, but not which build of PyTorch to use, because +that depends on your hardware. The package index you install from picks the CPU +or CUDA build. Nightly builds of ExecuTorch declare no PyTorch at all, so +installing one on its own gives an environment where exporting a model stops +with `No module named 'torch'`. Both packages come from the same package index. Find your machine in the table below and put the name from it in place of ``: diff --git a/install_utils.py b/install_utils.py index 4e1397413b3..5308e7433de 100644 --- a/install_utils.py +++ b/install_utils.py @@ -6,6 +6,7 @@ # LICENSE file in the root directory of this source tree. import functools +import importlib.metadata import os import platform import re @@ -296,6 +297,37 @@ def _normalize_cmake_bool(value: Optional[str], default: bool = False) -> bool: return cmake_boolean_is_true(value) +def release_torch_requirement(build_version: Optional[str]) -> Optional[str]: + """The torch requirement a release wheel declares, or None for any other build. + + The native extensions link torch's C++ library, which has no stable ABI. With a wheel built on + 2.14, torch 2.12 imports and then fails at runtime, and with 2.11 a shipped library fails to + load. So a release declares the minor it was built on. The cap matters as much as the floor: + an open range let a later torch break a release that had already shipped. + + Only a release declares it. test-infra sets BUILD_VERSION to a plain version for a release or a + release candidate, such as 1.6.0+cu132, and to a .dev version for a nightly. A local build + leaves it unset. + + The version comes from the torch installed in the build environment, because that is what the + native code compiles against. torch_pin.py is not what the installer reads, and on one release + branch it named 2.11 while the build installed 2.10. + + The floor is a0 so that a torch built from source, which reports a version like + 2.14.0a0+git0123abc, still satisfies it. A nightly snapshot such as 2.14.0.dev20260810 sorts + below a0, so a build on one uses that snapshot as the floor, or the range would exclude the + torch the wheel was built on. + """ + public = (build_version or "").strip().split("+", 1)[0] + if not re.fullmatch(r"\d+(\.\d+)*", public): + return None + torch_version = importlib.metadata.version("torch").split("+", 1)[0] + major, minor = (int(part) for part in torch_version.split(".")[:2]) + snapshot = re.fullmatch(rf"{major}\.{minor}\.0\.dev\d+", torch_version) + floor = snapshot.group(0) if snapshot else f"{major}.{minor}.0a0" + return f"torch>={floor},<{major}.{minor + 1}" + + def _cuda_version_to_pytorch_suffix(major, minor): """ Generate PyTorch CUDA wheel suffix from CUDA version numbers. diff --git a/setup.py b/setup.py index c0a23cb0195..9a6b6587b6d 100644 --- a/setup.py +++ b/setup.py @@ -1197,7 +1197,8 @@ def _minimal_dependencies() -> List[str]: Derived as the subset of _base_dependencies() that executorch.exir needs to lower and serialize a .pte, so version pins and markers stay in sync with the - full set. torch is intentionally absent from both (consumers bring their own). + full set. torch is intentionally absent, as it is from a nightly full wheel + (consumers bring their own); only a release full wheel declares it. mpmath is intentionally dropped too: it is pulled transitively by sympy, whose "mpmath<1.4" cap resolves to the same 1.3.0 the full wheel pins. Keep the name set below in sync with the `expected` set in .ci/scripts/test_minimal_wheel.sh. @@ -1228,6 +1229,16 @@ def _name(dep: str) -> str: return minimal +def _torch_dependencies() -> List[str]: + """The torch requirement of a release wheel, or nothing for any other build. + + Reads the BUILD_VERSION environment variable rather than Version.string(), which appends the git + hash to version.txt when it is unset and so makes a local build look like a release. + """ + requirement = install_utils.release_torch_requirement(os.getenv("BUILD_VERSION")) + return [requirement] if requirement else [] + + class Version: """Static strings that describe the version of the pip package.""" @@ -2852,7 +2863,9 @@ def iter_distribution_names(self): setup_kwargs["packages"] = _full_packages() # A CUDA wheel links the CUDA runtime but does not bundle it, so the wheels that # carry it are declared here. A CPU wheel adds nothing. - setup_kwargs["install_requires"] = _base_dependencies() + _cuda_dependencies() + setup_kwargs["install_requires"] = ( + _base_dependencies() + _cuda_dependencies() + _torch_dependencies() + ) setup(