From 7fc73b3264993c9ad856524eadaedc2dd6b20a24 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Fri, 18 Sep 2026 12:31:10 -0700 Subject: [PATCH 01/18] cuda.core: require a per-major cuda-bindings floor at build and run time cuda.core accepted any cuda-bindings of the right major at build time and at run time, and built against whatever cuda.h was on the include path. Builds succeeded for configurations we never test, and an older cuda-bindings at run time surfaced as an ImportError for a missing C function, a silently disabled feature, or a null-pointer crash (#2783). Each supported CUDA major now has a floor, the newest cuda-bindings release its CI source root can build: 12.9.8 for CUDA 12 and 13.4.1 for CUDA 13, in the import-free module cuda/core/_bindings_floor.py. Build time: the pip build requirement becomes `cuda-bindings>=,==.*`, and because conda-forge, pixi and --no-build-isolation installs bypass it, the backend itself checks the imported cuda-bindings against the floor and requires the cuda.h it compiles against to have the same major.minor as that cuda-bindings (the header cuda-bindings was generated from), even when CUDA_CORE_BUILD_MAJOR is set. It records the header version and the floor in the generated, gitignored cuda/core/_build_info.py, shipped like _version.py. Run time: cuda/core/__init__.py reads that record from the selected build (the cu12/cu13 subpackage of the merged wheel, or the top level of a plain build) and requires the installed cuda-bindings to be of the build's major and at least the floor, or at least the header's minor when that is newer. The error names the version found, the version required, and the pip command that fixes it. Pre-release and dev builds of an accepted version pass. Packaging: the cu12/cu13 extras pin the floor; a test keeps them and the ci/versions.yml toolkit pins in step with the module. CI: the BINDINGS_SOURCE=published rows, which paired a new wheel with cuda-bindings 13.0 (no longer supported), become BINDINGS_SOURCE=floor: they install the floor bindings, read from the wheel under test by ci/tools/cuda_core_bindings_floor.py, and keep their older CTK libraries. The prior-major rows whose CTK minor differs from prev_build do the same. Docs: support policy section, install guide, 1.3.0 release note. Co-Authored-By: Claude Fable 5.1 --- .github/workflows/test-wheel-linux.yml | 4 +- .github/workflows/test-wheel-windows.yml | 4 +- .gitignore | 2 + ci/tools/cuda_core_bindings_floor.py | 60 +++++++ ci/tools/env-vars | 29 ++- ci/tools/run-tests | 10 +- .../tests/test_cuda_core_bindings_floor.py | 71 ++++++++ cuda_core/build_hooks.py | 168 +++++++++++++++--- cuda_core/cuda/core/__init__.py | 49 ++++- cuda_core/cuda/core/_bindings_floor.py | 122 +++++++++++++ cuda_core/docs/source/install.rst | 19 +- cuda_core/docs/source/release/1.3.0-notes.rst | 14 ++ cuda_core/docs/source/support.rst | 45 ++++- cuda_core/pyproject.toml | 4 +- cuda_core/tests/test_bindings_floor.py | 161 +++++++++++++++++ cuda_core/tests/test_build_hooks.py | 123 +++++++++++++ 16 files changed, 835 insertions(+), 50 deletions(-) create mode 100644 ci/tools/cuda_core_bindings_floor.py create mode 100644 ci/tools/tests/test_cuda_core_bindings_floor.py create mode 100644 cuda_core/cuda/core/_bindings_floor.py create mode 100644 cuda_core/tests/test_bindings_floor.py diff --git a/.github/workflows/test-wheel-linux.yml b/.github/workflows/test-wheel-linux.yml index 2f8e9a47a02..8abb5ac48e7 100644 --- a/.github/workflows/test-wheel-linux.yml +++ b/.github/workflows/test-wheel-linux.yml @@ -248,14 +248,14 @@ jobs: fi - name: Display structure of downloaded cuda-python artifacts - if: ${{ env.TEST_PYTHON == 'true' && env.BINDINGS_SOURCE != 'published' }} + if: ${{ env.TEST_PYTHON == 'true' && env.BINDINGS_SOURCE != 'floor' }} run: | pwd ls -lah cuda_python*.whl cuda_pathfinder/ - name: Display structure of downloaded cuda.bindings artifacts if: ${{ (env.TEST_BINDINGS == 'true' || env.TEST_CORE == 'true' || env.TEST_PYTHON == 'true') && - env.BINDINGS_SOURCE != 'published' }} + env.BINDINGS_SOURCE != 'floor' }} run: | pwd ls -lahR $CUDA_BINDINGS_ARTIFACTS_DIR diff --git a/.github/workflows/test-wheel-windows.yml b/.github/workflows/test-wheel-windows.yml index 1b9fc0ff9cb..16db1ab8d4c 100644 --- a/.github/workflows/test-wheel-windows.yml +++ b/.github/workflows/test-wheel-windows.yml @@ -228,14 +228,14 @@ jobs: fi - name: Display structure of downloaded cuda-python artifacts - if: ${{ env.TEST_PYTHON == 'true' && env.BINDINGS_SOURCE != 'published' }} + if: ${{ env.TEST_PYTHON == 'true' && env.BINDINGS_SOURCE != 'floor' }} run: | Get-Location Get-ChildItem cuda_python*.whl | Select-Object Mode, LastWriteTime, Length, FullName - name: Display structure of downloaded cuda.bindings artifacts if: ${{ (env.TEST_BINDINGS == 'true' || env.TEST_CORE == 'true' || env.TEST_PYTHON == 'true') && - env.BINDINGS_SOURCE != 'published' }} + env.BINDINGS_SOURCE != 'floor' }} run: | Get-Location Get-ChildItem -Recurse -Force $env:CUDA_BINDINGS_ARTIFACTS_DIR | Select-Object Mode, LastWriteTime, Length, FullName diff --git a/.gitignore b/.gitignore index d2b0d4ffdde..a7e10ed1b7e 100644 --- a/.gitignore +++ b/.gitignore @@ -39,6 +39,8 @@ cuda_bindings/cuda/bindings/utils/_get_handle.pyx # Version files from setuptools_scm _version.py +# Generated by cuda_core/build_hooks.py at build time (see cuda/core/__init__.py). +cuda_core/cuda/core/_build_info.py # Distribution / packaging .Python diff --git a/ci/tools/cuda_core_bindings_floor.py b/ci/tools/cuda_core_bindings_floor.py new file mode 100644 index 00000000000..d0983547baa --- /dev/null +++ b/ci/tools/cuda_core_bindings_floor.py @@ -0,0 +1,60 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# SPDX-License-Identifier: Apache-2.0 + +"""Print the cuda-bindings floor of a cuda-core wheel for one CUDA major. + + cuda_core_bindings_floor.py --wheel dist/cuda_core-*.whl --major 13 + -> 13.4.1 + +CI installs `cuda-bindings==` next to a freshly built cuda-core wheel to +test the oldest cuda-bindings that wheel supports (BINDINGS_SOURCE=floor in +ci/tools/env-vars). The floor is read from the wheel under test rather than +from the checkout, so a nightly job that tests a wheel built from another +commit reads that wheel's floor. + +The wheel carries the import-free module cuda/core/_bindings_floor.py, at top +level in a single-major build and under cuda/core/cu/ in the merged +wheel; this script evaluates it and prints CUDA_BINDINGS_FLOOR[major]. +""" + +from __future__ import annotations + +import argparse +import sys +import zipfile +from pathlib import Path + +MODULE = "_bindings_floor.py" + + +def floor_from_source(source: str, major: int) -> str: + namespace: dict = {} + exec(compile(source, MODULE, "exec"), namespace) + floors = namespace["CUDA_BINDINGS_FLOOR"] + if major not in floors: + raise SystemExit(f"CUDA {major} is not a supported major (floors: {sorted(floors)})") + return namespace["format_version"](floors[major]) + + +def floor_from_wheel(wheel: Path, major: int) -> str: + with zipfile.ZipFile(wheel) as zf: + names = set(zf.namelist()) + for candidate in (f"cuda/core/cu{major}/{MODULE}", f"cuda/core/{MODULE}"): + if candidate in names: + return floor_from_source(zf.read(candidate).decode("utf-8"), major) + raise SystemExit(f"{wheel.name} contains no {MODULE}; is it a cuda-core wheel?") + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--wheel", type=Path, required=True, help="the cuda-core wheel under test") + parser.add_argument("--major", type=int, required=True, help="CUDA major series (12 or 13)") + args = parser.parse_args(argv) + print(floor_from_wheel(args.wheel, args.major)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/ci/tools/env-vars b/ci/tools/env-vars index 4155753a4ee..03d074b5a30 100755 --- a/ci/tools/env-vars +++ b/ci/tools/env-vars @@ -64,23 +64,40 @@ elif [[ "${1}" == "test" ]]; then # BINDINGS_SOURCE controls which cuda-bindings to install at test time: # main — use the just-built bindings wheel from this CI run # backport — fetch bindings from the prior (N-1) branch - # published — install from PyPI (cuda-bindings==${TEST_CUDA_MAJOR}.${TEST_CUDA_MINOR}.*) + # floor — install the oldest cuda-bindings the cuda-core wheel under test + # supports, from PyPI (its per-major floor; see + # cuda_core/cuda/core/_bindings_floor.py and ci/tools/run-tests). + # Selected when the test CTK minor differs from the one the wheel + # was built against, so those rows exercise new cuda-core + floor + # bindings + older CTK libraries, the skew cuda-core supports. + # (cuda-bindings older than the floor is unsupported and fails at + # import: https://github.com/NVIDIA/cuda-python/issues/2783.) # # SKIP_CUDA_BINDINGS_TEST / SKIP_CYTHON_TEST control which *tests* to run # (they do NOT affect installation — that's BINDINGS_SOURCE's job). BUILD_CUDA_MINOR="$(cut -d '.' -f 2 <<< ${BUILD_CUDA_VER})" TEST_CUDA_MINOR="$(cut -d '.' -f 2 <<< ${CUDA_VER})" + # The prior-major half of the cuda-core wheel is built against ci/versions.yml's + # prev_build toolkit (and the backport branch's bindings). + BUILD_PREV_CUDA_VER="$(sed -n '/prev_build:/,/version:/s/.*version: *"\([^"]*\)".*/\1/p' ci/versions.yml)" + BUILD_PREV_CUDA_MINOR="$(cut -d '.' -f 2 <<< ${BUILD_PREV_CUDA_VER})" if [[ ${BUILD_CUDA_MAJOR} != ${TEST_CUDA_MAJOR} ]]; then - # Major mismatch (e.g. build=13.x, test=12.x): use the backport branch. - BINDINGS_SOURCE=backport SKIP_CUDA_BINDINGS_TEST=1 SKIP_CYTHON_TEST=1 + if [[ ${BUILD_PREV_CUDA_MINOR} != ${TEST_CUDA_MINOR} ]]; then + # Prior major, minor mismatch (e.g. built against 12.9, test=12.6): floor + # bindings from PyPI with the older CTK libraries. + BINDINGS_SOURCE=floor + else + # Prior major, same minor (e.g. build=13.x, test=12.9): the backport branch. + BINDINGS_SOURCE=backport + fi elif [[ ${BUILD_CUDA_MINOR} != ${TEST_CUDA_MINOR} ]]; then - # Same major, minor mismatch (e.g. build=13.2, test=13.0): use published - # bindings from PyPI to test the real-world backward-compat scenario. - BINDINGS_SOURCE=published + # Same major, minor mismatch (e.g. build=13.4, test=13.0): floor bindings + # from PyPI with the older CTK libraries. + BINDINGS_SOURCE=floor SKIP_CUDA_BINDINGS_TEST=1 SKIP_CYTHON_TEST=1 else diff --git a/ci/tools/run-tests b/ci/tools/run-tests index cfa7e9d6a7a..7a9d90955a4 100755 --- a/ci/tools/run-tests +++ b/ci/tools/run-tests @@ -74,10 +74,14 @@ elif [[ "${test_module}" == "core" || "${test_module}" == nightly-* ]]; then # Resolve bindings based on BINDINGS_SOURCE (set by env-vars): # main/backport → local wheel from artifacts dir - # published → install from PyPI by version + # floor → the oldest cuda-bindings the core wheel under test supports, + # read from that wheel, installed from PyPI BINDINGS_ARGS=() - if [[ "${BINDINGS_SOURCE}" == "published" ]]; then - BINDINGS_ARGS+=("cuda-bindings==${TEST_CUDA_MAJOR}.${TEST_CUDA_MINOR}.*") + if [[ "${BINDINGS_SOURCE}" == "floor" ]]; then + CORE_WHL_FOR_FLOOR=("${CUDA_CORE_ARTIFACTS_DIR}"/*.whl) + BINDINGS_FLOOR="$(python ci/tools/cuda_core_bindings_floor.py --wheel "${CORE_WHL_FOR_FLOOR[0]}" --major "${TEST_CUDA_MAJOR}")" + echo "cuda-bindings floor of ${CORE_WHL_FOR_FLOOR[0]##*/} for CUDA ${TEST_CUDA_MAJOR}: ${BINDINGS_FLOOR}" + BINDINGS_ARGS+=("cuda-bindings==${BINDINGS_FLOOR}") else BINDINGS_ARGS=("${CUDA_BINDINGS_ARTIFACTS_DIR}"/*.whl) if [[ "${LOCAL_CTK}" != 1 ]]; then diff --git a/ci/tools/tests/test_cuda_core_bindings_floor.py b/ci/tools/tests/test_cuda_core_bindings_floor.py new file mode 100644 index 00000000000..69f2d81050d --- /dev/null +++ b/ci/tools/tests/test_cuda_core_bindings_floor.py @@ -0,0 +1,71 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# SPDX-License-Identifier: Apache-2.0 + +import importlib.util +import zipfile +from pathlib import Path + +import pytest + +TOOLS = Path(__file__).resolve().parent.parent +REPO = TOOLS.parent.parent +FLOOR_MODULE = REPO / "cuda_core" / "cuda" / "core" / "_bindings_floor.py" + + +def _load_tool(): + spec = importlib.util.spec_from_file_location("cuda_core_bindings_floor", TOOLS / "cuda_core_bindings_floor.py") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +tool = _load_tool() + + +def _expected(major): + namespace = {} + exec(FLOOR_MODULE.read_text(encoding="utf-8"), namespace) + return namespace["format_version"](namespace["CUDA_BINDINGS_FLOOR"][major]) + + +def _wheel(tmp_path, entries): + path = tmp_path / "cuda_core-1.3.0-cp312-cp312-linux_x86_64.whl" + with zipfile.ZipFile(path, "w") as zf: + for name in entries: + zf.writestr(name, FLOOR_MODULE.read_text(encoding="utf-8")) + return path + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +@pytest.mark.parametrize("major", [12, 13]) +def test_reads_the_merged_wheel_layout(tmp_path, major): + wheel = _wheel(tmp_path, ["cuda/core/cu12/_bindings_floor.py", "cuda/core/cu13/_bindings_floor.py"]) + assert tool.floor_from_wheel(wheel, major) == _expected(major) + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_reads_a_single_major_wheel(tmp_path): + wheel = _wheel(tmp_path, ["cuda/core/_bindings_floor.py"]) + assert tool.floor_from_wheel(wheel, 13) == _expected(13) + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_rejects_a_wheel_without_the_module(tmp_path): + wheel = _wheel(tmp_path, []) + with pytest.raises(SystemExit, match="contains no _bindings_floor.py"): + tool.floor_from_wheel(wheel, 13) + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_rejects_an_unsupported_major(tmp_path): + wheel = _wheel(tmp_path, ["cuda/core/_bindings_floor.py"]) + with pytest.raises(SystemExit, match="CUDA 11 is not a supported major"): + tool.floor_from_wheel(wheel, 11) + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_cli_prints_the_floor(tmp_path, capsys): + wheel = _wheel(tmp_path, ["cuda/core/_bindings_floor.py"]) + assert tool.main(["--wheel", str(wheel), "--major", "13"]) == 0 + assert capsys.readouterr().out.strip() == _expected(13) diff --git a/cuda_core/build_hooks.py b/cuda_core/build_hooks.py index 545523867d1..f1b64bc62d6 100644 --- a/cuda_core/build_hooks.py +++ b/cuda_core/build_hooks.py @@ -9,6 +9,7 @@ import functools import glob +import importlib.util import os import re import sys @@ -75,6 +76,43 @@ def _get_cuda_path() -> str: return cuda_path +_PACKAGE_DIR = Path(__file__).parent / "cuda" / "core" + +# Generated at build time by _write_build_info(); read by cuda/core/__init__.py. +_BUILD_INFO_PATH = _PACKAGE_DIR / "_build_info.py" + + +@functools.cache +def _load_bindings_floor(): + """Load cuda/core/_bindings_floor.py, the floor's single source of truth. + + Loaded by file path: the package this backend builds is not importable + during its own build, and the module is deliberately import-free. + """ + path = _PACKAGE_DIR / "_bindings_floor.py" + spec = importlib.util.spec_from_file_location("_cuda_core_bindings_floor", path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def _read_cuda_h_version(cuda_path: str) -> int: + """The CUDA_VERSION macro (e.g. 13040 for 13.4) of the cuda.h under cuda_path.""" + cuda_h = os.path.join(cuda_path, "include", "cuda.h") + try: + with open(cuda_h, encoding="utf-8") as f: + for line in f: + m = re.match(r"^#\s*define\s+CUDA_VERSION\s+(\d+)\s*$", line) + if m: + return int(m.group(1)) + except OSError: + pass + raise RuntimeError( + f"Cannot read CUDA_VERSION from {cuda_h}. " + "Ensure CUDA_PATH or CUDA_HOME points to a valid CUDA installation with include/cuda.h." + ) + + @functools.cache def _determine_cuda_major_version() -> str: """Determine the CUDA major version for building cuda.core. @@ -88,7 +126,9 @@ def _determine_cuda_major_version() -> str: 2. CUDA_VERSION macro in cuda.h from CUDA_PATH or CUDA_HOME Since CUDA_PATH or CUDA_HOME is required for the build (to provide include - directories), the cuda.h header should always be available. + directories), the cuda.h header should always be available. The override + only short-circuits this detection; _check_build_configuration() still + reads the header and rejects one whose major disagrees. """ # Explicit override, e.g. in CI. cuda_major = os.environ.get("CUDA_CORE_BUILD_MAJOR") @@ -97,27 +137,99 @@ def _determine_cuda_major_version() -> str: return cuda_major # Derive from the CUDA headers (the authoritative source for what we compile against). - cuda_path = _get_cuda_path() - cuda_h = os.path.join(cuda_path, "include", "cuda.h") try: - with open(cuda_h, encoding="utf-8") as f: - for line in f: - m = re.match(r"^#\s*define\s+CUDA_VERSION\s+(\d+)\s*$", line) - if m: - v = int(m.group(1)) - # CUDA_VERSION is e.g. 12020 for 12.2. - cuda_major = str(v // 1000) - print("CUDA MAJOR VERSION:", cuda_major) - return cuda_major - except OSError: - pass + cuda_version = _read_cuda_h_version(_get_cuda_path()) + except RuntimeError as exc: + # CUDA_PATH or CUDA_HOME is required for the build, so we should not reach + # here in normal circumstances. Raise an error to make the issue clear. + raise RuntimeError( + "Cannot determine CUDA major version. " + "Set CUDA_CORE_BUILD_MAJOR environment variable, or ensure CUDA_PATH or CUDA_HOME " + "points to a valid CUDA installation with include/cuda.h." + ) from exc + # CUDA_VERSION is e.g. 12020 for 12.2. + cuda_major = str(cuda_version // 1000) + print("CUDA MAJOR VERSION:", cuda_major) + return cuda_major - # CUDA_PATH or CUDA_HOME is required for the build, so we should not reach here - # in normal circumstances. Raise an error to make the issue clear. - raise RuntimeError( - "Cannot determine CUDA major version. " - "Set CUDA_CORE_BUILD_MAJOR environment variable, or ensure CUDA_PATH or CUDA_HOME " - "points to a valid CUDA installation with include/cuda.h." + +def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: + """Reject build configurations cuda.core does not support, then record the build. + + cuda.core supports one configuration per CUDA major series: the installed + cuda-bindings is at least the series' floor (cuda/core/_bindings_floor.py) + and the cuda.h it compiles against has the same major.minor as that + cuda-bindings, which is the header cuda-bindings itself was generated from. + The pip build requirement (get_requires_for_build_*) states the floor, but + conda-forge, pixi and --no-build-isolation installs bypass it, so the check + lives here, where every build path passes. + + A too-old cuda-bindings used to surface late as an ImportError at module + init or as a feature that was silently compiled out; a mismatched header as + an unclear Cython error (see https://github.com/NVIDIA/cuda-python/issues/2783). + """ + floor = _load_bindings_floor() + major = int(cuda_major) + if major not in floor.CUDA_BINDINGS_FLOOR: + raise RuntimeError( + f"cuda.core does not support CUDA {major}; supported CUDA major versions: " + f"{', '.join(str(m) for m in floor.SUPPORTED_CUDA_MAJORS)}" + ) + requirement = floor.pip_requirement(major) + + try: + bindings_module = importlib.import_module("cuda.bindings") + except ImportError as exc: + raise RuntimeError( + f"cuda.core requires cuda-bindings to build (install '{requirement}'). " + "Isolated builds install it automatically; other builds must provide it." + ) from exc + bindings_version = bindings_module.__version__ + bindings = floor.release_triple(bindings_version) + if bindings is None: + raise RuntimeError( + f"Cannot parse the installed cuda-bindings version {bindings_version!r}. " + "A shallow git clone of cuda-bindings reports a bogus version; see CONTRIBUTING.md." + ) + if bindings[0] != major: + raise RuntimeError( + f"Building cuda.core for CUDA {major}, but the installed cuda-bindings is " + f"{bindings_version}. Install '{requirement}'." + ) + if bindings < floor.CUDA_BINDINGS_FLOOR[major]: + raise RuntimeError( + f"cuda.core requires cuda-bindings >= {floor.format_version(floor.CUDA_BINDINGS_FLOOR[major])} " + f"for CUDA {major}, but {bindings_version} is installed. Install '{requirement}'." + ) + + cuda_version = _read_cuda_h_version(cuda_path) + header = (cuda_version // 1000, cuda_version // 10 % 100) + if header != bindings[:2]: + raise RuntimeError( + f"cuda.h under {cuda_path} is CUDA {header[0]}.{header[1]}, but the installed cuda-bindings " + f"is {bindings_version}. cuda.core must be built against a cuda.h of the same " + "major.minor as its cuda-bindings (the header cuda-bindings was generated from). " + "Point CUDA_PATH or CUDA_HOME at a matching CUDA Toolkit, or install matching cuda-bindings." + ) + print(f"Build configuration: CUDA {header[0]}.{header[1]} headers, cuda-bindings {bindings_version}") + _write_build_info(major, cuda_version, floor.CUDA_BINDINGS_FLOOR[major], bindings_version) + + +def _write_build_info(cuda_major: int, cuda_version: int, floor: tuple, bindings_version: str) -> None: + """Record what this build compiled against, for the import-time check. + + cuda/core/__init__.py reads this module before it selects the versioned + subpackage and refuses an installed cuda-bindings older than the floor or + older, by minor, than the header (see _bindings_floor.required_minimum). + Like _version.py, the file is generated, gitignored, and shipped. + """ + _BUILD_INFO_PATH.write_text( + "# Generated by build_hooks.py at build time. Do not edit or commit.\n" + f"CUDA_MAJOR = {cuda_major}\n" + f"CUDA_VERSION = {cuda_version} # the cuda.h this build compiled against\n" + f"CUDA_BINDINGS_FLOOR = {tuple(floor)!r}\n" + f"CUDA_BINDINGS_BUILD_VERSION = {bindings_version!r}\n", + encoding="utf-8", ) @@ -310,6 +422,7 @@ def module_names(): # _get_cuda_path() and reads cuda.h, which must not run before the # pathfinder import has repaired PEP 517 namespace shadowing. cuda_major = _check_build_major() + _check_build_configuration(cuda_path, cuda_major) nthreads = int(os.environ.get("CUDA_PYTHON_PARALLEL_LEVEL", os.cpu_count() // 2)) compile_time_env = {"CUDA_CORE_BUILD_MAJOR": int(cuda_major)} @@ -441,8 +554,19 @@ def build_wheel(wheel_directory, config_settings=None, metadata_directory=None): def _get_cuda_bindings_require(): - cuda_major = _determine_cuda_major_version() - return [f"cuda-bindings=={cuda_major}.*"] + """The cuda-bindings build requirement: the floor of the CUDA major being built. + + Honored by isolated builds only; _check_build_configuration() enforces the + same rule for every other build path. + """ + floor = _load_bindings_floor() + cuda_major = int(_determine_cuda_major_version()) + if cuda_major not in floor.CUDA_BINDINGS_FLOOR: + raise RuntimeError( + f"cuda.core does not support CUDA {cuda_major}; supported CUDA major versions: " + f"{', '.join(str(m) for m in floor.SUPPORTED_CUDA_MAJORS)}" + ) + return [floor.pip_requirement(cuda_major)] def get_requires_for_build_editable(config_settings=None): diff --git a/cuda_core/cuda/core/__init__.py b/cuda_core/cuda/core/__init__.py index 5020f0f2a83..5c03c958611 100644 --- a/cuda_core/cuda/core/__init__.py +++ b/cuda_core/cuda/core/__init__.py @@ -6,13 +6,52 @@ def _import_versioned_module() -> None: + """Select the build for the installed cuda-bindings, after checking it is supported. + + The published wheel carries one build per CUDA major series, as the + subpackages ``cuda.core.cu12`` and ``cuda.core.cu13``; a conda or local + build carries one build at the top level. Each build records the CUDA + header it compiled against and its cuda-bindings floor in ``_build_info`` + (generated by build_hooks.py). The installed cuda-bindings must be of the + build's major and at least as new as the build's minimum + (see ``_bindings_floor.required_minimum``), or import fails here with an + actionable message instead of later with a missing C function or a + silently disabled feature. + """ import importlib - from cuda import bindings - - cuda_major = bindings.__version__.split(".")[0] - if cuda_major not in ("12", "13"): - raise ImportError("cuda.bindings 12.x or 13.x must be installed") + try: + from cuda import bindings + except ModuleNotFoundError as exc: + if exc.name in ("cuda", "cuda.bindings"): + raise ImportError("cuda.core requires cuda-bindings; install cuda-core[cu12] or cuda-core[cu13]") from None + raise + + def load_build_module(name: str, cuda_major: int): + # Prefer this major's build in the merged wheel; fall back to a plain build. + try: + return importlib.import_module(f".cu{cuda_major}.{name}", __package__) + except ModuleNotFoundError as exc: + if exc.name != f"{__package__}.cu{cuda_major}": + raise + return importlib.import_module(f".{name}", __package__) + + version_str = bindings.__version__ + # The major decides which build to consult; _bindings_floor validates everything else. + try: + cuda_major = int(version_str.split(".")[0]) + except ValueError: + cuda_major = -1 + if cuda_major not in (12, 13): + raise ImportError(f"cuda-bindings 12.x or 13.x must be installed (found {version_str})") + try: + floor = load_build_module("_bindings_floor", cuda_major) + info = load_build_module("_build_info", cuda_major) + except ModuleNotFoundError as exc: + raise ImportError( + f"this cuda.core installation has no build for CUDA {cuda_major} (installed cuda-bindings: {version_str})" + ) from exc + floor.check_installed_bindings(version_str, info.CUDA_MAJOR, info.CUDA_VERSION, __version__) subdir = f"cu{cuda_major}" try: diff --git a/cuda_core/cuda/core/_bindings_floor.py b/cuda_core/cuda/core/_bindings_floor.py new file mode 100644 index 00000000000..6e79f8392a1 --- /dev/null +++ b/cuda_core/cuda/core/_bindings_floor.py @@ -0,0 +1,122 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# SPDX-License-Identifier: Apache-2.0 + +"""The cuda-bindings version floor: the single source of truth. + +cuda.core supports two CUDA major series at a time and requires, for each, a +minimum cuda-bindings version (the *floor*) at build time and at run time. The +floor is the newest cuda-bindings release of that series that the CI source +root can build, normally the release cuda.core's own wheels are built against. +See https://github.com/NVIDIA/cuda-python/issues/2783 and the support policy. + +This module is imported by the build backend (``build_hooks.py``, by file +path, because the package is not importable during its own build), by +``cuda/core/__init__.py`` at import time, and by tests that keep the static +pins in ``pyproject.toml`` and ``ci/versions.yml`` in step. It therefore uses +the standard library only and must not import anything from ``cuda``. + +Bumping a floor is a release-note item under "Breaking Changes". Bump it in the +same PR that first uses a cuda-bindings API newer than the old floor; the CI +rows that install the floor bindings fail otherwise. +""" + +from __future__ import annotations + +import re + +__all__ = [ + "CUDA_BINDINGS_FLOOR", + "SUPPORTED_CUDA_MAJORS", + "check_installed_bindings", + "cuda_version_of", + "format_version", + "pip_requirement", + "release_triple", + "required_minimum", +] + +# Minimum cuda-bindings release per CUDA major series, as a (major, minor, patch) +# triple. Keep in step with the `cu12`/`cu13` extras in pyproject.toml (tested). +CUDA_BINDINGS_FLOOR: dict[int, tuple[int, int, int]] = { + 12: (12, 9, 8), + 13: (13, 4, 1), +} + +SUPPORTED_CUDA_MAJORS = tuple(sorted(CUDA_BINDINGS_FLOOR)) + +_RELEASE_RE = re.compile(r"^(\d+)\.(\d+)\.(\d+)") + + +def release_triple(version: str) -> tuple[int, int, int] | None: + """The leading ``major.minor.patch`` of a version string, or None. + + Pre-release and development suffixes are ignored, so ``13.4.1``, + ``13.4.1a0`` and ``13.4.2.dev249+gabcdef`` yield (13, 4, 1), (13, 4, 1) + and (13, 4, 2). A string without three leading integers (for example the + ``0.1.dev1`` that setuptools-scm reports for a shallow clone) yields None. + """ + m = _RELEASE_RE.match(version.strip()) + if m is None: + return None + return int(m.group(1)), int(m.group(2)), int(m.group(3)) + + +def format_version(triple: tuple[int, ...]) -> str: + return ".".join(str(part) for part in triple) + + +def cuda_version_of(triple: tuple[int, int, int]) -> int: + """The ``CUDA_VERSION`` macro value (e.g. 13040) for a version triple's major.minor.""" + return triple[0] * 1000 + triple[1] * 10 + + +def pip_requirement(major: int) -> str: + """The pip requirement that pins cuda-bindings to the floor and the major.""" + return f"cuda-bindings>={format_version(CUDA_BINDINGS_FLOOR[major])},=={major}.*" + + +def required_minimum(cuda_major: int, header_cuda_version: int) -> tuple[int, int, int]: + """The minimum cuda-bindings a build accepts at run time. + + A build accepts the floor of its major series, and never a cuda-bindings + whose minor is older than the ``cuda.h`` the build compiled against: the + driver function-pointer keys the C++ layer looks up in cuda-bindings are + derived from that header's macros, so an older minor may lack them. + """ + floor = CUDA_BINDINGS_FLOOR[cuda_major] + header_minor = (header_cuda_version // 1000, header_cuda_version // 10 % 100, 0) + return max(floor, header_minor) + + +def check_installed_bindings( + installed_version: str, + build_cuda_major: int, + build_cuda_version: int, + core_version: str, +) -> tuple[int, int, int]: + """Validate the installed cuda-bindings against this build; return its triple. + + Raises ImportError with an actionable message when the installed + cuda-bindings is not a release of a supported major, is not the major + this build was compiled for, or is older than the build's minimum. + """ + installed = release_triple(installed_version) + if installed is None or installed[0] not in CUDA_BINDINGS_FLOOR: + majors = " or ".join(f"{m}.x" for m in SUPPORTED_CUDA_MAJORS) + raise ImportError(f"cuda-bindings {majors} must be installed (found {installed_version})") + major = installed[0] + if major != build_cuda_major: + raise ImportError( + f"this cuda.core {core_version} build is for CUDA {build_cuda_major}, but the installed " + f"cuda-bindings is {installed_version}. Install cuda-bindings {build_cuda_major}.x, " + f"or a cuda.core build for CUDA {major}." + ) + minimum = required_minimum(major, build_cuda_version) + if installed < minimum: + floor = format_version(minimum) + raise ImportError( + f"cuda.core {core_version} requires cuda-bindings >= {floor} for CUDA {major} " + f"(found {installed_version}). Upgrade with: pip install -U 'cuda-bindings>={floor},=={major}.*'" + ) + return installed diff --git a/cuda_core/docs/source/install.rst b/cuda_core/docs/source/install.rst index c048cfb2a2c..3fdb8f71cc4 100644 --- a/cuda_core/docs/source/install.rst +++ b/cuda_core/docs/source/install.rst @@ -45,7 +45,9 @@ Starting ``cuda-core`` 0.4.0, **experimental** packages for the `free-threaded i Installing from PyPI -------------------- -``cuda.core`` works with ``cuda.bindings`` (part of ``cuda-python``) 12 or 13. Test dependencies now use the ``cuda-toolkit`` metapackage for improved dependency resolution. For example with CUDA 12: +``cuda.core`` works with ``cuda-bindings`` (part of ``cuda-python``) 12 or 13, at or above the +release's per-major floor (see :ref:`cuda-core-bindings-floor`); the ``cu12`` and ``cu13`` extras +install a compatible version. Test dependencies now use the ``cuda-toolkit`` metapackage for improved dependency resolution. For example with CUDA 12: .. code-block:: console @@ -53,8 +55,9 @@ Installing from PyPI and likewise use ``[cu13]`` for CUDA 13. -Note that using ``cuda.core`` with NVRTC installed from PyPI via ``pip install`` requires -``cuda.bindings`` 12.8.0+. Likewise, with nvJitLink it requires 12.8.0+. +Upgrading ``cuda-core`` on its own can leave an older ``cuda-bindings`` installed than the new +release requires; ``import cuda.core`` then reports the required version and the ``pip`` command +that installs it. Installing from Conda (conda-forge) @@ -68,7 +71,8 @@ Same as above, ``cuda.core`` can be installed in a CUDA 12 or 13 environment. Fo and likewise use ``cuda-version=13`` for CUDA 13. -Note that to use ``cuda.core`` with nvJitLink installed from conda-forge requires ``cuda.bindings`` 12.8.0+. +The conda-forge package pins ``cuda-bindings`` to the version it was built against, so a +compatible ``cuda-bindings`` is installed alongside it. Development environment @@ -155,7 +159,12 @@ Installing from Source $ cd cuda-python/cuda_core $ pip install . -``cuda-bindings`` 12.x or 13.x is a required dependency. +A source build requires two things to agree (see :ref:`cuda-core-bindings-floor`): + +- ``cuda-bindings`` 12.x or 13.x at or above the release's floor for that major. An isolated + build (the default ``pip install``) installs it; other builds must provide it. +- A CUDA Toolkit, located through ``CUDA_PATH`` or ``CUDA_HOME``, whose ``cuda.h`` has the same + major.minor as that ``cuda-bindings``. The build fails early otherwise. .. note:: diff --git a/cuda_core/docs/source/release/1.3.0-notes.rst b/cuda_core/docs/source/release/1.3.0-notes.rst index 31598150775..b0fa9aa4624 100644 --- a/cuda_core/docs/source/release/1.3.0-notes.rst +++ b/cuda_core/docs/source/release/1.3.0-notes.rst @@ -6,6 +6,20 @@ ``cuda.core`` 1.3.0 Release Notes ================================== +Breaking Changes +---------------- + +- ``cuda.core`` now requires a minimum ``cuda-bindings`` version per CUDA major, at build time and + at run time: 12.9.8 for CUDA 12 and 13.4.1 for CUDA 13 (see the + :ref:`support policy `). ``import cuda.core`` with an older + ``cuda-bindings`` fails with a message that names the required version and how to install it; + ``pip install cuda-core[cu12]`` / ``[cu13]`` installs a compatible version. A source build must + also use a ``cuda.h`` of the same major.minor as its ``cuda-bindings``; it fails early otherwise. + Supported CUDA drivers and CUDA Toolkit libraries are unchanged. Previously any + ``cuda-bindings`` of the right major was accepted, and an older one produced import errors for + missing C functions, silently disabled features, or crashes. + (https://github.com/NVIDIA/cuda-python/issues/2783) + New features ------------ diff --git a/cuda_core/docs/source/support.rst b/cuda_core/docs/source/support.rst index 3a6548ce204..6117ea83328 100644 --- a/cuda_core/docs/source/support.rst +++ b/cuda_core/docs/source/support.rst @@ -45,7 +45,10 @@ CUDA Version Support example, ``cuda.core`` 1.x supports CUDA 12 and 13. In particular, what this entails is that all CUDA minor versions within the two major releases -(12.x, 13.x) are supported by the same ``cuda-core`` package. +(12.x, 13.x) are supported by the same ``cuda-core`` package, at run time: any CUDA driver and any +CUDA Toolkit libraries of a supported major work with the same ``cuda-core`` wheel. The one input +this does not extend to is ``cuda-bindings``, which has a per-release minimum (see +:ref:`cuda-core-bindings-floor` below). When a new CUDA major version is released and support for the oldest major version is dropped, ``cuda.core`` will release a new major version (e.g., 1.x → 2.0.0). @@ -59,8 +62,44 @@ When a new CUDA major version is released and support for the oldest major versi - 12, 13 As with any CUDA library, certain features may impose additional requirements on the minimum -``cuda-bindings``, CUDA library, or CUDA driver versions. Refer to the individual module -documentation for details. +CUDA library or CUDA driver versions. Refer to the individual module documentation for details. + +.. _cuda-core-bindings-floor: + +``cuda-bindings`` Version Requirements +************************************** + +Each ``cuda-core`` release declares, for each supported CUDA major version, a minimum +``cuda-bindings`` version, its *floor*: the newest ``cuda-bindings`` release of that major at the +time of the ``cuda-core`` release, which is the version the published wheels are built against. +The floors of the current release are recorded in ``cuda/core/_bindings_floor.py`` and in the +``cu12``/``cu13`` extras of ``cuda-core``. + +.. list-table:: ``cuda-bindings`` floors + :header-rows: 1 + + * - ``cuda-core`` version + - CUDA 12 + - CUDA 13 + * - 1.3.x + - ``cuda-bindings`` >= 12.9.8 + - ``cuda-bindings`` >= 13.4.1 + +- **At run time**, ``import cuda.core`` requires an installed ``cuda-bindings`` of the same major + as the ``cuda-core`` build in use and at least as new as that build's floor. An older + ``cuda-bindings`` fails at import with a message that names the version found, the version + required, and the ``pip`` command that fixes it. A newer ``cuda-bindings`` of the same major is + supported. +- **At build time**, a source build requires ``cuda-bindings`` at or above the floor and a + ``cuda.h`` (``CUDA_PATH`` or ``CUDA_HOME``) of the same major.minor as that ``cuda-bindings``, + which is the header ``cuda-bindings`` itself was generated from. Any other configuration fails + the build with a message that names what was found and what is required. Building against an + older CUDA Toolkit than the floor's minor is not supported. +- **The CUDA driver** is unaffected. Feature availability is decided by the driver alone: a + feature the installed driver lacks raises when it is used, as before. + +A floor is raised only in a release that needs a newer ``cuda-bindings`` API, and every such +change is listed under "Breaking Changes" in the :doc:`release notes `. Python Version Support ---------------------- diff --git a/cuda_core/pyproject.toml b/cuda_core/pyproject.toml index 23a72a685f6..c21f81c5533 100644 --- a/cuda_core/pyproject.toml +++ b/cuda_core/pyproject.toml @@ -54,8 +54,8 @@ dependencies = [ ] [project.optional-dependencies] -cu12 = ["cuda-bindings[all]==12.*", "cuda-toolkit==12.*"] -cu13 = ["cuda-bindings[all]==13.*", "cuda-toolkit==13.*"] +cu12 = ["cuda-bindings[all]>=12.9.8,==12.*", "cuda-toolkit==12.*"] +cu13 = ["cuda-bindings[all]>=13.4.1,==13.*", "cuda-toolkit==13.*"] [dependency-groups] test = [ diff --git a/cuda_core/tests/test_bindings_floor.py b/cuda_core/tests/test_bindings_floor.py new file mode 100644 index 00000000000..d01700d63ec --- /dev/null +++ b/cuda_core/tests/test_bindings_floor.py @@ -0,0 +1,161 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# SPDX-License-Identifier: Apache-2.0 + +"""The cuda-bindings version floor (cuda/core/_bindings_floor.py) and the +import-time check built on it. + +Source-tree properties and pure functions only: no GPU, so this file also runs +with --noconftest (conftest.py initializes CUDA). The consistency tests read +pyproject.toml and ci/versions.yml from the checkout, so they need the source +tree next to the tests, which every CI job that runs tests/ has. + + pytest tests/test_bindings_floor.py -v --noconftest +""" + +import re +from pathlib import Path + +import pytest + +from cuda.core import _bindings_floor as floor_mod +from cuda.core._bindings_floor import ( + CUDA_BINDINGS_FLOOR, + SUPPORTED_CUDA_MAJORS, + check_installed_bindings, + cuda_version_of, + format_version, + pip_requirement, + release_triple, + required_minimum, +) + +CUDA_CORE = Path(__file__).resolve().parent.parent +REPO = CUDA_CORE.parent + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_floor_module_is_import_free(): + """build_hooks.py loads it by file path during the build; it must stay standard-library only.""" + source = Path(floor_mod.__file__).read_text(encoding="utf-8") + imports = re.findall(r"^\s*(?:from|import)\s+(\w+)", source, re.M) + assert set(imports) <= {"__future__", "re"} + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_floors_are_release_triples_of_their_major(): + assert SUPPORTED_CUDA_MAJORS == (12, 13) + for major, floor in CUDA_BINDINGS_FLOOR.items(): + assert len(floor) == 3 + assert floor[0] == major + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +@pytest.mark.parametrize( + ("version", "expected"), + [ + ("13.4.1", (13, 4, 1)), + ("13.4.1a0", (13, 4, 1)), + ("13.4.2.dev249+g471618971c2", (13, 4, 2)), + ("12.9.8", (12, 9, 8)), + (" 12.9.9.dev2 ", (12, 9, 9)), + ("0.1.dev1+g0d22cb444", None), # shallow clone + ("13.4", None), + ("", None), + ], +) +def test_release_triple(version, expected): + assert release_triple(version) == expected + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_formatting_helpers(): + assert format_version((13, 4, 1)) == "13.4.1" + assert cuda_version_of((13, 4, 1)) == 13040 + assert cuda_version_of((12, 9, 8)) == 12090 + assert pip_requirement(13) == f"cuda-bindings>={format_version(CUDA_BINDINGS_FLOOR[13])},==13.*" + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_required_minimum_is_the_floor_or_the_header_minor(): + floor = CUDA_BINDINGS_FLOOR[13] + header_at_floor = cuda_version_of(floor) + assert required_minimum(13, header_at_floor) == floor + # A build against a newer header than the floor's minor demands that minor: + # the driver-pointer keys are derived from the header's macros. + assert required_minimum(13, header_at_floor + 10) == (13, floor[1] + 1, 0) + # An older header cannot win over the floor. + assert required_minimum(13, 13000) == floor + + +class TestCheckInstalledBindings: + FLOOR = CUDA_BINDINGS_FLOOR[13] + HEADER = cuda_version_of(FLOOR) + + @pytest.mark.agent_authored(model="claude-fable-5-1") + @pytest.mark.parametrize( + "installed", + [ + format_version(FLOOR), + f"{FLOOR[0]}.{FLOOR[1]}.{FLOOR[2] + 1}", + f"{FLOOR[0]}.{FLOOR[1]}.{FLOOR[2] + 1}.dev249+gabcdef0", # main-built bindings in CI + f"{FLOOR[0]}.{FLOOR[1] + 1}.0b1", # newer bindings than the build: supported + ], + ) + def test_accepts_the_floor_and_newer(self, installed): + assert check_installed_bindings(installed, 13, self.HEADER, "1.3.0") == release_triple(installed) + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_rejects_older_than_the_floor_with_the_fix(self): + older = f"{self.FLOOR[0]}.{self.FLOOR[1] - 1}.1" + with pytest.raises(ImportError) as excinfo: + check_installed_bindings(older, 13, self.HEADER, "1.3.0") + message = str(excinfo.value) + assert f"requires cuda-bindings >= {format_version(self.FLOOR)} for CUDA 13" in message + assert f"(found {older})" in message + assert f"pip install -U 'cuda-bindings>={format_version(self.FLOOR)},==13.*'" in message + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_rejects_a_minor_older_than_the_header(self): + # Built against a header one minor above the floor; the floor itself no longer suffices. + header = self.HEADER + 10 + with pytest.raises(ImportError, match=rf"requires cuda-bindings >= 13\.{self.FLOOR[1] + 1}\.0"): + check_installed_bindings(format_version(self.FLOOR), 13, header, "1.3.0") + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_rejects_another_major_than_the_build(self): + with pytest.raises(ImportError, match="build is for CUDA 12, but the installed cuda-bindings is 13.4.1"): + check_installed_bindings("13.4.1", 12, 12090, "1.3.0") + + @pytest.mark.agent_authored(model="claude-fable-5-1") + @pytest.mark.parametrize("installed", ["0.1.dev1+g0d22cb444", "11.8.0", "14.0.0", "garbage"]) + def test_rejects_unsupported_or_unparseable_versions(self, installed): + with pytest.raises( + ImportError, match=rf"cuda-bindings 12\.x or 13\.x must be installed \(found {re.escape(installed)}\)" + ): + check_installed_bindings(installed, 13, self.HEADER, "1.3.0") + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_pyproject_extras_pin_the_floor(): + """The static `cu12`/`cu13` extras cannot read the module; keep them in step by test.""" + pyproject = (CUDA_CORE / "pyproject.toml").read_text(encoding="utf-8") + for major in SUPPORTED_CUDA_MAJORS: + m = re.search(rf'^cu{major} = \["cuda-bindings\[all\]([^"]+)"', pyproject, re.M) + assert m, f"no cu{major} extra in pyproject.toml" + assert m.group(1) == pip_requirement(major).removeprefix("cuda-bindings") + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_ci_toolkit_pins_match_the_floors_minor(): + """CI builds each major against the toolkit pinned in ci/versions.yml; the + build requires that header's major.minor to equal the bindings', so the + floor of each major must sit in the same minor as its toolkit pin.""" + versions = (REPO / "ci" / "versions.yml").read_text(encoding="utf-8") + pins = dict(re.findall(r"^\s+(build|prev_build):\s*\n\s+version:\s*\"(\d+\.\d+)", versions, re.M)) + assert set(pins) == {"build", "prev_build"}, pins + by_major = {int(v.split(".")[0]): v for v in pins.values()} + for major, floor in CUDA_BINDINGS_FLOOR.items(): + assert by_major[major] == f"{floor[0]}.{floor[1]}", ( + f"ci/versions.yml builds CUDA {major} against {by_major[major]} but the floor is {format_version(floor)}" + ) diff --git a/cuda_core/tests/test_build_hooks.py b/cuda_core/tests/test_build_hooks.py index 27c057e0297..429bf358737 100644 --- a/cuda_core/tests/test_build_hooks.py +++ b/cuda_core/tests/test_build_hooks.py @@ -232,6 +232,9 @@ def fake_cythonize(ext_modules, **kwargs): # Builds resolve the CTK for include dirs; stub it so the test runs # where no toolkit is installed (e.g. the wheels CI jobs). monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: "/nonexistent-cuda") + # The configuration check reads that header and the installed cuda-bindings; + # it has its own tests (TestBuildConfigurationCheck). + monkeypatch.setattr(build_hooks, "_check_build_configuration", lambda cuda_path, cuda_major: None) monkeypatch.setattr(build_hooks, "cythonize", fake_cythonize) monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", cuda_major) build_hooks._determine_cuda_major_version.cache_clear() @@ -451,3 +454,123 @@ def test_serial_builds_and_compilers_without_the_hook_keep_the_stock_path(self, cmd = self._build_ext(monkeypatch, 4, self.MsvcLikeCompiler()) with cmd._parallel_source_compilation(): assert cmd.compiler.compile(["a.cpp"]) == "stock" + + +def _fake_bindings(monkeypatch, version): + """Install a stand-in cuda.bindings whose __version__ is `version`.""" + import types + + module = types.ModuleType("cuda.bindings") + module.__version__ = version + monkeypatch.setitem(sys.modules, "cuda.bindings", module) + + +def _write_cuda_h(tmp_path, cuda_version): + include = tmp_path / "include" + include.mkdir(exist_ok=True) + (include / "cuda.h").write_text(f"#define CUDA_VERSION {cuda_version}\n") + return str(tmp_path) + + +class TestBuildConfigurationCheck: + """_check_build_configuration() accepts exactly one configuration per CUDA + major: cuda-bindings at or above the floor, and a cuda.h of the same + major.minor as that cuda-bindings. Anything else is a build error that + names what was found and what is required.""" + + FLOOR = build_hooks._load_bindings_floor().CUDA_BINDINGS_FLOOR + + @pytest.fixture(autouse=True) + def _isolate_build_info(self, tmp_path, monkeypatch): + monkeypatch.setattr(build_hooks, "_BUILD_INFO_PATH", tmp_path / "_build_info.py") + + @pytest.mark.agent_authored(model="claude-fable-5-1") + @pytest.mark.parametrize("major", [12, 13]) + def test_floor_bindings_and_matching_header_pass_and_are_recorded(self, tmp_path, monkeypatch, major): + floor = self.FLOOR[major] + version = f"{floor[0]}.{floor[1]}.{floor[2] + 1}.dev3+gabcdef0" + _fake_bindings(monkeypatch, version) + cuda_path = _write_cuda_h(tmp_path, floor[0] * 1000 + floor[1] * 10) + + build_hooks._check_build_configuration(cuda_path, str(major)) + + info = {} + exec(build_hooks._BUILD_INFO_PATH.read_text(), info) + assert info["CUDA_MAJOR"] == major + assert info["CUDA_VERSION"] == floor[0] * 1000 + floor[1] * 10 + assert info["CUDA_BINDINGS_FLOOR"] == floor + assert info["CUDA_BINDINGS_BUILD_VERSION"] == version + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_bindings_below_the_floor_fail(self, tmp_path, monkeypatch): + floor = self.FLOOR[13] + _fake_bindings(monkeypatch, f"{floor[0]}.{floor[1]}.{floor[2] - 1}" if floor[2] else "13.0.0") + cuda_path = _write_cuda_h(tmp_path, 13040) + with pytest.raises(RuntimeError, match=r"requires cuda-bindings >= 13\.\d+\.\d+ for CUDA 13"): + build_hooks._check_build_configuration(cuda_path, "13") + assert not build_hooks._BUILD_INFO_PATH.exists() + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_bindings_of_another_major_fail(self, tmp_path, monkeypatch): + _fake_bindings(monkeypatch, "13.4.1") + cuda_path = _write_cuda_h(tmp_path, 12090) + with pytest.raises( + RuntimeError, match="Building cuda.core for CUDA 12, but the installed cuda-bindings is 13.4.1" + ): + build_hooks._check_build_configuration(cuda_path, "12") + + @pytest.mark.agent_authored(model="claude-fable-5-1") + @pytest.mark.parametrize("header", [13030, 13050, 12090]) + def test_header_minor_must_match_bindings(self, tmp_path, monkeypatch, header): + _fake_bindings(monkeypatch, "13.4.1") + cuda_path = _write_cuda_h(tmp_path, header) + with pytest.raises(RuntimeError, match="same major.minor as its cuda-bindings"): + build_hooks._check_build_configuration(cuda_path, "13") + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_header_is_read_even_when_the_major_override_is_set(self, tmp_path, monkeypatch): + # CUDA_CORE_BUILD_MAJOR skips header detection of the major, not this check. + monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", "13") + _fake_bindings(monkeypatch, "13.4.1") + cuda_path = _write_cuda_h(tmp_path, 13030) + with pytest.raises(RuntimeError, match="same major.minor"): + build_hooks._check_build_configuration(cuda_path, "13") + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_missing_bindings_is_a_build_error(self, tmp_path, monkeypatch): + monkeypatch.setitem(sys.modules, "cuda.bindings", None) # makes `import cuda.bindings` fail + cuda_path = _write_cuda_h(tmp_path, 13040) + with pytest.raises(RuntimeError, match="requires cuda-bindings to build"): + build_hooks._check_build_configuration(cuda_path, "13") + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_unparseable_bindings_version_is_a_build_error(self, tmp_path, monkeypatch): + _fake_bindings(monkeypatch, "0.1.dev1+g0d22cb444") # a shallow clone of cuda-bindings + cuda_path = _write_cuda_h(tmp_path, 13040) + with pytest.raises(RuntimeError, match="Cannot parse the installed cuda-bindings version"): + build_hooks._check_build_configuration(cuda_path, "13") + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_unsupported_major_is_a_build_error(self, tmp_path, monkeypatch): + _fake_bindings(monkeypatch, "14.0.0") + cuda_path = _write_cuda_h(tmp_path, 14000) + with pytest.raises(RuntimeError, match="does not support CUDA 14"): + build_hooks._check_build_configuration(cuda_path, "14") + + +class TestBuildRequirement: + @pytest.mark.agent_authored(model="claude-fable-5-1") + @pytest.mark.parametrize("major", ["12", "13"]) + def test_pins_the_floor_and_the_major(self, monkeypatch, major): + monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", major) + build_hooks._determine_cuda_major_version.cache_clear() + floor = build_hooks._load_bindings_floor().CUDA_BINDINGS_FLOOR[int(major)] + (requirement,) = build_hooks._get_cuda_bindings_require() + assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},=={major}.*" + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_unsupported_major_names_the_supported_ones(self, monkeypatch): + monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", "11") + build_hooks._determine_cuda_major_version.cache_clear() + with pytest.raises(RuntimeError, match="does not support CUDA 11.*12, 13"): + build_hooks._get_cuda_bindings_require() From f4757aa820925031e42395ce331db7caa05440cd Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Fri, 18 Sep 2026 12:35:25 -0700 Subject: [PATCH 02/18] cuda.core: fence C++ on the CUDA major only, never on CUDA_VERSION The two `#if CUDA_VERSION >= 130x0` fences in _cpp/rt/driver_api.* compiled cuDevSmResourceSplit and cuMemcpyWithAttributesAsync out of a source build against an older 13.x header, while the run-time gates, which looked at the bindings and the driver, still reported the features as available; the calls then failed with CUDA_ERROR_NOT_SUPPORTED (#2783, "guards disable working features"). Both features also went through void*-typed C++ shims and has_*() presence probes whose only purpose was to avoid an eager cimport of a cydriver function that cuda-bindings 13.0 did not export. With the cuda-bindings floor in place (13.4.1 for CUDA 13), the cimport is safe, so the shims, the presence probes and the two fenced pointers go away: _device_resources.pyx and _buffer.pyx call cydriver.cuDevSmResourceSplit and cydriver.cuMemcpyWithAttributesAsync directly under the existing `IF CUDA_CORE_BUILD_MAJOR >= 13`, gated on the driver version alone. build_hooks.py now defines CUDA_CORE_BUILD_MAJOR (until now a Cython-only compile-time constant) and CUDA_CORE_MIN_CUDA_VERSION (the floor's major.minor) for the C++ compiler. The new _cpp/rt/versions.hpp, the first include of the tree via types.hpp, re-checks cuda.h against both with #error, so a build that bypasses the backend still cannot compile against an unsupported header. tests/test_rt_layout.py enforces that versions.hpp is the only file under _cpp/ that names CUDA_VERSION. Co-Authored-By: Claude Fable 5.1 --- cuda_core/build_hooks.py | 25 ++++++-- cuda_core/cuda/core/_cpp/rt/DESIGN.md | 19 ++++++ cuda_core/cuda/core/_cpp/rt/driver_api.cpp | 58 ------------------- cuda_core/cuda/core/_cpp/rt/driver_api.hpp | 53 ----------------- cuda_core/cuda/core/_cpp/rt/types.hpp | 1 + cuda_core/cuda/core/_cpp/rt/versions.hpp | 48 +++++++++++++++ cuda_core/cuda/core/_device_resources.pyx | 15 ++--- cuda_core/cuda/core/_memory/_buffer.pyx | 7 +-- .../cuda/core/_memory/_copy_attributes.pxd | 17 ++---- cuda_core/cuda/core/_rt.pxd | 17 ------ cuda_core/cuda/core/_rt.pyx | 30 ---------- cuda_core/tests/test_build_hooks.py | 31 ++++++++++ cuda_core/tests/test_rt_layout.py | 16 ++++- 13 files changed, 144 insertions(+), 193 deletions(-) create mode 100644 cuda_core/cuda/core/_cpp/rt/versions.hpp diff --git a/cuda_core/build_hooks.py b/cuda_core/build_hooks.py index f1b64bc62d6..70938f4091e 100644 --- a/cuda_core/build_hooks.py +++ b/cuda_core/build_hooks.py @@ -215,6 +215,16 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: _write_build_info(major, cuda_version, floor.CUDA_BINDINGS_FLOOR[major], bindings_version) +def _build_define_macros(cuda_major: str) -> list: + """Preprocessor macros that carry the build decision into the C++ (see _cpp/rt/versions.hpp).""" + floor = _load_bindings_floor() + major = int(cuda_major) + return [ + ("CUDA_CORE_BUILD_MAJOR", str(major)), + ("CUDA_CORE_MIN_CUDA_VERSION", str(floor.cuda_version_of(floor.CUDA_BINDINGS_FLOOR[major]))), + ] + + def _write_build_info(cuda_major: int, cuda_version: int, floor: tuple, bindings_version: str) -> None: """Record what this build compiled against, for the import-time check. @@ -400,6 +410,12 @@ def module_names(): # related to free-threading builds. extra_compile_args += ["-DCYTHON_TRACE_NOGIL=1", "-DCYTHON_USE_SYS_MONITORING=0"] + # Deliberately after the cuda.bindings import above: this re-enters + # _get_cuda_path() and reads cuda.h, which must not run before the + # pathfinder import has repaired PEP 517 namespace shadowing. + cuda_major = _check_build_major() + _check_build_configuration(cuda_path, cuda_major) + depends = _extension_depends() ext_modules = tuple( Extension( @@ -411,6 +427,9 @@ def module_names(): "cuda/core/_cpp", ] + all_include_dirs, + # The C++ branches on the CUDA major series only; _cpp/rt/versions.hpp + # re-checks cuda.h against both macros (see _check_build_configuration). + define_macros=_build_define_macros(cuda_major), language="c++", extra_compile_args=extra_compile_args, extra_link_args=extra_link_args, @@ -418,12 +437,6 @@ def module_names(): for mod in module_names() ) - # Deliberately after the cuda.bindings import above: this re-enters - # _get_cuda_path() and reads cuda.h, which must not run before the - # pathfinder import has repaired PEP 517 namespace shadowing. - cuda_major = _check_build_major() - _check_build_configuration(cuda_path, cuda_major) - nthreads = int(os.environ.get("CUDA_PYTHON_PARALLEL_LEVEL", os.cpu_count() // 2)) compile_time_env = {"CUDA_CORE_BUILD_MAJOR": int(cuda_major)} compiler_directives = {"embedsignature": True, "warn.deprecated.IF": False, "freethreading_compatible": True} diff --git a/cuda_core/cuda/core/_cpp/rt/DESIGN.md b/cuda_core/cuda/core/_cpp/rt/DESIGN.md index 0374d5a08be..6c1194f9cd8 100644 --- a/cuda_core/cuda/core/_cpp/rt/DESIGN.md +++ b/cuda_core/cuda/core/_cpp/rt/DESIGN.md @@ -205,6 +205,25 @@ This approach: will return errors like `CUDA_ERROR_NO_DEVICE`) - Requires no custom capsule infrastructure—uses Cython's built-in mechanism +## Build-time version guards + +cuda.core supports one build configuration per CUDA major series: the `cuda.h` +it compiles against has the same major.minor as the cuda-bindings it is built +with, and that cuda-bindings is at or above the series' floor +(`cuda/core/_bindings_floor.py`). `build_hooks.py` enforces both before +compiling and defines `CUDA_CORE_BUILD_MAJOR` and `CUDA_CORE_MIN_CUDA_VERSION` +for the C++ compiler; `versions.hpp`, the first include of the tree, re-checks +`cuda.h` against them with `#error`. + +The C++ branches on `CUDA_CORE_BUILD_MAJOR` only, and only where the two major +series differ. Minor-version fences (`#if CUDA_VERSION >= 130x0`) are not +allowed: they compiled features out of source builds against an older header +while the run-time checks, which looked at the bindings and the driver, never +noticed (https://github.com/NVIDIA/cuda-python/issues/2783). Whether the +*driver* provides a function is decided by the driver-version gates in Cython, +never by the C++ layer. `tests/test_rt_layout.py` enforces that `versions.hpp` +is the only file under `_cpp/` that names `CUDA_VERSION`. + ## Key Implementation Details ### Structural Dependencies diff --git a/cuda_core/cuda/core/_cpp/rt/driver_api.cpp b/cuda_core/cuda/core/_cpp/rt/driver_api.cpp index 860a29a3554..34b080c6658 100644 --- a/cuda_core/cuda/core/_cpp/rt/driver_api.cpp +++ b/cuda_core/cuda/core/_cpp/rt/driver_api.cpp @@ -94,20 +94,6 @@ decltype(&cuTexObjectDestroy) p_cuTexObjectDestroy = nullptr; decltype(&cuSurfObjectCreate) p_cuSurfObjectCreate = nullptr; decltype(&cuSurfObjectDestroy) p_cuSurfObjectDestroy = nullptr; -// SM resource split (13.1+ — may be null on older drivers/bindings) -#if CUDA_VERSION >= 13010 -decltype(&cuDevSmResourceSplit) p_cuDevSmResourceSplit = nullptr; -#else -void* p_cuDevSmResourceSplit = nullptr; -#endif - -// cuMemcpyWithAttributesAsync (13.2+ — may be null on older drivers/bindings) -#if CUDA_VERSION >= 13020 -decltype(&cuMemcpyWithAttributesAsync) p_cuMemcpyWithAttributesAsync = nullptr; -#else -void* p_cuMemcpyWithAttributesAsync = nullptr; -#endif - // NVRTC function pointers decltype(&nvrtcDestroyProgram) p_nvrtcDestroyProgram = nullptr; @@ -117,48 +103,4 @@ NvvmDestroyProgramFn p_nvvmDestroyProgram = nullptr; // nvJitLink function pointers (may be null if nvJitLink is not available) NvJitLinkDestroyFn p_nvJitLinkDestroy = nullptr; -// ============================================================================ -// SM resource split wrapper -// ============================================================================ - -CUresult sm_resource_split(CUdevResource* result, unsigned int nbGroups, - const CUdevResource* input, CUdevResource* remainder, - unsigned int flags, void* groupParams) { -#if CUDA_VERSION >= 13010 - if (!p_cuDevSmResourceSplit) { - return CUDA_ERROR_NOT_SUPPORTED; - } - return p_cuDevSmResourceSplit( - result, nbGroups, input, remainder, flags, - static_cast(groupParams)); -#else - return CUDA_ERROR_NOT_SUPPORTED; -#endif -} - -bool has_sm_resource_split() noexcept { - return p_cuDevSmResourceSplit != nullptr; -} - -// ============================================================================ -// cuMemcpyWithAttributesAsync wrapper -// ============================================================================ - -CUresult memcpy_with_attributes_async(CUdeviceptr dst, CUdeviceptr src, size_t size, - void* attr, CUstream hStream) { -#if CUDA_VERSION >= 13020 - if (!p_cuMemcpyWithAttributesAsync) { - return CUDA_ERROR_NOT_SUPPORTED; - } - return p_cuMemcpyWithAttributesAsync( - dst, src, size, static_cast(attr), hStream); -#else - return CUDA_ERROR_NOT_SUPPORTED; -#endif -} - -bool has_memcpy_with_attributes_async() noexcept { - return p_cuMemcpyWithAttributesAsync != nullptr; -} - } // namespace cuda_core::rt diff --git a/cuda_core/cuda/core/_cpp/rt/driver_api.hpp b/cuda_core/cuda/core/_cpp/rt/driver_api.hpp index 87ed3bd7906..475b1b15579 100644 --- a/cuda_core/cuda/core/_cpp/rt/driver_api.hpp +++ b/cuda_core/cuda/core/_cpp/rt/driver_api.hpp @@ -99,24 +99,6 @@ extern decltype(&cuTexObjectDestroy) p_cuTexObjectDestroy; extern decltype(&cuSurfObjectCreate) p_cuSurfObjectCreate; extern decltype(&cuSurfObjectDestroy) p_cuSurfObjectDestroy; -// SM resource split (13.1+ — may be null on older drivers/bindings) -#if CUDA_VERSION >= 13010 -extern decltype(&cuDevSmResourceSplit) p_cuDevSmResourceSplit; -#else -// cuDevSmResourceSplit doesn't exist in CUDA < 13.1 headers, so use a -// void* placeholder. The pointer is always null when built against 12.x. -extern void* p_cuDevSmResourceSplit; -#endif - -// cuMemcpyWithAttributesAsync (13.2+ — may be null on older drivers/bindings) -#if CUDA_VERSION >= 13020 -extern decltype(&cuMemcpyWithAttributesAsync) p_cuMemcpyWithAttributesAsync; -#else -// cuMemcpyWithAttributesAsync doesn't exist in CUDA < 13.2 headers, so use a -// void* placeholder. The pointer is always null when built against older CUDA. -extern void* p_cuMemcpyWithAttributesAsync; -#endif - // ============================================================================ // NVRTC function pointers // @@ -152,39 +134,4 @@ extern NvvmDestroyProgramFn p_nvvmDestroyProgram; using NvJitLinkDestroyFn = int (*)(nvJitLink_t*); extern NvJitLinkDestroyFn p_nvJitLinkDestroy; -// ============================================================================ -// SM resource split wrapper (13.1+) -// -// Calls through p_cuDevSmResourceSplit if available, otherwise returns -// CUDA_ERROR_NOT_SUPPORTED. This avoids a direct Cython cimport of the -// cydriver cdef function, which would fail at module init on cuda-bindings -// < 13.1 (see https://github.com/NVIDIA/cuda-python/issues/2063). -// ============================================================================ - -// groupParams is void* so the Cython declaration doesn't reference -// CU_DEV_SM_RESOURCE_GROUP_PARAMS (absent from cuda-bindings 13.0 .pxd). -CUresult sm_resource_split(CUdevResource* result, unsigned int nbGroups, - const CUdevResource* input, CUdevResource* remainder, - unsigned int flags, void* groupParams); - -// Returns true if the cuDevSmResourceSplit function pointer is available. -bool has_sm_resource_split() noexcept; - -// ============================================================================ -// cuMemcpyWithAttributesAsync wrapper (13.2+) -// -// Calls through p_cuMemcpyWithAttributesAsync if available, otherwise returns -// CUDA_ERROR_NOT_SUPPORTED. This avoids a direct Cython cimport of the -// cydriver cdef function, which would fail at module init on cuda-bindings -// < 13.2 (see https://github.com/NVIDIA/cuda-python/issues/2063). -// ============================================================================ - -// attr is void* so the Cython declaration doesn't reference CUmemcpyAttributes -// (absent from cuda-bindings built against CUDA < 12.8). The C++ side casts it. -CUresult memcpy_with_attributes_async(CUdeviceptr dst, CUdeviceptr src, size_t size, - void* attr, CUstream hStream); - -// Returns true if the cuMemcpyWithAttributesAsync function pointer is available. -bool has_memcpy_with_attributes_async() noexcept; - } // namespace cuda_core::rt diff --git a/cuda_core/cuda/core/_cpp/rt/types.hpp b/cuda_core/cuda/core/_cpp/rt/types.hpp index 59389f78fca..9b1ebaa6468 100644 --- a/cuda_core/cuda/core/_cpp/rt/types.hpp +++ b/cuda_core/cuda/core/_cpp/rt/types.hpp @@ -4,6 +4,7 @@ #pragma once +#include "versions.hpp" #include #include #include diff --git a/cuda_core/cuda/core/_cpp/rt/versions.hpp b/cuda_core/cuda/core/_cpp/rt/versions.hpp new file mode 100644 index 00000000000..eda836cb507 --- /dev/null +++ b/cuda_core/cuda/core/_cpp/rt/versions.hpp @@ -0,0 +1,48 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// +// SPDX-License-Identifier: Apache-2.0 + +#pragma once + +// The one place the C++ under _cpp/ consults CUDA_VERSION. +// +// cuda.core supports one build configuration per CUDA major series: the +// cuda.h it compiles against has the same major.minor as the cuda-bindings it +// is built with, and that cuda-bindings is at or above the series' floor +// (cuda/core/_bindings_floor.py; https://github.com/NVIDIA/cuda-python/issues/2783). +// build_hooks.py enforces both before compiling and passes the decision down +// as two macros: +// +// CUDA_CORE_BUILD_MAJOR the CUDA major series being built (12 or 13); +// the only version the C++ may branch on, as +// `#if CUDA_CORE_BUILD_MAJOR >= 13`, and always +// for a difference between major series. +// CUDA_CORE_MIN_CUDA_VERSION the floor's major.minor as a CUDA_VERSION +// value (e.g. 13040). +// +// This header re-checks the header against both macros so that a build that +// bypasses build_hooks.py still cannot compile against an unsupported header. +// Minor-version fences (`#if CUDA_VERSION >= 130x0`) are not allowed anywhere +// else: they compiled features out of source builds against an older header +// while the run-time checks, which looked at the bindings and the driver, +// never noticed. tests/test_rt_layout.py enforces that this is the only file +// that names CUDA_VERSION. +// +// Downstream Cython code that cimports _rt includes this header without the +// macros; it then only learns the major from cuda.h and skips the floor check. + +#include + +#ifndef CUDA_CORE_BUILD_MAJOR +#define CUDA_CORE_BUILD_MAJOR (CUDA_VERSION / 1000) +#endif + +#if (CUDA_VERSION / 1000) != CUDA_CORE_BUILD_MAJOR +#error "cuda.h does not belong to the CUDA major series cuda.core is being built for (CUDA_CORE_BUILD_MAJOR)" +#endif + +#ifdef CUDA_CORE_MIN_CUDA_VERSION +#if CUDA_VERSION < CUDA_CORE_MIN_CUDA_VERSION +#error "cuda.h is older than the minimum this cuda.core release supports for its CUDA major series (see the cuda.core support policy)" +#endif +#endif diff --git a/cuda_core/cuda/core/_device_resources.pyx b/cuda_core/cuda/core/_device_resources.pyx index c9fe33c4fd7..1d697a0c6d6 100644 --- a/cuda_core/cuda/core/_device_resources.pyx +++ b/cuda_core/cuda/core/_device_resources.pyx @@ -225,20 +225,17 @@ cdef inline unsigned int _to_sm_count(object value) except? 0: return (value) -IF CUDA_CORE_BUILD_MAJOR >= 13: - from cuda.core._rt cimport sm_resource_split, has_sm_resource_split - cdef int _structured_split_checked = 0 cdef inline bint _can_use_structured_sm_split(): - """Check if cuDevSmResourceSplit (13.1+) is available. Cached.""" + """Whether the driver provides cuDevSmResourceSplit (13.1+). Cached. + + cuda-bindings 13.4+ (the floor) always exports it; only the driver can lack it.""" global _structured_split_checked if _structured_split_checked != 0: return _structured_split_checked == 1 IF CUDA_CORE_BUILD_MAJOR >= 13: - if (has_sm_resource_split() - and cy_driver_version() >= (13, 1, 0) - and cy_binding_version() >= (13, 1, 0)): + if cy_driver_version() >= (13, 1, 0): _structured_split_checked = 1 return True _structured_split_checked = -1 @@ -326,13 +323,13 @@ IF CUDA_CORE_BUILD_MAJOR >= 13: memset(&remaining, 0, sizeof(cydriver.CUdevResource)) with nogil: - HANDLE_RETURN(sm_resource_split( + HANDLE_RETURN(cydriver.cuDevSmResourceSplit( result, (n_groups), &sm._resource, &remaining, 0, - params, + params, )) if result != NULL: diff --git a/cuda_core/cuda/core/_memory/_buffer.pyx b/cuda_core/cuda/core/_memory/_buffer.pyx index ae4546eb5be..e6c4c6710bb 100644 --- a/cuda_core/cuda/core/_memory/_buffer.pyx +++ b/cuda_core/cuda/core/_memory/_buffer.pyx @@ -30,9 +30,6 @@ from cuda.core.typing import DevicePointerType from cuda.core._memory._copy_attributes cimport _with_attributes_available from cuda.core._memory._copy_attributes cimport _to_cu_memcpy_attributes # no-cython-lint -IF CUDA_CORE_BUILD_MAJOR >= 13: - from cuda.core._rt cimport memcpy_with_attributes_async - from cuda.core._stream cimport Stream, Stream_accept, Stream_is_legacy_default_token, default_stream from cuda.core._utils.cuda_utils cimport HANDLE_RETURN, _parse_fill_value @@ -187,11 +184,9 @@ cdef void _do_copy_with_attributes( object options, cydriver.CUstream hstream, ): IF CUDA_CORE_BUILD_MAJOR >= 13: - # Routed through the memcpy_with_attributes_async() C++ shim since - # cydriver.cuMemcpyWithAttributesAsync is absent from cuda-bindings < 13.2. cdef cydriver.CUmemcpyAttributes cu_attr = _to_cu_memcpy_attributes(options) with nogil: - HANDLE_RETURN(memcpy_with_attributes_async(dst, src, nbytes, &cu_attr, hstream)) + HANDLE_RETURN(cydriver.cuMemcpyWithAttributesAsync(dst, src, nbytes, &cu_attr, hstream)) ELSE: pass # unreachable: _with_attributes_available() is always False on CUDA 12 diff --git a/cuda_core/cuda/core/_memory/_copy_attributes.pxd b/cuda_core/cuda/core/_memory/_copy_attributes.pxd index 3e4ae2e778f..e85d1b71a99 100644 --- a/cuda_core/cuda/core/_memory/_copy_attributes.pxd +++ b/cuda_core/cuda/core/_memory/_copy_attributes.pxd @@ -7,23 +7,14 @@ # without either depending on the other. from cuda.bindings cimport cydriver -from cuda.core._utils.version cimport cy_binding_version, cy_driver_version # no-cython-lint +from cuda.core._utils.version cimport cy_driver_version # no-cython-lint IF CUDA_CORE_BUILD_MAJOR >= 13: - from cuda.core._rt cimport has_memcpy_with_attributes_async - cdef inline bint _with_attributes_available(): - # has_memcpy_with_attributes_async() says whether the installed - # cuda-bindings actually exports cuMemcpyWithAttributesAsync (13.2+); - # the version checks alone are not sufficient, since cuda.core's build - # can be paired with a cuda-bindings install older than what it built - # against (see https://github.com/NVIDIA/cuda-python/issues/2063). - return ( - has_memcpy_with_attributes_async() - and cy_driver_version() >= (13, 2, 0) - and cy_binding_version() >= (13, 2, 0) - ) + # cuMemcpyWithAttributesAsync is a 13.2 driver API; cuda-bindings 13.4+ + # (the floor) always exports it, so only the driver can lack it. + return cy_driver_version() >= (13, 2, 0) ELSE: cdef inline bint _with_attributes_available(): return False diff --git a/cuda_core/cuda/core/_rt.pxd b/cuda_core/cuda/core/_rt.pxd index 76082d7ec0e..e12cefb0f71 100644 --- a/cuda_core/cuda/core/_rt.pxd +++ b/cuda_core/cuda/core/_rt.pxd @@ -375,20 +375,3 @@ cdef TexObjectHandle create_tex_object_handle_linear( cdef SurfObjectHandle create_surf_object_handle( const ContextHandle& h_context, const cydriver.CUDA_RESOURCE_DESC& res, const OpaqueArrayHandle& h_backing) except+ nogil - -# SM resource split (13.1+ — calls through function pointer, safe on older bindings) -# groupParams is void* here to avoid referencing CU_DEV_SM_RESOURCE_GROUP_PARAMS -# (which doesn't exist in cuda-bindings 13.0 .pxd). The C++ side casts it. -cdef cydriver.CUresult sm_resource_split( - cydriver.CUdevResource* result, unsigned int nbGroups, - const cydriver.CUdevResource* input, cydriver.CUdevResource* remainder, - unsigned int flags, void* groupParams) nogil -cdef bint has_sm_resource_split() noexcept nogil - -# cuMemcpyWithAttributesAsync (13.2+ — calls through function pointer, safe on older bindings) -# attr is void* here to avoid referencing CUmemcpyAttributes (absent from -# cuda-bindings built against CUDA < 12.8). The C++ side casts it. -cdef cydriver.CUresult memcpy_with_attributes_async( - cydriver.CUdeviceptr dst, cydriver.CUdeviceptr src, size_t size, - void* attr, cydriver.CUstream hStream) nogil -cdef bint has_memcpy_with_attributes_async() noexcept nogil diff --git a/cuda_core/cuda/core/_rt.pyx b/cuda_core/cuda/core/_rt.pyx index f92b051c0d7..f2e2ad1cb3c 100644 --- a/cuda_core/cuda/core/_rt.pyx +++ b/cuda_core/cuda/core/_rt.pyx @@ -271,23 +271,6 @@ cdef extern from "_cpp/rt/rt.hpp" namespace "cuda_core::rt": FileDescriptorHandle create_fd_handle_ref "cuda_core::rt::create_fd_handle_ref" ( int fd) except+ nogil - # SM resource split (13.1+ wrapper — avoids direct cydriver cimport) - # groupParams is void* to avoid referencing CU_DEV_SM_RESOURCE_GROUP_PARAMS - # (which doesn't exist in cuda-bindings 13.0 .pxd). The C++ side casts it. - cydriver.CUresult sm_resource_split "cuda_core::rt::sm_resource_split" ( - cydriver.CUdevResource* result, unsigned int nbGroups, - const cydriver.CUdevResource* input, cydriver.CUdevResource* remainder, - unsigned int flags, void* groupParams) nogil - bint has_sm_resource_split "cuda_core::rt::has_sm_resource_split" () noexcept nogil - - # cuMemcpyWithAttributesAsync (13.2+ wrapper — avoids direct cydriver cimport) - # attr is void* to avoid referencing CUmemcpyAttributes (absent from - # cuda-bindings built against CUDA < 12.8). The C++ side casts it. - cydriver.CUresult memcpy_with_attributes_async "cuda_core::rt::memcpy_with_attributes_async" ( - cydriver.CUdeviceptr dst, cydriver.CUdeviceptr src, size_t size, - void* attr, cydriver.CUstream hStream) nogil - bint has_memcpy_with_attributes_async "cuda_core::rt::has_memcpy_with_attributes_async" () noexcept nogil - # Array / mipmapped-array / texture / surface handles (PR #467) OpaqueArrayHandle create_array_handle "cuda_core::rt::create_array_handle" ( const ContextHandle& h_context, const cydriver.CUDA_ARRAY3D_DESCRIPTOR& desc) except+ nogil @@ -420,12 +403,6 @@ cdef extern from "_cpp/rt/rt.hpp" namespace "cuda_core::rt": void* p_cuSurfObjectCreate "reinterpret_cast(cuda_core::rt::p_cuSurfObjectCreate)" void* p_cuSurfObjectDestroy "reinterpret_cast(cuda_core::rt::p_cuSurfObjectDestroy)" - # SM resource split (13.1+) - void* p_cuDevSmResourceSplit "reinterpret_cast(cuda_core::rt::p_cuDevSmResourceSplit)" - - # cuMemcpyWithAttributesAsync (13.2+) - void* p_cuMemcpyWithAttributesAsync "reinterpret_cast(cuda_core::rt::p_cuMemcpyWithAttributesAsync)" - # NVRTC void* p_nvrtcDestroyProgram "reinterpret_cast(cuda_core::rt::p_nvrtcDestroyProgram)" @@ -473,8 +450,6 @@ cdef void _init_driver_fn_pointers() noexcept: global p_cuGraphNodeFindInClone, p_cuGraphChildGraphNodeGetGraph global p_cuLinkDestroy global p_cuGraphicsUnmapResources, p_cuGraphicsUnregisterResource - global p_cuDevSmResourceSplit - global p_cuMemcpyWithAttributesAsync global p_cuArray3DCreate, p_cuArrayDestroy global p_cuMipmappedArrayCreate, p_cuMipmappedArrayDestroy, p_cuMipmappedArrayGetLevel global p_cuTexObjectCreate, p_cuTexObjectDestroy @@ -570,11 +545,6 @@ cdef void _init_driver_fn_pointers() noexcept: p_cuSurfObjectCreate = _get_driver_fn("cuSurfObjectCreate") p_cuSurfObjectDestroy = _get_driver_fn("cuSurfObjectDestroy") - # SM resource split (13.1+ — may not exist in older cuda-bindings) - p_cuDevSmResourceSplit = _get_optional_driver_fn("cuDevSmResourceSplit") - - # cuMemcpyWithAttributesAsync (13.2+ — may not exist in older cuda-bindings) - p_cuMemcpyWithAttributesAsync = _get_optional_driver_fn("cuMemcpyWithAttributesAsync") _init_driver_fn_pointers() initialize_deferred_cleanup() diff --git a/cuda_core/tests/test_build_hooks.py b/cuda_core/tests/test_build_hooks.py index 429bf358737..34fc10d534d 100644 --- a/cuda_core/tests/test_build_hooks.py +++ b/cuda_core/tests/test_build_hooks.py @@ -574,3 +574,34 @@ def test_unsupported_major_names_the_supported_ones(self, monkeypatch): build_hooks._determine_cuda_major_version.cache_clear() with pytest.raises(RuntimeError, match="does not support CUDA 11.*12, 13"): build_hooks._get_cuda_bindings_require() + + +class TestDefineMacros: + """The C++ learns the build decision through two macros (see _cpp/rt/versions.hpp).""" + + @pytest.mark.agent_authored(model="claude-fable-5-1") + @pytest.mark.parametrize("major", ["12", "13"]) + def test_major_and_floor_header_version(self, major): + floor = build_hooks._load_bindings_floor().CUDA_BINDINGS_FLOOR[int(major)] + assert build_hooks._build_define_macros(major) == [ + ("CUDA_CORE_BUILD_MAJOR", major), + ("CUDA_CORE_MIN_CUDA_VERSION", str(floor[0] * 1000 + floor[1] * 10)), + ] + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_extensions_receive_the_macros(self, monkeypatch): + captured = {} + + def fake_cythonize(ext_modules, **kwargs): + captured["macros"] = {tuple(ext.define_macros) for ext in ext_modules} + return [] + + monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: "/nonexistent-cuda") + monkeypatch.setattr(build_hooks, "_check_build_configuration", lambda cuda_path, cuda_major: None) + monkeypatch.setattr(build_hooks, "cythonize", fake_cythonize) + monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", "13") + build_hooks._determine_cuda_major_version.cache_clear() + monkeypatch.chdir(Path(__file__).parent.parent) + monkeypatch.setattr(sys, "path", list(sys.path)) + build_hooks._build_cuda_core() + assert captured["macros"] == {tuple(build_hooks._build_define_macros("13"))} diff --git a/cuda_core/tests/test_rt_layout.py b/cuda_core/tests/test_rt_layout.py index 53158370968..d70d24ee374 100644 --- a/cuda_core/tests/test_rt_layout.py +++ b/cuda_core/tests/test_rt_layout.py @@ -90,7 +90,7 @@ def test_umbrellas_are_named_only_by_their_cython_file(): @pytest.mark.agent_authored(model="claude-fable-5-1") def test_consumer_closure_is_types_and_the_python_seam(): closure = {p.name for p in include_closure(RT / "handles.hpp")} - assert closure == {"handles.hpp", "py.hpp", "types.hpp"} + assert closure == {"handles.hpp", "py.hpp", "types.hpp", "versions.hpp"} # Consumers are RTLD_LOCAL extensions that cannot link to _rt: nothing with storage. for name in sorted(closure): text = read(RT / name) @@ -98,6 +98,20 @@ def test_consumer_closure_is_types_and_the_python_seam(): assert not re.search(r"^(static|thread_local)\b", text, re.M), f"{name} defines storage" +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_cuda_version_is_named_only_in_versions_hpp(): + """The C++ may branch on CUDA_CORE_BUILD_MAJOR only. A `#if CUDA_VERSION >= 130x0` + fence compiled a feature out of source builds against an older header while the + run-time checks never noticed (https://github.com/NVIDIA/cuda-python/issues/2783); + versions.hpp checks the header once and is the only file allowed to name it.""" + cpp = CORE / "_cpp" + files = sorted(p for p in cpp.rglob("*") if p.suffix in (".hpp", ".h", ".cpp")) + assert len(files) > 20 + spellers = sorted(p.relative_to(cpp).as_posix() for p in files if re.search(r"\bCUDA_VERSION\b", read(p))) + assert spellers == ["rt/versions.hpp"] + assert "CUDA_CORE_BUILD_MAJOR" in read(RT / "versions.hpp") + + @pytest.mark.agent_authored(model="claude-fable-5-1") def test_pxd_functions_are_not_called_by_name_inside_the_module(): """Cython emits a static prototype for each cdef function the .pxd declares, so From b9ada0d9bed9b22bd09d3a79edb878317b43da86 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Fri, 18 Sep 2026 12:55:50 -0700 Subject: [PATCH 03/18] cuda.core: call the driver through cuda-bindings' resolved pointers (#2783) The C++ under _cpp/rt/ used to call the driver through cuda-bindings' Cython wrappers, extracted from cydriver.__pyx_capi__ at import. A wrapper for a function the driver lacks raises a Python exception the C++ never sees, and a wrapper the installed cuda-bindings lacks made the pointer optional and probed for null at every use. driver_api.hpp now lists every driver function the C++ calls, with the CUDA version cuda-bindings requests it at. The table is filled on first use from cuda.bindings._internal.driver._inspect_function_pointers(), which holds the driver's own entry points, so `import cuda.core` never touches the driver. Calls go through DRIVER_CALL(name, args...), which fills the table if needed and, when a pointer is still null after the fill (a missing feature gate or a failed fill), reports once and returns an error status from a trampoline instead of dereferencing null. The fill rejects a driver older than the CUDA major series. pw_ deleter wrappers use the same path, so the C++ never null-checks a pointer; functions the driver may lack are gated on the driver version in Cython. NVRTC, NVVM and nvJitLink each have their own one-entry table, filled when a program or linker handle is created. Consequences in this commit: the null-check fallbacks to CUDA_ERROR_NOT_SUPPORTED are gone; green-context stream creation is gated in _stream.pyx on driver 12.5; deviceptr_import_ipc resolves the table before taking ipc_import_mutex and keeps raw calls under the lock (marked `// raw:`); _rt.pyx no longer imports cydriver/cynvrtc/cynvvm/ cynvjitlink or reads __pyx_capi__. test_rt_layout.py checks the table against cuda-bindings' loader and forbids raw p_ calls elsewhere. DESIGN.md and AGENTS.md describe the mechanism and settle the GIL contract for entry points. --- cuda_core/AGENTS.md | 18 +- cuda_core/cuda/core/_cpp/rt/DESIGN.md | 104 ++++-- cuda_core/cuda/core/_cpp/rt/context.cpp | 41 +-- cuda_core/cuda/core/_cpp/rt/context_scope.hpp | 2 +- cuda_core/cuda/core/_cpp/rt/driver_api.cpp | 138 +++----- cuda_core/cuda/core/_cpp/rt/driver_api.hpp | 325 ++++++++++++------ cuda_core/cuda/core/_cpp/rt/error.cpp | 8 +- cuda_core/cuda/core/_cpp/rt/event.cpp | 6 +- cuda_core/cuda/core/_cpp/rt/graph.cpp | 39 +-- cuda_core/cuda/core/_cpp/rt/graph_exec.cpp | 24 +- cuda_core/cuda/core/_cpp/rt/internal.hpp | 15 +- cuda_core/cuda/core/_cpp/rt/memory.cpp | 38 +- cuda_core/cuda/core/_cpp/rt/program.cpp | 38 +- cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp | 229 ++++++++++++ cuda_core/cuda/core/_cpp/rt/stream.cpp | 7 +- cuda_core/cuda/core/_cpp/rt/texture.cpp | 16 +- cuda_core/cuda/core/_rt.pyx | 299 ---------------- cuda_core/cuda/core/_stream.pyx | 15 +- cuda_core/tests/test_rt_layout.py | 47 +++ 19 files changed, 724 insertions(+), 685 deletions(-) create mode 100644 cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp diff --git a/cuda_core/AGENTS.md b/cuda_core/AGENTS.md index 7196f95638f..115e6312619 100644 --- a/cuda_core/AGENTS.md +++ b/cuda_core/AGENTS.md @@ -91,10 +91,20 @@ and agents should flag violations. objects that are not meant to be shared (e.g., the thread-local `Device`) do not need such guards (see #2321). Reference-count integrity is guaranteed; cache value-identity/idempotency is not. -- **Entry points assume the GIL is held**: the helpers in `_cpp/rt/` - are called from Cython with the GIL held and do not re-acquire it. Driver and - destructor callbacks run at arbitrary times, so they take the GIL (`with gil`) - and probe for interpreter shutdown before touching Python objects. +- **Entry points work with or without the GIL**: the helpers in `_cpp/rt/` + are called from Cython both inside and outside `with nogil` blocks. They + never require the GIL, release it around driver calls, and never acquire it + while holding a C++ lock; the only paths that acquire it are the reporting + wrappers (`pw_*`, `report_*`) and the one-time driver function-table fill + (`ensure_fn_table()`, see `_cpp/rt/DESIGN.md`). Driver and destructor + callbacks run at arbitrary times, so they take the GIL (`with gil`) and probe + for interpreter shutdown before touching Python objects. +- **Driver calls go through the table**: C++ calls the driver with + `DRIVER_CALL(name, args...)`, whose pointers come from cuda-bindings' + resolved table, never from the Cython wrappers. A function the installed + driver may lack is gated in Cython on `cy_driver_version()` at the version + cuda-bindings requests it at (the number in `driver_api.hpp`); the C++ + never checks a pointer for null. - **Lock ordering -- release the GIL before entering the driver**: any CUDA work reachable from a host callback or a retained object's `__del__` must release the GIL before calling the driver, to avoid GIL/driver-lock deadlocks (see the diff --git a/cuda_core/cuda/core/_cpp/rt/DESIGN.md b/cuda_core/cuda/core/_cpp/rt/DESIGN.md index 6c1194f9cd8..ed0073a7392 100644 --- a/cuda_core/cuda/core/_cpp/rt/DESIGN.md +++ b/cuda_core/cuda/core/_cpp/rt/DESIGN.md @@ -165,45 +165,67 @@ functions, Cython generates calls through `_rt.so` at runtime. This ensures all static and thread-local state lives in a single shared library, avoiding the duplicate state problem. -## CUDA driver function pointers via cuda-bindings' `__pyx_capi__` +## CUDA driver function pointers from cuda-bindings -**Problem**: cuda.core cannot directly call CUDA driver functions because: +**Problem**: cuda.core cannot link against `libcuda.so` at build time, and it +must not load the driver itself: cuda-bindings owns driver loading and symbol +resolution (with `cuGetProcAddress`, which also selects the ABI variant and the +per-thread-default-stream variant). Until #2783, the C++ called cuda-bindings' +*Cython wrappers*, extracted from `cydriver.__pyx_capi__`. When the driver +lacked a function, a wrapper raised a Python exception that C++ never saw and +returned a sentinel `CUresult`; the exception surfaced later as `SystemError`. +And a wrapper could be absent when the installed cuda-bindings was older than +the build, so some pointers were optional and probed for null. -1. We don't want to link against `libcuda.so` at build time. -2. The driver symbols must be resolved dynamically through cuda-bindings. - -**Solution**: The C++ code declares extern function pointer variables: +**Solution**: `driver_api.hpp` lists every driver function the C++ calls, with +the CUDA version cuda-bindings requests it at: ```cpp -// driver_api.hpp -extern decltype(&cuStreamCreateWithPriority) p_cuStreamCreateWithPriority; -extern decltype(&cuMemPoolCreate) p_cuMemPoolCreate; -// ... etc -``` - -At module import time, `_rt.pyx` populates these pointers by -extracting them from `cuda.bindings.cydriver.__pyx_capi__`: - -```cython -import cuda.bindings.cydriver as cydriver - -cdef void* _get_driver_fn(str name): - capsule = cydriver.__pyx_capi__[name] - return PyCapsule_GetPointer(capsule, PyCapsule_GetName(capsule)) - -p_cuStreamCreateWithPriority = _get_driver_fn("cuStreamCreateWithPriority") +#define CUDA_CORE_DRIVER_FUNCTIONS(X) \ + X(cuStreamCreateWithPriority, 5050) \ + X(cuGreenCtxStreamCreate, 12050) \ + ... ``` -The `__pyx_capi__` dictionary contains PyCapsules that Cython automatically -generates for each `cdef` function declared in a `.pxd` file. Each capsule's -name is the function's C signature; we query it with `PyCapsule_GetName()` -rather than hardcoding signatures. - -This approach: -- Avoids linking against `libcuda.so` at build time -- Works on CPU-only machines (capsule extraction succeeds; actual driver calls - will return errors like `CUDA_ERROR_NO_DEVICE`) -- Requires no custom capsule infrastructure—uses Cython's built-in mechanism +The list declares the `p_cuXxx` pointers and builds a table of +`{key, name, slot, introduced}` entries. The key is `"__"` plus the symbol +`cuda.h` maps the name to (`cuStreamDestroy` -> `"__cuStreamDestroy_v2"`), +which is how cuda-bindings names the slot in +`cuda.bindings._internal.driver._inspect_function_pointers()`. Because the key +follows the header's macros, the build requires the header's major.minor to +equal cuda-bindings' (see "Build-time version guards"). + +The table is filled lazily by `ensure_fn_table()` (`py_driver_fns.cpp`), the +first time a `DRIVER_CALL(name, args...)` finds its pointer null, so +`import cuda.core` never touches the driver. The fill acquires the GIL, calls +`_inspect_function_pointers()`, and copies every entry's address into its +`p_` pointer under a mutex that is never held across a Python call. It then +checks that every function introduced at or before the CUDA major series' +first release is present; a null one means the driver is older than the series +and the fill fails with that message. + +After the fill a pointer is either the driver's entry point or null because +the installed driver does not provide a newer function. Those functions are +gated on the driver version in Cython (`cy_driver_version()`), never by a null +check in C++. A `DRIVER_CALL` that still finds null after the fill is a gate +bug or a failed fill: it reports through `report_message()` and returns +`CUDA_ERROR_NOT_INITIALIZED` from a trampoline of the right signature, so it +never dereferences null and never throws, which makes it safe in `noexcept` +deleters. The `pw_` wrappers go through the same path. Owning handle +constructors whose deleter calls the driver but which do not call it +themselves (`create_graph_handle`, ...) call `ensure_fn_table()` so the fill +never happens in a deleter. NVRTC, NVVM and nvJitLink have one table each, +filled by `create_*_handle` after the library has loaded. + +Two rules follow. A `DRIVER_CALL` must not be made while a C++ lock is held, +because the fill acquires the GIL; resolve the table before the lock +(`ensure_fn_table(FnTable::driver)`) and use the raw pointer inside it, marked +`// raw:` (see `deviceptr_import_ipc`). And a driver function the Cython layer +gates must be gated at the version cuda-bindings requests it at (the number in +the table), not the version the driver first shipped it. + +`tests/test_rt_layout.py` checks the table against cuda-bindings' loader and +that no raw `p_` call exists outside the machinery and the marked lines. ## Build-time version guards @@ -252,8 +274,13 @@ Handle destructors may run from any thread. The implementation includes RAII gua - Handle Python finalization gracefully (avoid GIL operations during shutdown) - Ensure Python object manipulation happens with GIL held -The handle API functions are safe to call with or without the GIL held. They -will release the GIL (if necessary) before calling CUDA driver API functions. +The handle API functions may be called with or without the GIL held (Cython +calls most of them from `with nogil` blocks and some with the GIL held). They +never require it and never take a C++ lock while acquiring it. They release the +GIL (if necessary) before calling CUDA driver API functions. The only places +that acquire the GIL are the reporting paths (`pw_*`, `report_*`) and the +one-time function-table fill (`ensure_fn_table()`), neither of which may run +while a C++ lock is held. **The GIL is the outermost lock.** Code that holds a C++ lock (a registry's mutex, `ipc_import_mutex`, any `std::mutex`) must not acquire or reacquire the @@ -387,9 +414,10 @@ exactly one of these; none is ever dropped. ### `p_` versus `pw_` -A `p_` function pointer calls the driver and nothing else. Its `pw_` twin calls -the driver and, if the call fails, acquires the GIL and runs Python: the warning -filters, `showwarning`, or `sys.unraisablehook`. Any of those can be user code, +A `DRIVER_CALL` (a `p_` function pointer) calls the driver and nothing else, +once the table is filled. Its `pw_` twin calls the driver and, if the call +fails, acquires the GIL and runs Python: the warning filters, `showwarning`, or +`sys.unraisablehook`. Any of those can be user code, and user code can call back into cuda.core. This is the one place where the handle layer runs code it does not control, and it is the entry point through which a thread holding a C++ lock can deadlock (see "GIL Management"). diff --git a/cuda_core/cuda/core/_cpp/rt/context.cpp b/cuda_core/cuda/core/_cpp/rt/context.cpp index 89478484738..b0b6d162ab2 100644 --- a/cuda_core/cuda/core/_cpp/rt/context.cpp +++ b/cuda_core/cuda/core/_cpp/rt/context.cpp @@ -43,11 +43,11 @@ CUresult enter_context(const ContextHandle& h_context, CUcontext* previous, int* } GILReleaseGuard gil; - CUresult status = p_cuCtxGetCurrent(previous); + CUresult status = DRIVER_CALL(cuCtxGetCurrent, previous); if (status != CUDA_SUCCESS || *previous == target) { return status; } - status = p_cuCtxSetCurrent(target); + status = DRIVER_CALL(cuCtxSetCurrent, target); *changed = status == CUDA_SUCCESS; return status; } @@ -62,7 +62,7 @@ CUresult restore_context(CUcontext previous) noexcept { return fault; } GILReleaseGuard gil; - return p_cuCtxSetCurrent(previous); + return DRIVER_CALL(cuCtxSetCurrent, previous); } // Restore the previous context and preserve an earlier operation error. The // operation error, if any, is returned; otherwise the restoration status is. @@ -82,7 +82,7 @@ CUresult exit_context(CUcontext previous, int changed, CUresult operation_status CUresult context_synchronize(const ContextHandle& h_context) noexcept { GILReleaseGuard gil; return invoke_in_context(h_context, []() noexcept { - return p_cuCtxSynchronize(); + return DRIVER_FN(cuCtxSynchronize)(); }); } @@ -92,14 +92,14 @@ CUresult context_get_stream_priority_range(const ContextHandle& h_context, int* greatest_priority) noexcept { GILReleaseGuard gil; return invoke_in_context(h_context, [&]() noexcept { - return p_cuCtxGetStreamPriorityRange(least_priority, greatest_priority); + return DRIVER_CALL(cuCtxGetStreamPriorityRange, least_priority, greatest_priority); }); } // Query the device of the provided context. CUresult context_get_device(const ContextHandle& h_context, CUdevice* device) noexcept { return invoke_in_context(h_context, [&]() noexcept { - return p_cuCtxGetDevice(device); + return DRIVER_CALL(cuCtxGetDevice, device); }); } @@ -157,13 +157,8 @@ ContextHandle create_context_handle_from_green_ctx(const GreenCtxHandle& h_green if (!h_green_ctx) { return {}; } - if (!p_cuCtxFromGreenCtx) { - err = CUDA_ERROR_NOT_SUPPORTED; - return {}; - } - CUcontext ctx = nullptr; - if (CUDA_SUCCESS != (err = p_cuCtxFromGreenCtx(&ctx, as_cu(h_green_ctx)))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuCtxFromGreenCtx, &ctx, as_cu(h_green_ctx)))) { return {}; } @@ -180,18 +175,13 @@ GreenCtxHandle get_context_green_ctx(const ContextHandle& h) noexcept { GreenCtxHandle create_green_ctx_handle(CUdevResource* resources, unsigned int nbResources, CUdevice dev, unsigned int flags) { GILReleaseGuard gil; - if (!p_cuDevResourceGenerateDesc || !p_cuGreenCtxCreate || !p_cuGreenCtxDestroy) { - err = CUDA_ERROR_NOT_SUPPORTED; - return {}; - } - CUdevResourceDesc desc = nullptr; - if (CUDA_SUCCESS != (err = p_cuDevResourceGenerateDesc(&desc, resources, nbResources))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuDevResourceGenerateDesc, &desc, resources, nbResources))) { return {}; } CUgreenCtx green_ctx = nullptr; - if (CUDA_SUCCESS != (err = p_cuGreenCtxCreate(&green_ctx, desc, dev, flags))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuGreenCtxCreate, &green_ctx, desc, dev, flags))) { return {}; } @@ -228,7 +218,7 @@ ContextHandle get_primary_context(int device_id) { // Cache miss - acquire primary context from driver GILReleaseGuard gil; CUcontext ctx; - if (CUDA_SUCCESS != (err = p_cuDevicePrimaryCtxRetain(&ctx, device_id))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuDevicePrimaryCtxRetain, &ctx, device_id))) { return {}; } @@ -236,13 +226,12 @@ ContextHandle get_primary_context(int device_id) { new ContextBox{ctx, {}}, [device_id](const ContextBox* b) { context_registry.unregister_handle(b->resource); - // The driver function pointer targets a Cython __pyx_capi__ - // wrapper, which touches the Python runtime even though the - // underlying CUDA call does not. During interpreter shutdown, - // leave primary-context cleanup to process teardown. + // During interpreter shutdown, leave primary-context cleanup to + // process teardown (an unavailable table entry would need Python + // to report itself). if (Py_IsInitialized() && !py_is_finalizing()) { GILReleaseGuard gil; - p_cuDevicePrimaryCtxRelease(device_id); + DRIVER_CALL(cuDevicePrimaryCtxRelease, device_id); } delete b; } @@ -261,7 +250,7 @@ ContextHandle get_primary_context(int device_id) { ContextHandle get_current_context() { GILReleaseGuard gil; CUcontext ctx = nullptr; - if (CUDA_SUCCESS != (err = p_cuCtxGetCurrent(&ctx))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuCtxGetCurrent, &ctx))) { return {}; } if (!ctx) { diff --git a/cuda_core/cuda/core/_cpp/rt/context_scope.hpp b/cuda_core/cuda/core/_cpp/rt/context_scope.hpp index d1965d55a52..040a8eb9fa8 100644 --- a/cuda_core/cuda/core/_cpp/rt/context_scope.hpp +++ b/cuda_core/cuda/core/_cpp/rt/context_scope.hpp @@ -65,7 +65,7 @@ CUresult invoke_in_context_or_undo(const ContextHandle& h_context, Fn&& operatio bool undo_ok = true; if (undo_requires_target_context) { CUcontext current = nullptr; - undo_ok = p_cuCtxGetCurrent(¤t) == CUDA_SUCCESS + undo_ok = DRIVER_CALL(cuCtxGetCurrent, ¤t) == CUDA_SUCCESS && current == as_cu(h_context); } if (undo_ok) { diff --git a/cuda_core/cuda/core/_cpp/rt/driver_api.cpp b/cuda_core/cuda/core/_cpp/rt/driver_api.cpp index 34b080c6658..83b6b92530e 100644 --- a/cuda_core/cuda/core/_cpp/rt/driver_api.cpp +++ b/cuda_core/cuda/core/_cpp/rt/driver_api.cpp @@ -8,99 +8,57 @@ namespace cuda_core::rt { -// ============================================================================ -// CUDA driver function pointers -// -// These are populated by _rt.pyx at module import time using -// function pointers extracted from cuda.bindings.cydriver.__pyx_capi__. -// ============================================================================ - -decltype(&cuGetErrorName) p_cuGetErrorName = nullptr; -decltype(&cuGetErrorString) p_cuGetErrorString = nullptr; - -decltype(&cuDevicePrimaryCtxRetain) p_cuDevicePrimaryCtxRetain = nullptr; -decltype(&cuDevicePrimaryCtxRelease) p_cuDevicePrimaryCtxRelease = nullptr; -decltype(&cuCtxGetCurrent) p_cuCtxGetCurrent = nullptr; -decltype(&cuCtxSetCurrent) p_cuCtxSetCurrent = nullptr; -decltype(&cuCtxSynchronize) p_cuCtxSynchronize = nullptr; -decltype(&cuCtxGetStreamPriorityRange) p_cuCtxGetStreamPriorityRange = nullptr; -decltype(&cuCtxGetDevice) p_cuCtxGetDevice = nullptr; -decltype(&cuGraphNodeSetParams) p_cuGraphNodeSetParams = nullptr; -decltype(&cuGreenCtxCreate) p_cuGreenCtxCreate = nullptr; -decltype(&cuGreenCtxDestroy) p_cuGreenCtxDestroy = nullptr; -decltype(&cuCtxFromGreenCtx) p_cuCtxFromGreenCtx = nullptr; -decltype(&cuDevResourceGenerateDesc) p_cuDevResourceGenerateDesc = nullptr; - -decltype(&cuGreenCtxStreamCreate) p_cuGreenCtxStreamCreate = nullptr; - -decltype(&cuStreamCreateWithPriority) p_cuStreamCreateWithPriority = nullptr; -decltype(&cuStreamDestroy) p_cuStreamDestroy = nullptr; -decltype(&cuStreamGetCtx) p_cuStreamGetCtx = nullptr; - -decltype(&cuEventCreate) p_cuEventCreate = nullptr; -decltype(&cuEventDestroy) p_cuEventDestroy = nullptr; -decltype(&cuIpcOpenEventHandle) p_cuIpcOpenEventHandle = nullptr; - -decltype(&cuDeviceGetCount) p_cuDeviceGetCount = nullptr; - -decltype(&cuMemPoolSetAccess) p_cuMemPoolSetAccess = nullptr; -decltype(&cuMemPoolDestroy) p_cuMemPoolDestroy = nullptr; -decltype(&cuMemPoolCreate) p_cuMemPoolCreate = nullptr; -decltype(&cuDeviceGetMemPool) p_cuDeviceGetMemPool = nullptr; -decltype(&cuMemPoolImportFromShareableHandle) p_cuMemPoolImportFromShareableHandle = nullptr; - -decltype(&cuMemAllocFromPoolAsync) p_cuMemAllocFromPoolAsync = nullptr; -decltype(&cuMemAllocAsync) p_cuMemAllocAsync = nullptr; -decltype(&cuMemAlloc) p_cuMemAlloc = nullptr; -decltype(&cuMemAllocHost) p_cuMemAllocHost = nullptr; - -decltype(&cuMemFreeAsync) p_cuMemFreeAsync = nullptr; -decltype(&cuMemFree) p_cuMemFree = nullptr; -decltype(&cuMemFreeHost) p_cuMemFreeHost = nullptr; +// The pointers. Null until ensure_fn_table() fills the table; see driver_api.hpp. +#define CUDA_CORE_DEFINE_DRIVER_FN(name, introduced) decltype(&name) p_##name = nullptr; +CUDA_CORE_DRIVER_FUNCTIONS(CUDA_CORE_DEFINE_DRIVER_FN) +#undef CUDA_CORE_DEFINE_DRIVER_FN -decltype(&cuMemPoolImportPointer) p_cuMemPoolImportPointer = nullptr; - -decltype(&cuLibraryLoadFromFile) p_cuLibraryLoadFromFile = nullptr; -decltype(&cuLibraryLoadData) p_cuLibraryLoadData = nullptr; -decltype(&cuLibraryUnload) p_cuLibraryUnload = nullptr; -decltype(&cuLibraryGetKernel) p_cuLibraryGetKernel = nullptr; - -// Graph -decltype(&cuGraphDestroy) p_cuGraphDestroy = nullptr; -decltype(&cuGraphInstantiateWithParams) p_cuGraphInstantiateWithParams = nullptr; -decltype(&cuGraphExecUpdate) p_cuGraphExecUpdate = nullptr; -decltype(&cuGraphExecDestroy) p_cuGraphExecDestroy = nullptr; -decltype(&cuUserObjectCreate) p_cuUserObjectCreate = nullptr; -decltype(&cuUserObjectRelease) p_cuUserObjectRelease = nullptr; -decltype(&cuGraphRetainUserObject) p_cuGraphRetainUserObject = nullptr; -decltype(&cuGraphReleaseUserObject) p_cuGraphReleaseUserObject = nullptr; -decltype(&cuGraphNodeFindInClone) p_cuGraphNodeFindInClone = nullptr; -decltype(&cuGraphChildGraphNodeGetGraph) p_cuGraphChildGraphNodeGetGraph = nullptr; - -// Linker -decltype(&cuLinkDestroy) p_cuLinkDestroy = nullptr; - -// GL interop pointers -decltype(&cuGraphicsUnmapResources) p_cuGraphicsUnmapResources = nullptr; -decltype(&cuGraphicsUnregisterResource) p_cuGraphicsUnregisterResource = nullptr; - -decltype(&cuArray3DCreate) p_cuArray3DCreate = nullptr; -decltype(&cuArrayDestroy) p_cuArrayDestroy = nullptr; -decltype(&cuMipmappedArrayCreate) p_cuMipmappedArrayCreate = nullptr; -decltype(&cuMipmappedArrayDestroy) p_cuMipmappedArrayDestroy = nullptr; -decltype(&cuMipmappedArrayGetLevel) p_cuMipmappedArrayGetLevel = nullptr; -decltype(&cuTexObjectCreate) p_cuTexObjectCreate = nullptr; -decltype(&cuTexObjectDestroy) p_cuTexObjectDestroy = nullptr; -decltype(&cuSurfObjectCreate) p_cuSurfObjectCreate = nullptr; -decltype(&cuSurfObjectDestroy) p_cuSurfObjectDestroy = nullptr; - -// NVRTC function pointers decltype(&nvrtcDestroyProgram) p_nvrtcDestroyProgram = nullptr; - -// NVVM function pointers (may be null if NVVM is not available) NvvmDestroyProgramFn p_nvvmDestroyProgram = nullptr; - -// nvJitLink function pointers (may be null if nvJitLink is not available) NvJitLinkDestroyFn p_nvJitLinkDestroy = nullptr; +namespace { + +#define CUDA_CORE_STR(x) #x +#define CUDA_CORE_XSTR(x) CUDA_CORE_STR(x) + +// "__" + the symbol cuda.h maps the public name to (macro-expanded), which is +// how cuda-bindings keys its table; #name is the public name, unexpanded. +#define CUDA_CORE_DRIVER_FN_ENTRY(name, introduced) \ + {"__" CUDA_CORE_XSTR(name), #name, reinterpret_cast(&p_##name), introduced}, + +const FnEntry driver_entries[] = {CUDA_CORE_DRIVER_FUNCTIONS(CUDA_CORE_DRIVER_FN_ENTRY)}; +#undef CUDA_CORE_DRIVER_FN_ENTRY + +const FnEntry nvrtc_entries[] = { + {"__nvrtcDestroyProgram", "nvrtcDestroyProgram", reinterpret_cast(&p_nvrtcDestroyProgram), 0}, +}; +const FnEntry nvvm_entries[] = { + {"__nvvmDestroyProgram", "nvvmDestroyProgram", reinterpret_cast(&p_nvvmDestroyProgram), 0}, +}; +const FnEntry nvjitlink_entries[] = { + {"__nvJitLinkDestroy", "nvJitLinkDestroy", reinterpret_cast(&p_nvJitLinkDestroy), 0}, +}; + +} // namespace + +const FnEntry* fn_table_entries(FnTable table, std::size_t* count) noexcept { + switch (table) { + case FnTable::driver: + *count = sizeof(driver_entries) / sizeof(driver_entries[0]); + return driver_entries; + case FnTable::nvrtc: + *count = sizeof(nvrtc_entries) / sizeof(nvrtc_entries[0]); + return nvrtc_entries; + case FnTable::nvvm: + *count = sizeof(nvvm_entries) / sizeof(nvvm_entries[0]); + return nvvm_entries; + case FnTable::nvjitlink: + *count = sizeof(nvjitlink_entries) / sizeof(nvjitlink_entries[0]); + return nvjitlink_entries; + } + *count = 0; + return nullptr; +} + } // namespace cuda_core::rt diff --git a/cuda_core/cuda/core/_cpp/rt/driver_api.hpp b/cuda_core/cuda/core/_cpp/rt/driver_api.hpp index 475b1b15579..b2955816647 100644 --- a/cuda_core/cuda/core/_cpp/rt/driver_api.hpp +++ b/cuda_core/cuda/core/_cpp/rt/driver_api.hpp @@ -12,126 +12,229 @@ namespace cuda_core::rt { // ============================================================================ -// CUDA driver function pointers +// Driver and compiler-library function pointers // -// These are populated by _rt.pyx at module import time using -// function pointers extracted from cuda.bindings.cydriver.__pyx_capi__. -// ============================================================================ - -extern decltype(&cuGetErrorName) p_cuGetErrorName; -extern decltype(&cuGetErrorString) p_cuGetErrorString; - -extern decltype(&cuDevicePrimaryCtxRetain) p_cuDevicePrimaryCtxRetain; -extern decltype(&cuDevicePrimaryCtxRelease) p_cuDevicePrimaryCtxRelease; -extern decltype(&cuCtxGetCurrent) p_cuCtxGetCurrent; -extern decltype(&cuCtxSetCurrent) p_cuCtxSetCurrent; -extern decltype(&cuCtxSynchronize) p_cuCtxSynchronize; -extern decltype(&cuCtxGetStreamPriorityRange) p_cuCtxGetStreamPriorityRange; -extern decltype(&cuCtxGetDevice) p_cuCtxGetDevice; -extern decltype(&cuGraphNodeSetParams) p_cuGraphNodeSetParams; -extern decltype(&cuGreenCtxCreate) p_cuGreenCtxCreate; -extern decltype(&cuGreenCtxDestroy) p_cuGreenCtxDestroy; -extern decltype(&cuCtxFromGreenCtx) p_cuCtxFromGreenCtx; -extern decltype(&cuDevResourceGenerateDesc) p_cuDevResourceGenerateDesc; - -extern decltype(&cuGreenCtxStreamCreate) p_cuGreenCtxStreamCreate; - -extern decltype(&cuStreamCreateWithPriority) p_cuStreamCreateWithPriority; -extern decltype(&cuStreamDestroy) p_cuStreamDestroy; -extern decltype(&cuStreamGetCtx) p_cuStreamGetCtx; - -extern decltype(&cuEventCreate) p_cuEventCreate; -extern decltype(&cuEventDestroy) p_cuEventDestroy; -extern decltype(&cuIpcOpenEventHandle) p_cuIpcOpenEventHandle; - -extern decltype(&cuDeviceGetCount) p_cuDeviceGetCount; - -extern decltype(&cuMemPoolSetAccess) p_cuMemPoolSetAccess; -extern decltype(&cuMemPoolDestroy) p_cuMemPoolDestroy; -extern decltype(&cuMemPoolCreate) p_cuMemPoolCreate; -extern decltype(&cuDeviceGetMemPool) p_cuDeviceGetMemPool; -extern decltype(&cuMemPoolImportFromShareableHandle) p_cuMemPoolImportFromShareableHandle; - -extern decltype(&cuMemAllocFromPoolAsync) p_cuMemAllocFromPoolAsync; -extern decltype(&cuMemAllocAsync) p_cuMemAllocAsync; -extern decltype(&cuMemAlloc) p_cuMemAlloc; -extern decltype(&cuMemAllocHost) p_cuMemAllocHost; - -extern decltype(&cuMemFreeAsync) p_cuMemFreeAsync; -extern decltype(&cuMemFree) p_cuMemFree; -extern decltype(&cuMemFreeHost) p_cuMemFreeHost; - -extern decltype(&cuMemPoolImportPointer) p_cuMemPoolImportPointer; - -// Library -extern decltype(&cuLibraryLoadFromFile) p_cuLibraryLoadFromFile; -extern decltype(&cuLibraryLoadData) p_cuLibraryLoadData; -extern decltype(&cuLibraryUnload) p_cuLibraryUnload; -extern decltype(&cuLibraryGetKernel) p_cuLibraryGetKernel; - -// Graph -extern decltype(&cuGraphDestroy) p_cuGraphDestroy; -extern decltype(&cuGraphInstantiateWithParams) p_cuGraphInstantiateWithParams; -extern decltype(&cuGraphExecUpdate) p_cuGraphExecUpdate; -extern decltype(&cuGraphExecDestroy) p_cuGraphExecDestroy; -extern decltype(&cuUserObjectCreate) p_cuUserObjectCreate; -extern decltype(&cuUserObjectRelease) p_cuUserObjectRelease; -extern decltype(&cuGraphRetainUserObject) p_cuGraphRetainUserObject; -extern decltype(&cuGraphReleaseUserObject) p_cuGraphReleaseUserObject; -extern decltype(&cuGraphNodeFindInClone) p_cuGraphNodeFindInClone; -extern decltype(&cuGraphChildGraphNodeGetGraph) p_cuGraphChildGraphNodeGetGraph; - -// Linker -extern decltype(&cuLinkDestroy) p_cuLinkDestroy; - -// Graphics interop -extern decltype(&cuGraphicsUnmapResources) p_cuGraphicsUnmapResources; -extern decltype(&cuGraphicsUnregisterResource) p_cuGraphicsUnregisterResource; - -// Texture / surface / array (PR #467) -extern decltype(&cuArray3DCreate) p_cuArray3DCreate; -extern decltype(&cuArrayDestroy) p_cuArrayDestroy; -extern decltype(&cuMipmappedArrayCreate) p_cuMipmappedArrayCreate; -extern decltype(&cuMipmappedArrayDestroy) p_cuMipmappedArrayDestroy; -extern decltype(&cuMipmappedArrayGetLevel) p_cuMipmappedArrayGetLevel; -extern decltype(&cuTexObjectCreate) p_cuTexObjectCreate; -extern decltype(&cuTexObjectDestroy) p_cuTexObjectDestroy; -extern decltype(&cuSurfObjectCreate) p_cuSurfObjectCreate; -extern decltype(&cuSurfObjectDestroy) p_cuSurfObjectDestroy; - -// ============================================================================ -// NVRTC function pointers +// The C++ under _cpp/rt/ calls the CUDA driver through the p_cuXxx pointers +// below. They hold the driver's own entry points, taken from the table that +// cuda-bindings builds when it loads the driver (cuGetProcAddress for each +// symbol, choosing the ABI variant and the per-thread-default-stream variant) +// and exposes as cuda.bindings._internal.driver._inspect_function_pointers(). +// cuda.core never loads the driver or resolves a symbol itself, and it never +// calls cuda-bindings' Cython wrappers from C++: those raise a Python +// exception when the driver lacks a function, which C++ cannot see +// (https://github.com/NVIDIA/cuda-python/issues/2783). // -// These are populated by _rt.pyx at module import time using -// function pointers extracted from cuda.bindings.cynvrtc.__pyx_capi__. -// ============================================================================ - -extern decltype(&nvrtcDestroyProgram) p_nvrtcDestroyProgram; - -// ============================================================================ -// NVVM function pointers +// The table is filled lazily, the first time a DRIVER_CALL finds its pointer +// null, so that `import cuda.core` never touches the driver. The fill acquires +// the GIL and runs Python (see py_driver_fns.cpp), so a DRIVER_CALL must not +// be made while a C++ lock is held; call ensure_fn_table() before taking the +// lock and use the raw pointer inside it (see deviceptr_import_ipc). // -// These are populated by _rt.pyx at module import time using -// function pointers extracted from cuda.bindings.cynvvm.__pyx_capi__. -// Note: May be null if NVVM is not available at runtime. +// After the fill a pointer is either the driver's entry point or null because +// the installed driver does not provide that function. Functions introduced +// at or before the first release of the CUDA major series being built are +// present in every driver cuda.core supports; the fill checks them and +// rejects an older driver. A newer function can legitimately be null, and the +// Cython layer gates its use on the driver version (cy_driver_version()), so +// the C++ never checks a pointer for null before a call. A DRIVER_CALL that +// still finds null after the fill is therefore a gate bug (or a failed fill); +// it reports an internal error and returns CUDA_ERROR_NOT_INITIALIZED from a +// trampoline of the right signature instead of dereferencing null. It never +// throws, so it is safe in noexcept deleters and cleanup paths. // ============================================================================ -// Function pointer type for nvvmDestroyProgram (avoids nvvm.h dependency) -// Signature: nvvmResult nvvmDestroyProgram(nvvmProgram *prog) +// Each X(name, introduced) names a driver function cuda.core calls and the CUDA +// version cuda-bindings requests it at (cuGetProcAddress's cudaVersion; the +// ABI's introduction). tests/test_rt_layout.py checks the list against +// cuda-bindings' loader. `name` is the public name; cuda.h may map it to a +// versioned symbol (cuStreamDestroy -> cuStreamDestroy_v2), and the table key +// follows that mapping, so the header cuda.core compiles against must be the one +// cuda-bindings was generated from (enforced by build_hooks.py). +#define CUDA_CORE_DRIVER_FUNCTIONS(X) \ + /* Error formatting */ \ + X(cuGetErrorName, 6000) \ + X(cuGetErrorString, 6000) \ + /* Context */ \ + X(cuDevicePrimaryCtxRetain, 7000) \ + X(cuDevicePrimaryCtxRelease, 11000) \ + X(cuCtxGetCurrent, 4000) \ + X(cuCtxSetCurrent, 4000) \ + X(cuCtxSynchronize, 2000) \ + X(cuCtxGetStreamPriorityRange, 5050) \ + X(cuCtxGetDevice, 2000) \ + X(cuGraphNodeSetParams, 12020) \ + X(cuGreenCtxCreate, 12040) \ + X(cuGreenCtxDestroy, 12040) \ + X(cuCtxFromGreenCtx, 12040) \ + X(cuDevResourceGenerateDesc, 12040) \ + X(cuGreenCtxStreamCreate, 12050) \ + /* Stream */ \ + X(cuStreamCreateWithPriority, 5050) \ + X(cuStreamDestroy, 4000) \ + X(cuStreamGetCtx, 9020) \ + /* Event */ \ + X(cuEventCreate, 2000) \ + X(cuEventDestroy, 4000) \ + X(cuIpcOpenEventHandle, 4010) \ + /* Device */ \ + X(cuDeviceGetCount, 2000) \ + /* Memory pool */ \ + X(cuMemPoolSetAccess, 11020) \ + X(cuMemPoolDestroy, 11020) \ + X(cuMemPoolCreate, 11020) \ + X(cuDeviceGetMemPool, 11020) \ + X(cuMemPoolImportFromShareableHandle, 11020) \ + /* Memory allocation */ \ + X(cuMemAllocFromPoolAsync, 11020) \ + X(cuMemAllocAsync, 11020) \ + X(cuMemAlloc, 3020) \ + X(cuMemAllocHost, 3020) \ + X(cuMemFreeAsync, 11020) \ + X(cuMemFree, 3020) \ + X(cuMemFreeHost, 2000) \ + /* IPC */ \ + X(cuMemPoolImportPointer, 11020) \ + /* Library */ \ + X(cuLibraryLoadFromFile, 12000) \ + X(cuLibraryLoadData, 12000) \ + X(cuLibraryUnload, 12000) \ + X(cuLibraryGetKernel, 12000) \ + /* Graph */ \ + X(cuGraphDestroy, 10000) \ + X(cuGraphInstantiateWithParams, 12000) \ + X(cuGraphExecUpdate, 12000) \ + X(cuGraphExecDestroy, 10000) \ + X(cuUserObjectCreate, 11030) \ + X(cuUserObjectRelease, 11030) \ + X(cuGraphRetainUserObject, 11030) \ + X(cuGraphReleaseUserObject, 11030) \ + X(cuGraphNodeFindInClone, 10000) \ + X(cuGraphChildGraphNodeGetGraph, 10000) \ + /* Linker */ \ + X(cuLinkDestroy, 5050) \ + /* Graphics interop */ \ + X(cuGraphicsUnmapResources, 7000) \ + X(cuGraphicsUnregisterResource, 3000) \ + /* Texture / surface / array (PR #467) */ \ + X(cuArray3DCreate, 3020) \ + X(cuArrayDestroy, 2000) \ + X(cuMipmappedArrayCreate, 5000) \ + X(cuMipmappedArrayDestroy, 5000) \ + X(cuMipmappedArrayGetLevel, 5000) \ + X(cuTexObjectCreate, 5000) \ + X(cuTexObjectDestroy, 5000) \ + X(cuSurfObjectCreate, 5000) \ + X(cuSurfObjectDestroy, 5000) + +#define CUDA_CORE_DECLARE_DRIVER_FN(name, introduced) extern decltype(&name) p_##name; +CUDA_CORE_DRIVER_FUNCTIONS(CUDA_CORE_DECLARE_DRIVER_FN) +#undef CUDA_CORE_DECLARE_DRIVER_FN + +// Compiler-library entry points, one table per library so that filling one +// does not load the others (NVVM and nvJitLink are optional at run time). +// Types are spelled out where the library header is not included; they match +// `nvvmResult nvvmDestroyProgram(nvvmProgram*)` and +// `nvJitLinkResult nvJitLinkDestroy(nvJitLinkHandle*)` as int-sized enums. +extern decltype(&nvrtcDestroyProgram) p_nvrtcDestroyProgram; using NvvmDestroyProgramFn = int (*)(nvvmProgram*); extern NvvmDestroyProgramFn p_nvvmDestroyProgram; +using NvJitLinkDestroyFn = int (*)(nvJitLink_t*); +extern NvJitLinkDestroyFn p_nvJitLinkDestroy; // ============================================================================ -// nvJitLink function pointers -// -// These are populated by _rt.pyx at module import time using -// function pointers extracted from cuda.bindings.cynvjitlink.__pyx_capi__. -// Note: May be null if nvJitLink is not available at runtime. +// Function tables // ============================================================================ -// Function pointer type for nvJitLinkDestroy (avoids nvJitLink.h dependency) -// Signature: nvJitLinkResult nvJitLinkDestroy(nvJitLinkHandle *handle) -using NvJitLinkDestroyFn = int (*)(nvJitLink_t*); -extern NvJitLinkDestroyFn p_nvJitLinkDestroy; - +enum class FnTable { driver, nvrtc, nvvm, nvjitlink }; + +struct FnEntry { + const char* key; // cuda-bindings' name for the slot, e.g. "__cuStreamDestroy_v2" + const char* name; // public name, for messages, e.g. "cuStreamDestroy" + void** slot; // the p_ pointer, as storage + int introduced; // CUDA version cuda-bindings requests the symbol at (0 = n/a) +}; + +// The entries of a table. Implemented in driver_api.cpp. +const FnEntry* fn_table_entries(FnTable table, std::size_t* count) noexcept; + +// Whether a table has been filled (acquire: a true result orders the slots). +bool fn_table_ready(FnTable table) noexcept; + +// Fill a table from cuda-bindings if it is not ready. Acquires the GIL (never +// call with a C++ lock held), imports cuda.bindings._internal., calls +// _inspect_function_pointers(), and stores every entry's pointer. Returns +// false, records the reason (fn_table_error) and reports it through +// report_message() when cuda-bindings cannot load the library, a key is +// missing (the installed cuda-bindings does not match the header this build +// compiled against), or a baseline driver function is null (the driver is +// older than the CUDA major series supports). Never leaves a Python error set. +// Implemented in py_driver_fns.cpp. +bool ensure_fn_table(FnTable table) noexcept; + +// The reason the last fill of `table` failed, or nullptr. Implemented in py_driver_fns.cpp. +const char* fn_table_error(FnTable table) noexcept; + +// Report, once per table, that `name` was called while unavailable: a gate +// bug, or a failed fill (whose reason is included). Implemented in py_driver_fns.cpp. +void report_unavailable_fn(FnTable table, const char* name) noexcept; + +namespace detail { + +// The status a trampoline returns in place of an unavailable function. +template +struct UnavailableStatus; +template <> +struct UnavailableStatus { + static constexpr CUresult value = CUDA_ERROR_NOT_INITIALIZED; +}; +template <> +struct UnavailableStatus { + static constexpr nvrtcResult value = NVRTC_ERROR_INTERNAL_ERROR; +}; +template <> +struct UnavailableStatus { + static constexpr int value = -1; +}; + +// A function of the same signature as an unavailable entry point, so that a +// call site never dereferences null. Its status flows to the caller's normal +// error handling; the cause has already been reported. +template +struct Unavailable; +template +struct Unavailable { + static R call(A...) noexcept { return UnavailableStatus::value; } +}; + +// The resolved pointer, filling the table on first use; the trampoline when +// the function is unavailable after the fill. Never throws. +template +inline F fn_or_unavailable(F& slot, FnTable table, const char* name) noexcept { + if (!fn_table_ready(table)) { + ensure_fn_table(table); + } + if (F fn = slot) { + return fn; + } + report_unavailable_fn(table, name); + return &Unavailable::call; +} + +} // namespace detail } // namespace cuda_core::rt + +// A driver function's table entry, resolved on first use. DRIVER_CALL(name, args...) +// calls it; use these for every driver call in the C++ layer except under a C++ +// lock (see the header comment; such a call is marked `// raw:`). Each macro +// pastes its own parameter: passing `name` through another macro would let +// cuda.h's versioning macros rewrite it (cuMemFree -> cuMemFree_v2) first. +#define DRIVER_FN(name) \ + (::cuda_core::rt::detail::fn_or_unavailable(::cuda_core::rt::p_##name, ::cuda_core::rt::FnTable::driver, #name)) +#define DRIVER_CALL(name, ...) \ + (::cuda_core::rt::detail::fn_or_unavailable(::cuda_core::rt::p_##name, ::cuda_core::rt::FnTable::driver, #name)(__VA_ARGS__)) +#define NVRTC_CALL(name, ...) \ + (::cuda_core::rt::detail::fn_or_unavailable(::cuda_core::rt::p_##name, ::cuda_core::rt::FnTable::nvrtc, #name)(__VA_ARGS__)) +#define NVVM_CALL(name, ...) \ + (::cuda_core::rt::detail::fn_or_unavailable(::cuda_core::rt::p_##name, ::cuda_core::rt::FnTable::nvvm, #name)(__VA_ARGS__)) +#define NVJITLINK_CALL(name, ...) \ + (::cuda_core::rt::detail::fn_or_unavailable(::cuda_core::rt::p_##name, ::cuda_core::rt::FnTable::nvjitlink, #name)(__VA_ARGS__)) diff --git a/cuda_core/cuda/core/_cpp/rt/error.cpp b/cuda_core/cuda/core/_cpp/rt/error.cpp index 38ceb20f240..2f9f06f285c 100644 --- a/cuda_core/cuda/core/_cpp/rt/error.cpp +++ b/cuda_core/cuda/core/_cpp/rt/error.cpp @@ -39,8 +39,8 @@ void format_cuda_error(char* buffer, size_t size, const char* operation, CUresul const char* error_name = nullptr; const char* error_description = nullptr; bool decoded = p_cuGetErrorName && p_cuGetErrorString - && p_cuGetErrorName(status, &error_name) == CUDA_SUCCESS - && p_cuGetErrorString(status, &error_description) == CUDA_SUCCESS; + && DRIVER_CALL(cuGetErrorName, status, &error_name) == CUDA_SUCCESS + && DRIVER_CALL(cuGetErrorString, status, &error_description) == CUDA_SUCCESS; const char* outcome = detail ? detail : "failed"; if (decoded) { std::snprintf(buffer, size, "%s %s: %s: %s", operation, outcome, error_name, error_description); @@ -100,13 +100,13 @@ namespace detail { void note_context_not_restored(CUcontext previous, CUresult operation_status, CUresult restore_status) noexcept { CUcontext current = nullptr; - if (p_cuCtxGetCurrent(¤t) != CUDA_SUCCESS) { + if (DRIVER_CALL(cuCtxGetCurrent, ¤t) != CUDA_SUCCESS) { current = nullptr; } char cause[128] = {0}; if (operation_status != CUDA_SUCCESS) { const char* error_name = nullptr; - if (p_cuGetErrorName && p_cuGetErrorName(restore_status, &error_name) == CUDA_SUCCESS) { + if (DRIVER_CALL(cuGetErrorName, restore_status, &error_name) == CUDA_SUCCESS) { std::snprintf(cause, sizeof(cause), " after this failure (cuCtxSetCurrent: %s)", error_name); } else { std::snprintf(cause, sizeof(cause), " after this failure (cuCtxSetCurrent: CUDA error %d)", diff --git a/cuda_core/cuda/core/_cpp/rt/event.cpp b/cuda_core/cuda/core/_cpp/rt/event.cpp index 7514f8b8e41..73b6e600a0e 100644 --- a/cuda_core/cuda/core/_cpp/rt/event.cpp +++ b/cuda_core/cuda/core/_cpp/rt/event.cpp @@ -69,7 +69,7 @@ EventHandle create_event_handle(const ContextHandle& h_ctx, unsigned int flags, CUevent event = nullptr; err = invoke_in_context_or_undo( h_ctx, - [&]() noexcept { return p_cuEventCreate(&event, flags); }, + [&]() noexcept { return DRIVER_CALL(cuEventCreate, &event, flags); }, [&]() noexcept { pw_cuEventDestroy(event); }, /*undo_requires_target_context=*/false); if (err != CUDA_SUCCESS) { @@ -97,7 +97,7 @@ EventHandle create_event_handle_for_stream(CUstream stream, unsigned int flags) CUcontext ctx = nullptr; { GILReleaseGuard gil; - err = p_cuStreamGetCtx(stream, &ctx); + err = DRIVER_CALL(cuStreamGetCtx, stream, &ctx); } if (err != CUDA_SUCCESS) { return {}; @@ -121,7 +121,7 @@ EventHandle create_event_handle_ipc(const CUipcEventHandle& ipc_handle, bool is_blocking_sync) { GILReleaseGuard gil; CUevent event; - if (CUDA_SUCCESS != (err = p_cuIpcOpenEventHandle(&event, ipc_handle))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuIpcOpenEventHandle, &event, ipc_handle))) { return {}; } diff --git a/cuda_core/cuda/core/_cpp/rt/graph.cpp b/cuda_core/cuda/core/_cpp/rt/graph.cpp index 4f6105dfdaa..d10633db7de 100644 --- a/cuda_core/cuda/core/_cpp/rt/graph.cpp +++ b/cuda_core/cuda/core/_cpp/rt/graph.cpp @@ -32,9 +32,6 @@ CUresult graph_node_set_params(CUgraphNode node, CUgraphNodeParams* params, const ContextHandle& h_context, CUresult* restore_status) noexcept { *restore_status = CUDA_SUCCESS; - if (!p_cuGraphNodeSetParams) { - return CUDA_ERROR_NOT_SUPPORTED; - } CUcontext previous = nullptr; int changed = 0; CUresult status = enter_context(h_context, &previous, &changed); @@ -43,7 +40,7 @@ CUresult graph_node_set_params(CUgraphNode node, CUgraphNodeParams* params, } { GILReleaseGuard gil; - status = p_cuGraphNodeSetParams(node, params); + status = DRIVER_CALL(cuGraphNodeSetParams, node, params); } if (!changed) { return status; @@ -149,15 +146,11 @@ CUresult rekey_attachments( if (!cloned_graph) { return CUDA_ERROR_INVALID_VALUE; } - if (!p_cuGraphNodeFindInClone) { - return CUDA_ERROR_NOT_SUPPORTED; - } - GraphAttachmentMap remapped; while (!attachments.empty()) { auto attachment = attachments.extract(attachments.begin()); CUgraphNode cloned_node = nullptr; - CUresult status = p_cuGraphNodeFindInClone( + CUresult status = DRIVER_CALL(cuGraphNodeFindInClone, &cloned_node, attachment.key(), cloned_graph); if (status != CUDA_SUCCESS) { return status; @@ -210,22 +203,18 @@ void stage_graph_metadata( // must be populated before entry. The caller must release the GIL. CUresult rekey_graph_metadata( StagedGraphMetadataList& staged) { - if (!p_cuGraphNodeFindInClone || !p_cuGraphChildGraphNodeGetGraph) { - return CUDA_ERROR_NOT_SUPPORTED; - } - CUresult status; for (size_t i = 0; i < staged.size(); ++i) { const GraphBox& source = *staged[i].source; GraphBox& clone = *staged[i].clone; if (i != 0) { CUgraphNode cloned_owner = nullptr; - status = p_cuGraphNodeFindInClone( + status = DRIVER_CALL(cuGraphNodeFindInClone, &cloned_owner, source.owner_node, clone.parent->resource); if (status == CUDA_SUCCESS) { - status = p_cuGraphChildGraphNodeGetGraph( + status = DRIVER_CALL(cuGraphChildGraphNodeGetGraph, cloned_owner, &clone.resource); } if (status != CUDA_SUCCESS) { @@ -307,6 +296,7 @@ struct PreparedChildGraphUpdateState { }; GraphHandle create_graph_handle(CUgraph graph) { + ensure_fn_table(FnTable::driver); // the deleter calls the driver; resolve before it can run if (!graph) { return {}; } @@ -432,9 +422,9 @@ CUresult graph_commit_child_graph_update( CUresult status = CUDA_ERROR_NOT_SUPPORTED; CUgraph cloned_root = nullptr; - if (p_cuGraphChildGraphNodeGetGraph) { + { GILReleaseGuard gil; - status = p_cuGraphChildGraphNodeGetGraph( + status = DRIVER_CALL(cuGraphChildGraphNodeGetGraph, state.owner_node, &cloned_root); if (status == CUDA_SUCCESS) { state.staged.front().clone->resource = cloned_root; @@ -504,19 +494,10 @@ CUresult graph_prepare_attachment( if (!box->resource) { return CUDA_ERROR_INVALID_VALUE; } - if (!p_cuGraphReleaseUserObject) { - return CUDA_ERROR_NOT_SUPPORTED; - } - PreparedAttachment prepared( new PreparedAttachmentState(h_graph), PreparedAttachmentDeleter{rollback_prepared_attachment}); if (owner0 || owner1) { - if (!p_cuUserObjectCreate || !p_cuUserObjectRelease || - !p_cuGraphRetainUserObject) { - return CUDA_ERROR_NOT_SUPPORTED; - } - ensure_deferred_cleanup_ready(); prepared->replacement = new NodeAttachment( std::move(owner0), std::move(owner1)); @@ -538,7 +519,7 @@ CUresult graph_prepare_attachment( CUresult status; { GILReleaseGuard gil; - status = p_cuUserObjectCreate( + status = DRIVER_CALL(cuUserObjectCreate, &object, cleanup_item, reinterpret_cast(enqueue_cleanup), 1, CU_USER_OBJECT_NO_DESTRUCTOR_SYNC); @@ -549,7 +530,7 @@ CUresult graph_prepare_attachment( return status; } prepared->replacement->object = object; - status = p_cuGraphRetainUserObject( + status = DRIVER_CALL(cuGraphRetainUserObject, box->resource, object, 1, CU_GRAPH_USER_OBJECT_MOVE); if (status != CUDA_SUCCESS) { prepared->replacement_entry.mapped() = nullptr; @@ -613,7 +594,7 @@ CUresult graph_commit_attachment( return CUDA_SUCCESS; } GILReleaseGuard gil; - return p_cuGraphReleaseUserObject( + return DRIVER_CALL(cuGraphReleaseUserObject, box->resource, previous->object, 1); } diff --git a/cuda_core/cuda/core/_cpp/rt/graph_exec.cpp b/cuda_core/cuda/core/_cpp/rt/graph_exec.cpp index dbf09cf77e0..4927274fb79 100644 --- a/cuda_core/cuda/core/_cpp/rt/graph_exec.cpp +++ b/cuda_core/cuda/core/_cpp/rt/graph_exec.cpp @@ -90,7 +90,7 @@ struct ExecAttachmentStaging { const GraphHandle source = std::move(h_source); accumulator = nullptr; GILReleaseGuard gil; - return p_cuGraphReleaseUserObject(*source, object, 1); + return DRIVER_CALL(cuGraphReleaseUserObject, *source, object, 1); } }; @@ -98,11 +98,6 @@ struct ExecAttachmentStaging { // instantiation or whole-graph update propagates a reference into the exec. CUresult stage_exec_attachments( const GraphHandle& h_source, ExecAttachmentStaging* out_staging) { - if (!p_cuUserObjectCreate || !p_cuUserObjectRelease || - !p_cuGraphRetainUserObject || !p_cuGraphReleaseUserObject) { - return CUDA_ERROR_NOT_SUPPORTED; - } - ensure_deferred_cleanup_ready(); auto* accumulator = new ExecAttachments; @@ -110,7 +105,7 @@ CUresult stage_exec_attachments( CUresult status; { GILReleaseGuard gil; - status = p_cuUserObjectCreate( + status = DRIVER_CALL(cuUserObjectCreate, &object, static_cast(accumulator), reinterpret_cast(enqueue_cleanup), @@ -121,7 +116,7 @@ CUresult stage_exec_attachments( return status; } accumulator->object = object; - status = p_cuGraphRetainUserObject( + status = DRIVER_CALL(cuGraphRetainUserObject, *h_source, object, 1, CU_GRAPH_USER_OBJECT_MOVE); if (status != CUDA_SUCCESS) { // Dropping the last reference retires the accumulator. @@ -174,11 +169,6 @@ GraphExecHandle create_graph_exec_handle( err = CUDA_ERROR_INVALID_VALUE; return {}; } - if (!p_cuGraphInstantiateWithParams) { - err = CUDA_ERROR_NOT_SUPPORTED; - return {}; - } - ExecAttachmentStaging staging; if (CUDA_SUCCESS != (err = stage_exec_attachments(h_source, &staging))) { return {}; @@ -187,7 +177,7 @@ GraphExecHandle create_graph_exec_handle( CUgraphExec graph_exec = nullptr; { GILReleaseGuard gil; - err = p_cuGraphInstantiateWithParams(&graph_exec, *h_source, params); + err = DRIVER_CALL(cuGraphInstantiateWithParams, &graph_exec, *h_source, params); } if (err != CUDA_SUCCESS) { return {}; @@ -218,10 +208,6 @@ CUresult graph_exec_update( if (!h_exec || !h_source || !*h_source || !result_info) { return CUDA_ERROR_INVALID_VALUE; } - if (!p_cuGraphExecUpdate) { - return CUDA_ERROR_NOT_SUPPORTED; - } - GraphExecBox* box = get_exec_box(h_exec); if (!box->resource) { return CUDA_ERROR_INVALID_VALUE; @@ -235,7 +221,7 @@ CUresult graph_exec_update( { GILReleaseGuard gil; - status = p_cuGraphExecUpdate(box->resource, *h_source, result_info); + status = DRIVER_CALL(cuGraphExecUpdate, box->resource, *h_source, result_info); } if (status != CUDA_SUCCESS) { return status; diff --git a/cuda_core/cuda/core/_cpp/rt/internal.hpp b/cuda_core/cuda/core/_cpp/rt/internal.hpp index 46b5a77f379..6b22acefc74 100644 --- a/cuda_core/cuda/core/_cpp/rt/internal.hpp +++ b/cuda_core/cuda/core/_cpp/rt/internal.hpp @@ -72,18 +72,21 @@ bool make_deallocation_stream(const StreamHandle& h, DeallocationStream& out) no // one while holding a C++ lock; the GIL must be the outermost lock. Where a // lock must stay held, call the p_ pointer, keep the status, and report after // the lock is released (see deviceptr_import_ipc and DESIGN.md). -template +template class WarnOnFailure { public: explicit WarnOnFailure(const char* operation) noexcept : operation_(operation) {} // The first argument is the resource being released; it is named in the // report so that independent failures are not collapsed by the warning - // registry (see format_operation). + // registry (see format_operation). The call goes through the function + // table like DRIVER_CALL: an unavailable entry is reported and yields an + // error status, never a null dereference. template auto operator()(First&& first, Rest&&... rest) const noexcept { const unsigned long long handle = handle_bits(first); - auto status = Function(std::forward(first), std::forward(rest)...); + auto status = detail::fn_or_unavailable(Function, Table, operation_)( + std::forward(first), std::forward(rest)...); report(status, handle); return status; } @@ -129,9 +132,9 @@ const WarnOnFailure pw_cuGraphicsUnregisterResou const WarnOnFailure pw_cuLinkDestroy{"cuLinkDestroy"}; const WarnOnFailure pw_cuUserObjectRelease{"cuUserObjectRelease"}; const WarnOnFailure pw_cuGraphReleaseUserObject{"cuGraphReleaseUserObject"}; -const WarnOnFailure pw_nvrtcDestroyProgram{"nvrtcDestroyProgram"}; -const WarnOnFailure pw_nvvmDestroyProgram{"nvvmDestroyProgram"}; -const WarnOnFailure pw_nvJitLinkDestroy{"nvJitLinkDestroy"}; +const WarnOnFailure pw_nvrtcDestroyProgram{"nvrtcDestroyProgram"}; +const WarnOnFailure pw_nvvmDestroyProgram{"nvvmDestroyProgram"}; +const WarnOnFailure pw_nvJitLinkDestroy{"nvJitLinkDestroy"}; // Intrusive base for payloads transferred out of CUDA's callback. struct DeferredCleanupItem { diff --git a/cuda_core/cuda/core/_cpp/rt/memory.cpp b/cuda_core/cuda/core/_cpp/rt/memory.cpp index 68a22c835bc..9ad1b64bccb 100644 --- a/cuda_core/cuda/core/_cpp/rt/memory.cpp +++ b/cuda_core/cuda/core/_cpp/rt/memory.cpp @@ -43,7 +43,7 @@ struct MemoryPoolBox { static void clear_mempool_peer_access(CUmemoryPool pool, int owner_device) noexcept { try { int device_count = 0; - if (p_cuDeviceGetCount(&device_count) != CUDA_SUCCESS || device_count <= 0) { + if (DRIVER_CALL(cuDeviceGetCount, &device_count) != CUDA_SUCCESS || device_count <= 0) { return; } @@ -55,7 +55,7 @@ static void clear_mempool_peer_access(CUmemoryPool pool, int owner_device) noexc continue; } revoke.location.id = i; - p_cuMemPoolSetAccess(pool, &revoke, 1); // Best effort + DRIVER_CALL(cuMemPoolSetAccess, pool, &revoke, 1); // Best effort } } catch (...) { // Swallow exceptions - this is best-effort cleanup in destructor context @@ -80,7 +80,7 @@ static MemoryPoolHandle wrap_mempool_owned(CUmemoryPool pool, int owner_device) MemoryPoolHandle create_mempool_handle(const CUmemPoolProps& props) { GILReleaseGuard gil; CUmemoryPool pool; - if (CUDA_SUCCESS != (err = p_cuMemPoolCreate(&pool, &props))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuMemPoolCreate, &pool, &props))) { return {}; } int owner_device = props.location.type == CU_MEM_LOCATION_TYPE_DEVICE ? props.location.id : -1; @@ -95,7 +95,7 @@ MemoryPoolHandle create_mempool_handle_ref(CUmemoryPool pool) { MemoryPoolHandle get_device_mempool(int device_id) { GILReleaseGuard gil; CUmemoryPool pool; - if (CUDA_SUCCESS != (err = p_cuDeviceGetMemPool(&pool, device_id))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuDeviceGetMemPool, &pool, device_id))) { return {}; } return create_mempool_handle_ref(pool); @@ -105,7 +105,7 @@ MemoryPoolHandle create_mempool_handle_ipc(int fd, CUmemAllocationHandleType han GILReleaseGuard gil; CUmemoryPool pool; auto handle_ptr = reinterpret_cast(static_cast(fd)); - if (CUDA_SUCCESS != (err = p_cuMemPoolImportFromShareableHandle(&pool, handle_ptr, handle_type, 0))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuMemPoolImportFromShareableHandle, &pool, handle_ptr, handle_type, 0))) { return {}; } return wrap_mempool_owned(pool, -1); @@ -158,7 +158,7 @@ CUresult set_deallocation_stream(const DevicePtrHandle& h, const StreamHandle& h DevicePtrHandle deviceptr_alloc_from_pool(size_t size, const MemoryPoolHandle& h_pool, const StreamHandle& h_stream) { GILReleaseGuard gil; CUdeviceptr ptr; - if (CUDA_SUCCESS != (err = p_cuMemAllocFromPoolAsync(&ptr, size, *h_pool, as_cu(h_stream)))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuMemAllocFromPoolAsync, &ptr, size, *h_pool, as_cu(h_stream)))) { return {}; } @@ -176,7 +176,7 @@ DevicePtrHandle deviceptr_alloc_from_pool(size_t size, const MemoryPoolHandle& h cleanup_in_context( deallocation_context(stream), "cuMemFreeAsync", handle_bits(b->resource), [&]() noexcept { - return p_cuMemFreeAsync( + return DRIVER_CALL(cuMemFreeAsync, b->resource, as_cu(stream.h_stream)); }); delete b; @@ -188,7 +188,7 @@ DevicePtrHandle deviceptr_alloc_from_pool(size_t size, const MemoryPoolHandle& h DevicePtrHandle deviceptr_alloc_async(size_t size, const StreamHandle& h_stream) { GILReleaseGuard gil; CUdeviceptr ptr; - if (CUDA_SUCCESS != (err = p_cuMemAllocAsync(&ptr, size, as_cu(h_stream)))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuMemAllocAsync, &ptr, size, as_cu(h_stream)))) { return {}; } @@ -206,7 +206,7 @@ DevicePtrHandle deviceptr_alloc_async(size_t size, const StreamHandle& h_stream) cleanup_in_context( deallocation_context(stream), "cuMemFreeAsync", handle_bits(b->resource), [&]() noexcept { - return p_cuMemFreeAsync( + return DRIVER_CALL(cuMemFreeAsync, b->resource, as_cu(stream.h_stream)); }); delete b; @@ -221,7 +221,7 @@ CUresult deviceptr_alloc_raw(CUdeviceptr* ptr, size_t size, GILReleaseGuard gil; return invoke_in_context_or_undo( h_context, - [&]() noexcept { return p_cuMemAlloc(ptr, size); }, + [&]() noexcept { return DRIVER_CALL(cuMemAlloc, ptr, size); }, [&]() noexcept { pw_cuMemFree(*ptr); }, /*undo_requires_target_context=*/false); } @@ -229,7 +229,7 @@ CUresult deviceptr_alloc_raw(CUdeviceptr* ptr, size_t size, DevicePtrHandle deviceptr_alloc_host(size_t size) { GILReleaseGuard gil; void* ptr; - if (CUDA_SUCCESS != (err = p_cuMemAllocHost(&ptr, size))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuMemAllocHost, &ptr, size))) { return {}; } @@ -291,7 +291,7 @@ DevicePtrHandle deviceptr_create_mapped_graphics( cleanup_in_context( deallocation_context(stream), "cuGraphicsUnmapResources", handle_bits(resource), [&]() noexcept { - return p_cuGraphicsUnmapResources( + return DRIVER_CALL(cuGraphicsUnmapResources, 1, &resource, as_cu(stream.h_stream)); }); delete b; @@ -404,6 +404,10 @@ DevicePtrHandle deviceptr_import_ipc(const MemoryPoolHandle& h_pool, const void* auto data = const_cast( reinterpret_cast(export_data)); + // Resolve the table before any lock is taken: a fill acquires the GIL, and + // nothing under ipc_import_mutex may (#2840). The raw p_ calls below rely on it. + ensure_fn_table(FnTable::driver); + if (use_ipc_ptr_cache()) { ExportDataKey key; std::memcpy(&key.data, data, sizeof(key.data)); @@ -425,7 +429,7 @@ DevicePtrHandle deviceptr_import_ipc(const MemoryPoolHandle& h_pool, const void* } CUdeviceptr ptr; - if (CUDA_SUCCESS != (err = p_cuMemPoolImportPointer(&ptr, *h_pool, data))) { + if (CUDA_SUCCESS != (err = p_cuMemPoolImportPointer(&ptr, *h_pool, data))) { // raw: under ipc_import_mutex return {}; } @@ -449,7 +453,7 @@ DevicePtrHandle deviceptr_import_ipc(const MemoryPoolHandle& h_pool, const void* cleanup_in_context( h_dealloc, "cuMemFreeAsync", handle_bits(b->resource), [&]() noexcept { - return p_cuMemFreeAsync(b->resource, as_cu(stream.h_stream)); + return p_cuMemFreeAsync(b->resource, as_cu(stream.h_stream)); // raw: under ipc_import_mutex }, [&]() noexcept { lock.unlock(); }); delete b; @@ -462,7 +466,7 @@ DevicePtrHandle deviceptr_import_ipc(const MemoryPoolHandle& h_pool, const void* // No deallocation stream could be recorded: discard the import with // the raw call (a pw_ report would acquire the GIL under the mutex). - discard_status = p_cuMemFreeAsync(ptr, as_cu(h_stream)); + discard_status = p_cuMemFreeAsync(ptr, as_cu(h_stream)); // raw: under ipc_import_mutex discarded = ptr; } if (discard_status != CUDA_SUCCESS) { @@ -477,7 +481,7 @@ DevicePtrHandle deviceptr_import_ipc(const MemoryPoolHandle& h_pool, const void* } else { GILReleaseGuard gil; CUdeviceptr ptr; - if (CUDA_SUCCESS != (err = p_cuMemPoolImportPointer(&ptr, *h_pool, data))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuMemPoolImportPointer, &ptr, *h_pool, data))) { return {}; } @@ -495,7 +499,7 @@ DevicePtrHandle deviceptr_import_ipc(const MemoryPoolHandle& h_pool, const void* cleanup_in_context( deallocation_context(stream), "cuMemFreeAsync", handle_bits(b->resource), [&]() noexcept { - return p_cuMemFreeAsync( + return DRIVER_CALL(cuMemFreeAsync, b->resource, as_cu(stream.h_stream)); }); delete b; diff --git a/cuda_core/cuda/core/_cpp/rt/program.cpp b/cuda_core/cuda/core/_cpp/rt/program.cpp index 74c6455264a..2a41f8fb125 100644 --- a/cuda_core/cuda/core/_cpp/rt/program.cpp +++ b/cuda_core/cuda/core/_cpp/rt/program.cpp @@ -28,7 +28,7 @@ struct LibraryBox { LibraryHandle create_library_handle_from_file(const char* path) { GILReleaseGuard gil; CUlibrary library; - if (CUDA_SUCCESS != (err = p_cuLibraryLoadFromFile(&library, path, nullptr, nullptr, 0, nullptr, nullptr, 0))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuLibraryLoadFromFile, &library, path, nullptr, nullptr, 0, nullptr, nullptr, 0))) { return {}; } @@ -47,7 +47,7 @@ LibraryHandle create_library_handle_from_file(const char* path) { LibraryHandle create_library_handle_from_data(const void* data) { GILReleaseGuard gil; CUlibrary library; - if (CUDA_SUCCESS != (err = p_cuLibraryLoadData(&library, data, nullptr, nullptr, 0, nullptr, nullptr, 0))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuLibraryLoadData, &library, data, nullptr, nullptr, 0, nullptr, nullptr, 0))) { return {}; } @@ -92,7 +92,7 @@ static HandleRegistry kernel_registry; KernelHandle create_kernel_handle(const LibraryHandle& h_library, const char* name) { GILReleaseGuard gil; CUkernel kernel; - if (CUDA_SUCCESS != (err = p_cuLibraryGetKernel(&kernel, *h_library, name))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuLibraryGetKernel, &kernel, *h_library, name))) { return {}; } @@ -126,15 +126,16 @@ struct NvrtcProgramBox { } // namespace NvrtcProgramHandle create_nvrtc_program_handle(nvrtcProgram prog) { + // Resolve the table now, while the library that created `prog` is loaded, + // so the deleter never has to. + ensure_fn_table(FnTable::nvrtc); auto box = std::shared_ptr( new NvrtcProgramBox{prog}, [](NvrtcProgramBox* b) { // Note: nvrtcDestroyProgram takes nvrtcProgram* and nulls it, // but we're deleting the box anyway so nulling is harmless. - if (p_nvrtcDestroyProgram) { - GILReleaseGuard gil; - pw_nvrtcDestroyProgram(&b->resource); - } + GILReleaseGuard gil; + pw_nvrtcDestroyProgram(&b->resource); delete b; } ); @@ -157,16 +158,14 @@ struct NvvmProgramBox { } // namespace NvvmProgramHandle create_nvvm_program_handle(nvvmProgram prog) { + ensure_fn_table(FnTable::nvvm); auto box = std::shared_ptr( new NvvmProgramBox{{prog}}, [](NvvmProgramBox* b) { // Note: nvvmDestroyProgram takes nvvmProgram* and nulls it, // but we're deleting the box anyway so nulling is harmless. - // If NVVM is not available, the function pointer is null. - if (p_nvvmDestroyProgram) { - GILReleaseGuard gil; - pw_nvvmDestroyProgram(&b->resource.raw); - } + GILReleaseGuard gil; + pw_nvvmDestroyProgram(&b->resource.raw); delete b; } ); @@ -189,16 +188,14 @@ struct NvJitLinkBox { } // namespace NvJitLinkHandle create_nvjitlink_handle(nvJitLink_t handle) { + ensure_fn_table(FnTable::nvjitlink); auto box = std::shared_ptr( new NvJitLinkBox{{handle}}, [](NvJitLinkBox* b) { // Note: nvJitLinkDestroy takes nvJitLinkHandle* and nulls it, // but we're deleting the box anyway so nulling is harmless. - // If nvJitLink is not available, the function pointer is null. - if (p_nvJitLinkDestroy) { - GILReleaseGuard gil; - pw_nvJitLinkDestroy(&b->resource.raw); - } + GILReleaseGuard gil; + pw_nvJitLinkDestroy(&b->resource.raw); delete b; } ); @@ -221,14 +218,13 @@ struct CuLinkBox { } // namespace CuLinkHandle create_culink_handle(CUlinkState state) { + ensure_fn_table(FnTable::driver); auto box = std::shared_ptr( new CuLinkBox{state}, [](CuLinkBox* b) { // cuLinkDestroy takes CUlinkState by value (not pointer). - if (p_cuLinkDestroy) { - GILReleaseGuard gil; - pw_cuLinkDestroy(b->resource); - } + GILReleaseGuard gil; + pw_cuLinkDestroy(b->resource); delete b; } ); diff --git a/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp b/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp new file mode 100644 index 00000000000..bf4a31e8e1e --- /dev/null +++ b/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp @@ -0,0 +1,229 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// +// SPDX-License-Identifier: Apache-2.0 + +// Filling the driver and compiler-library function tables from cuda-bindings. +// +// cuda-bindings loads each library and resolves its symbols once (for the +// driver, with cuGetProcAddress). cuda.bindings._internal. +// ._inspect_function_pointers() returns that table as {name: address}, where a +// zero address means the library does not provide the symbol. This file copies +// the entries cuda.core uses into the p_ pointers declared in driver_api.hpp. +// +// The fill runs Python, so it acquires the GIL and must never run under a C++ +// lock (the GIL is the outermost lock; see DESIGN.md). The slot stores happen +// under fill_mutex with no Python call inside, and the ready flag is published +// with release semantics after them, so readers that see the flag see the +// pointers. Two threads may both compute the table; they store identical +// values, one after the other. +// +// Failures never propagate as exceptions and never leave a Python error set: +// they are recorded, reported through report_message(), and every affected +// DRIVER_CALL then returns an error status from a trampoline (driver_api.hpp). + +#include "py.hpp" +#include "driver_api.hpp" +#include "error.hpp" +#include +#include +#include +#include +#include + +namespace cuda_core::rt { + +namespace { + +constexpr std::size_t kTables = 4; +constexpr std::size_t kMaxEntries = 128; + +std::atomic table_ready[kTables]; +std::atomic unavailable_reported[kTables]; +std::mutex fill_mutex; +char fill_error[kTables][512] = {}; + +std::size_t index_of(FnTable table) noexcept { return static_cast(table); } + +const char* module_name(FnTable table) noexcept { + switch (table) { + case FnTable::driver: return "cuda.bindings._internal.driver"; + case FnTable::nvrtc: return "cuda.bindings._internal.nvrtc"; + case FnTable::nvvm: return "cuda.bindings._internal.nvvm"; + case FnTable::nvjitlink: return "cuda.bindings._internal.nvjitlink"; + } + return "cuda.bindings._internal"; +} + +const char* library_name(FnTable table) noexcept { + switch (table) { + case FnTable::driver: return "CUDA driver"; + case FnTable::nvrtc: return "NVRTC"; + case FnTable::nvvm: return "NVVM"; + case FnTable::nvjitlink: return "nvJitLink"; + } + return "library"; +} + +// Copy the pending Python exception's text into buf and clear it. +void take_python_error(char* buf, std::size_t size) noexcept { +#if PY_VERSION_HEX >= 0x030C0000 + PyObject* exc = PyErr_GetRaisedException(); +#else + PyObject *type, *value, *traceback; + PyErr_Fetch(&type, &value, &traceback); + PyErr_NormalizeException(&type, &value, &traceback); + PyObject* exc = value; + Py_XDECREF(type); + Py_XDECREF(traceback); +#endif + PyObject* text = exc ? PyObject_Str(exc) : nullptr; + const char* utf8 = text ? PyUnicode_AsUTF8(text) : nullptr; + std::snprintf(buf, size, "%s", utf8 ? utf8 : "unknown error"); + Py_XDECREF(text); + Py_XDECREF(exc); + PyErr_Clear(); +} + +void record_failure(FnTable table, const char* message) noexcept { + { + std::lock_guard lock(fill_mutex); + std::snprintf(fill_error[index_of(table)], sizeof(fill_error[0]), "%s", message); + } + report_message(message); +} + +} // namespace + +bool fn_table_ready(FnTable table) noexcept { + return table_ready[index_of(table)].load(std::memory_order_acquire); +} + +const char* fn_table_error(FnTable table) noexcept { + const char* text = fill_error[index_of(table)]; + return text[0] ? text : nullptr; +} + +bool ensure_fn_table(FnTable table) noexcept { + const std::size_t idx = index_of(table); + if (table_ready[idx].load(std::memory_order_acquire)) { + return true; + } + std::size_t count = 0; + const FnEntry* entries = fn_table_entries(table, &count); + if (entries == nullptr || count > kMaxEntries) { + record_failure(table, "internal cuda.core error, please report: function table has an unexpected size"); + return false; + } + if (!Py_IsInitialized() || py_is_finalizing()) { + record_failure(table, "cuda.core cannot resolve driver functions while the interpreter is shutting down"); + return false; + } + + char message[640]; + char cause[384]; + void* values[kMaxEntries] = {}; + + GILAcquireGuard gil; + if (!gil.acquired()) { + record_failure(table, "cuda.core cannot resolve driver functions while the interpreter is shutting down"); + return false; + } + + PyObject* module = PyImport_ImportModule(module_name(table)); + if (module == nullptr) { + take_python_error(cause, sizeof(cause)); + std::snprintf(message, sizeof(message), + "cuda.core cannot import %s from the installed cuda-bindings: %s", module_name(table), cause); + record_failure(table, message); + return false; + } + PyObject* pointers = PyObject_CallMethod(module, "_inspect_function_pointers", nullptr); + Py_DECREF(module); + if (pointers == nullptr) { + take_python_error(cause, sizeof(cause)); + std::snprintf(message, sizeof(message), "cuda-bindings could not load the %s: %s", library_name(table), cause); + record_failure(table, message); + return false; + } + if (!PyDict_Check(pointers)) { + Py_DECREF(pointers); + std::snprintf(message, sizeof(message), + "internal cuda.core error, please report: %s._inspect_function_pointers() did not return a dict", + module_name(table)); + record_failure(table, message); + return false; + } + for (std::size_t i = 0; i < count; ++i) { + PyObject* item = PyDict_GetItemString(pointers, entries[i].key); // borrowed; no error on a miss + if (item == nullptr) { + Py_DECREF(pointers); + std::snprintf(message, sizeof(message), + "the installed cuda-bindings has no entry for %s (%s). cuda.core was compiled against a " + "cuda.h that names this symbol differently than the cuda-bindings in use; install the " + "cuda-bindings this cuda.core requires", + entries[i].name, entries[i].key); + record_failure(table, message); + return false; + } + void* address = PyLong_AsVoidPtr(item); + if (address == nullptr && PyErr_Occurred()) { + Py_DECREF(pointers); + take_python_error(cause, sizeof(cause)); + std::snprintf(message, sizeof(message), + "internal cuda.core error, please report: the entry for %s in %s is not an address: %s", + entries[i].key, module_name(table), cause); + record_failure(table, message); + return false; + } + values[i] = address; + } + Py_DECREF(pointers); + + // Every function introduced at or before the first release of the CUDA + // major series is present in every driver cuda.core supports; a null one + // means the driver is older than that. Newer functions may be null and are + // gated on the driver version in Cython. + if (table == FnTable::driver) { + for (std::size_t i = 0; i < count; ++i) { + if (values[i] == nullptr && entries[i].introduced <= CUDA_CORE_BUILD_MAJOR * 1000) { + std::snprintf(message, sizeof(message), + "the installed CUDA driver does not provide %s, which every driver of the CUDA %d " + "series provides; this cuda.core build requires a CUDA %d driver", + entries[i].name, CUDA_CORE_BUILD_MAJOR, CUDA_CORE_BUILD_MAJOR); + record_failure(table, message); + return false; + } + } + } + + { + std::lock_guard lock(fill_mutex); + if (!table_ready[idx].load(std::memory_order_relaxed)) { + for (std::size_t i = 0; i < count; ++i) { + *entries[i].slot = values[i]; + } + fill_error[idx][0] = 0; + table_ready[idx].store(true, std::memory_order_release); + } + } + return true; +} + +void report_unavailable_fn(FnTable table, const char* name) noexcept { + // Once per table: the first unavailable call is the informative one. + if (unavailable_reported[index_of(table)].exchange(true)) { + return; + } + char message[768]; + if (const char* reason = fn_table_error(table)) { + std::snprintf(message, sizeof(message), "cuda.core could not call %s: %s", name, reason); + } else { + std::snprintf(message, sizeof(message), + "internal cuda.core error, please report: %s was called but the installed %s does not " + "provide it; a feature gate is missing or wrong. The call returned an error instead.", + name, library_name(table)); + } + report_message(message); +} + +} // namespace cuda_core::rt diff --git a/cuda_core/cuda/core/_cpp/rt/stream.cpp b/cuda_core/cuda/core/_cpp/rt/stream.cpp index eb17b428f11..1cff809e50a 100644 --- a/cuda_core/cuda/core/_cpp/rt/stream.cpp +++ b/cuda_core/cuda/core/_cpp/rt/stream.cpp @@ -70,13 +70,12 @@ StreamHandle create_stream_handle(const ContextHandle& h_ctx, unsigned int flags CUstream stream = nullptr; GreenCtxHandle h_green = get_context_green_ctx(h_ctx); if (h_green) { - err = p_cuGreenCtxStreamCreate - ? p_cuGreenCtxStreamCreate(&stream, as_cu(h_green), flags, priority) - : CUDA_ERROR_NOT_SUPPORTED; + // Gated in Cython on driver >= 12.5 (cuGreenCtxStreamCreate's introduction). + err = DRIVER_CALL(cuGreenCtxStreamCreate, &stream, as_cu(h_green), flags, priority); } else { err = invoke_in_context_or_undo( h_ctx, - [&]() noexcept { return p_cuStreamCreateWithPriority(&stream, flags, priority); }, + [&]() noexcept { return DRIVER_CALL(cuStreamCreateWithPriority, &stream, flags, priority); }, [&]() noexcept { pw_cuStreamDestroy(stream); }, /*undo_requires_target_context=*/false); } diff --git a/cuda_core/cuda/core/_cpp/rt/texture.cpp b/cuda_core/cuda/core/_cpp/rt/texture.cpp index c667df28bd8..b3f34e08cd5 100644 --- a/cuda_core/cuda/core/_cpp/rt/texture.cpp +++ b/cuda_core/cuda/core/_cpp/rt/texture.cpp @@ -27,6 +27,7 @@ struct GraphicsResourceBox { } // namespace GraphicsResourceHandle create_graphics_resource_handle(CUgraphicsResource resource) { + ensure_fn_table(FnTable::driver); // the deleter calls the driver; resolve before it can run auto box = std::shared_ptr( new GraphicsResourceBox{resource}, [](const GraphicsResourceBox* b) { @@ -112,7 +113,7 @@ OpaqueArrayHandle create_array_handle(const ContextHandle& h_context, const CUDA CUarray arr = nullptr; err = invoke_in_context_or_undo( h_context, - [&]() noexcept { return p_cuArray3DCreate(&arr, &desc); }, + [&]() noexcept { return DRIVER_CALL(cuArray3DCreate, &arr, &desc); }, [&]() noexcept { pw_cuArrayDestroy(arr); }, /*undo_requires_target_context=*/false); if (err != CUDA_SUCCESS) { @@ -130,6 +131,7 @@ OpaqueArrayHandle create_array_handle_ref(CUarray arr) { } OpaqueArrayHandle create_array_handle_owning(CUarray arr) { + ensure_fn_table(FnTable::driver); // the deleter calls the driver; resolve before it can run if (!arr) { return {}; } @@ -145,7 +147,7 @@ OpaqueArrayHandle create_array_level_handle(const MipmappedArrayHandle& h_mip, u GILReleaseGuard gil; CUarray arr; ContextHandle h_context = h_mip ? get_box(h_mip)->h_context : ContextHandle{}; - if (CUDA_SUCCESS != (err = p_cuMipmappedArrayGetLevel(&arr, as_cu(h_mip), level))) { + if (CUDA_SUCCESS != (err = DRIVER_CALL(cuMipmappedArrayGetLevel, &arr, as_cu(h_mip), level))) { return {}; } // Non-owning level view: storage belongs to the mipmap. Embed the mipmap @@ -164,7 +166,7 @@ MipmappedArrayHandle create_mipmapped_array_handle(const ContextHandle& h_contex CUmipmappedArray mip = nullptr; err = invoke_in_context_or_undo( h_context, - [&]() noexcept { return p_cuMipmappedArrayCreate(&mip, &desc, num_levels); }, + [&]() noexcept { return DRIVER_CALL(cuMipmappedArrayCreate, &mip, &desc, num_levels); }, [&]() noexcept { pw_cuMipmappedArrayDestroy(mip); }, /*undo_requires_target_context=*/false); if (err != CUDA_SUCCESS) { @@ -195,7 +197,7 @@ TexObjectHandle make_tex_object_handle(const CUDA_RESOURCE_DESC& res, CUtexObject obj = 0; err = invoke_in_context_or_undo( h_context, - [&]() noexcept { return p_cuTexObjectCreate(&obj, &res, &tex, nullptr); }, + [&]() noexcept { return DRIVER_CALL(cuTexObjectCreate, &obj, &res, &tex, nullptr); }, [&]() noexcept { pw_cuTexObjectDestroy(obj); }, /*undo_requires_target_context=*/true); if (err != CUDA_SUCCESS) { @@ -206,7 +208,7 @@ TexObjectHandle make_tex_object_handle(const CUDA_RESOURCE_DESC& res, [](const TexObjectBox* b) { GILReleaseGuard gil; cleanup_in_context(b->h_context, "cuTexObjectDestroy", handle_bits(b->resource.raw), [&]() noexcept { - return p_cuTexObjectDestroy(b->resource.raw); + return DRIVER_CALL(cuTexObjectDestroy, b->resource.raw); }); delete b; } @@ -243,7 +245,7 @@ SurfObjectHandle create_surf_object_handle(const ContextHandle& h_context, CUsurfObject obj = 0; err = invoke_in_context_or_undo( h_context, - [&]() noexcept { return p_cuSurfObjectCreate(&obj, &res); }, + [&]() noexcept { return DRIVER_CALL(cuSurfObjectCreate, &obj, &res); }, [&]() noexcept { pw_cuSurfObjectDestroy(obj); }, /*undo_requires_target_context=*/true); if (err != CUDA_SUCCESS) { @@ -254,7 +256,7 @@ SurfObjectHandle create_surf_object_handle(const ContextHandle& h_context, [](const SurfObjectBox* b) { GILReleaseGuard gil; cleanup_in_context(b->h_context, "cuSurfObjectDestroy", handle_bits(b->resource.raw), [&]() noexcept { - return p_cuSurfObjectDestroy(b->resource.raw); + return DRIVER_CALL(cuSurfObjectDestroy, b->resource.raw); }); delete b; } diff --git a/cuda_core/cuda/core/_rt.pyx b/cuda_core/cuda/core/_rt.pyx index f2e2ad1cb3c..74ec27f942f 100644 --- a/cuda_core/cuda/core/_rt.pyx +++ b/cuda_core/cuda/core/_rt.pyx @@ -15,7 +15,6 @@ # without needing separate wrapper functions. from cpython.object cimport PyObject -from cpython.pycapsule cimport PyCapsule_GetName, PyCapsule_GetPointer from libc.stddef cimport size_t from cuda.bindings cimport cydriver @@ -23,10 +22,6 @@ from cuda.bindings cimport cynvrtc from cuda.bindings cimport cynvvm from cuda.bindings cimport cynvjitlink -import cuda.bindings.cydriver as cydriver -import cuda.bindings.cynvrtc as cynvrtc -import cuda.bindings.cynvvm as cynvvm -import cuda.bindings.cynvjitlink as cynvjitlink # ============================================================================= # C++ function declarations (non-inline, implemented under _cpp/rt/) @@ -301,252 +296,6 @@ cdef extern from "_cpp/rt/rt.hpp" namespace "cuda_core::rt": const OpaqueArrayHandle& h_backing) except+ nogil -# ============================================================================= -# CUDA driver function pointer initialization -# -# The C++ code declares extern function pointers (p_cuXxx) that need to be -# populated before any handle creation functions are called. We extract these -# from cuda.bindings.cydriver.__pyx_capi__ at module import time. -# -# The Cython string substitution (e.g., "reinterpret_cast(...)") -# allows us to assign void* values to typed function pointer variables. -# ============================================================================= - -# Declare extern variables with reinterpret_cast to allow void* assignment -cdef extern from "_cpp/rt/rt.hpp" namespace "cuda_core::rt": - # Error formatting - void* p_cuGetErrorName "reinterpret_cast(cuda_core::rt::p_cuGetErrorName)" - void* p_cuGetErrorString "reinterpret_cast(cuda_core::rt::p_cuGetErrorString)" - - # Context - void* p_cuDevicePrimaryCtxRetain "reinterpret_cast(cuda_core::rt::p_cuDevicePrimaryCtxRetain)" - void* p_cuDevicePrimaryCtxRelease "reinterpret_cast(cuda_core::rt::p_cuDevicePrimaryCtxRelease)" - void* p_cuCtxGetCurrent "reinterpret_cast(cuda_core::rt::p_cuCtxGetCurrent)" - void* p_cuCtxSetCurrent "reinterpret_cast(cuda_core::rt::p_cuCtxSetCurrent)" - void* p_cuCtxSynchronize "reinterpret_cast(cuda_core::rt::p_cuCtxSynchronize)" - void* p_cuCtxGetStreamPriorityRange "reinterpret_cast(cuda_core::rt::p_cuCtxGetStreamPriorityRange)" - void* p_cuCtxGetDevice "reinterpret_cast(cuda_core::rt::p_cuCtxGetDevice)" - void* p_cuGraphNodeSetParams "reinterpret_cast(cuda_core::rt::p_cuGraphNodeSetParams)" - void* p_cuGreenCtxCreate "reinterpret_cast(cuda_core::rt::p_cuGreenCtxCreate)" - void* p_cuGreenCtxDestroy "reinterpret_cast(cuda_core::rt::p_cuGreenCtxDestroy)" - void* p_cuCtxFromGreenCtx "reinterpret_cast(cuda_core::rt::p_cuCtxFromGreenCtx)" - void* p_cuDevResourceGenerateDesc "reinterpret_cast(cuda_core::rt::p_cuDevResourceGenerateDesc)" - void* p_cuGreenCtxStreamCreate "reinterpret_cast(cuda_core::rt::p_cuGreenCtxStreamCreate)" - - # Stream - void* p_cuStreamCreateWithPriority "reinterpret_cast(cuda_core::rt::p_cuStreamCreateWithPriority)" - void* p_cuStreamDestroy "reinterpret_cast(cuda_core::rt::p_cuStreamDestroy)" - void* p_cuStreamGetCtx "reinterpret_cast(cuda_core::rt::p_cuStreamGetCtx)" - - # Event - void* p_cuEventCreate "reinterpret_cast(cuda_core::rt::p_cuEventCreate)" - void* p_cuEventDestroy "reinterpret_cast(cuda_core::rt::p_cuEventDestroy)" - void* p_cuIpcOpenEventHandle "reinterpret_cast(cuda_core::rt::p_cuIpcOpenEventHandle)" - - # Device - void* p_cuDeviceGetCount "reinterpret_cast(cuda_core::rt::p_cuDeviceGetCount)" - - # Memory pool - void* p_cuMemPoolSetAccess "reinterpret_cast(cuda_core::rt::p_cuMemPoolSetAccess)" - void* p_cuMemPoolDestroy "reinterpret_cast(cuda_core::rt::p_cuMemPoolDestroy)" - void* p_cuMemPoolCreate "reinterpret_cast(cuda_core::rt::p_cuMemPoolCreate)" - void* p_cuDeviceGetMemPool "reinterpret_cast(cuda_core::rt::p_cuDeviceGetMemPool)" - void* p_cuMemPoolImportFromShareableHandle "reinterpret_cast(cuda_core::rt::p_cuMemPoolImportFromShareableHandle)" - - # Memory allocation - void* p_cuMemAllocFromPoolAsync "reinterpret_cast(cuda_core::rt::p_cuMemAllocFromPoolAsync)" - void* p_cuMemAllocAsync "reinterpret_cast(cuda_core::rt::p_cuMemAllocAsync)" - void* p_cuMemAlloc "reinterpret_cast(cuda_core::rt::p_cuMemAlloc)" - void* p_cuMemAllocHost "reinterpret_cast(cuda_core::rt::p_cuMemAllocHost)" - - # Memory deallocation - void* p_cuMemFreeAsync "reinterpret_cast(cuda_core::rt::p_cuMemFreeAsync)" - void* p_cuMemFree "reinterpret_cast(cuda_core::rt::p_cuMemFree)" - void* p_cuMemFreeHost "reinterpret_cast(cuda_core::rt::p_cuMemFreeHost)" - - # IPC - void* p_cuMemPoolImportPointer "reinterpret_cast(cuda_core::rt::p_cuMemPoolImportPointer)" - - # Library - void* p_cuLibraryLoadFromFile "reinterpret_cast(cuda_core::rt::p_cuLibraryLoadFromFile)" - void* p_cuLibraryLoadData "reinterpret_cast(cuda_core::rt::p_cuLibraryLoadData)" - void* p_cuLibraryUnload "reinterpret_cast(cuda_core::rt::p_cuLibraryUnload)" - void* p_cuLibraryGetKernel "reinterpret_cast(cuda_core::rt::p_cuLibraryGetKernel)" - - # Graph - void* p_cuGraphDestroy "reinterpret_cast(cuda_core::rt::p_cuGraphDestroy)" - void* p_cuGraphInstantiateWithParams "reinterpret_cast(cuda_core::rt::p_cuGraphInstantiateWithParams)" - void* p_cuGraphExecUpdate "reinterpret_cast(cuda_core::rt::p_cuGraphExecUpdate)" - void* p_cuGraphExecDestroy "reinterpret_cast(cuda_core::rt::p_cuGraphExecDestroy)" - void* p_cuUserObjectCreate "reinterpret_cast(cuda_core::rt::p_cuUserObjectCreate)" - void* p_cuUserObjectRelease "reinterpret_cast(cuda_core::rt::p_cuUserObjectRelease)" - void* p_cuGraphRetainUserObject "reinterpret_cast(cuda_core::rt::p_cuGraphRetainUserObject)" - void* p_cuGraphReleaseUserObject "reinterpret_cast(cuda_core::rt::p_cuGraphReleaseUserObject)" - void* p_cuGraphNodeFindInClone "reinterpret_cast(cuda_core::rt::p_cuGraphNodeFindInClone)" - void* p_cuGraphChildGraphNodeGetGraph "reinterpret_cast(cuda_core::rt::p_cuGraphChildGraphNodeGetGraph)" - - # Linker - void* p_cuLinkDestroy "reinterpret_cast(cuda_core::rt::p_cuLinkDestroy)" - - # Graphics interop - void* p_cuGraphicsUnmapResources "reinterpret_cast(cuda_core::rt::p_cuGraphicsUnmapResources)" - void* p_cuGraphicsUnregisterResource "reinterpret_cast(cuda_core::rt::p_cuGraphicsUnregisterResource)" - - # Texture / surface / array (PR #467) - void* p_cuArray3DCreate "reinterpret_cast(cuda_core::rt::p_cuArray3DCreate)" - void* p_cuArrayDestroy "reinterpret_cast(cuda_core::rt::p_cuArrayDestroy)" - void* p_cuMipmappedArrayCreate "reinterpret_cast(cuda_core::rt::p_cuMipmappedArrayCreate)" - void* p_cuMipmappedArrayDestroy "reinterpret_cast(cuda_core::rt::p_cuMipmappedArrayDestroy)" - void* p_cuMipmappedArrayGetLevel "reinterpret_cast(cuda_core::rt::p_cuMipmappedArrayGetLevel)" - void* p_cuTexObjectCreate "reinterpret_cast(cuda_core::rt::p_cuTexObjectCreate)" - void* p_cuTexObjectDestroy "reinterpret_cast(cuda_core::rt::p_cuTexObjectDestroy)" - void* p_cuSurfObjectCreate "reinterpret_cast(cuda_core::rt::p_cuSurfObjectCreate)" - void* p_cuSurfObjectDestroy "reinterpret_cast(cuda_core::rt::p_cuSurfObjectDestroy)" - - # NVRTC - void* p_nvrtcDestroyProgram "reinterpret_cast(cuda_core::rt::p_nvrtcDestroyProgram)" - - # NVVM - void* p_nvvmDestroyProgram "reinterpret_cast(cuda_core::rt::p_nvvmDestroyProgram)" - - # nvJitLink - void* p_nvJitLinkDestroy "reinterpret_cast(cuda_core::rt::p_nvJitLinkDestroy)" - - -# Initialize driver function pointers from cydriver.__pyx_capi__ at module load -cdef void* _get_driver_fn(str name): - capsule = cydriver.__pyx_capi__[name] - return PyCapsule_GetPointer(capsule, PyCapsule_GetName(capsule)) - - -cdef void* _get_optional_driver_fn(str name): - try: - capsule = cydriver.__pyx_capi__[name] - except KeyError: - return NULL - return PyCapsule_GetPointer(capsule, PyCapsule_GetName(capsule)) - - -cdef void _init_driver_fn_pointers() noexcept: - global p_cuGetErrorName, p_cuGetErrorString - global p_cuDevicePrimaryCtxRetain, p_cuDevicePrimaryCtxRelease, p_cuCtxGetCurrent - global p_cuCtxSetCurrent, p_cuCtxSynchronize, p_cuCtxGetStreamPriorityRange - global p_cuCtxGetDevice, p_cuGraphNodeSetParams - global p_cuGreenCtxCreate, p_cuGreenCtxDestroy, p_cuCtxFromGreenCtx - global p_cuDevResourceGenerateDesc, p_cuGreenCtxStreamCreate - global p_cuStreamCreateWithPriority, p_cuStreamDestroy, p_cuStreamGetCtx - global p_cuEventCreate, p_cuEventDestroy, p_cuIpcOpenEventHandle - global p_cuDeviceGetCount - global p_cuMemPoolSetAccess, p_cuMemPoolDestroy, p_cuMemPoolCreate - global p_cuDeviceGetMemPool, p_cuMemPoolImportFromShareableHandle - global p_cuMemAllocFromPoolAsync, p_cuMemAllocAsync, p_cuMemAlloc, p_cuMemAllocHost - global p_cuMemFreeAsync, p_cuMemFree, p_cuMemFreeHost - global p_cuMemPoolImportPointer - global p_cuLibraryLoadFromFile, p_cuLibraryLoadData, p_cuLibraryUnload, p_cuLibraryGetKernel - global p_cuGraphDestroy, p_cuGraphInstantiateWithParams - global p_cuGraphExecUpdate, p_cuGraphExecDestroy - global p_cuUserObjectCreate, p_cuUserObjectRelease - global p_cuGraphRetainUserObject, p_cuGraphReleaseUserObject - global p_cuGraphNodeFindInClone, p_cuGraphChildGraphNodeGetGraph - global p_cuLinkDestroy - global p_cuGraphicsUnmapResources, p_cuGraphicsUnregisterResource - global p_cuArray3DCreate, p_cuArrayDestroy - global p_cuMipmappedArrayCreate, p_cuMipmappedArrayDestroy, p_cuMipmappedArrayGetLevel - global p_cuTexObjectCreate, p_cuTexObjectDestroy - global p_cuSurfObjectCreate, p_cuSurfObjectDestroy - - # Error formatting - p_cuGetErrorName = _get_driver_fn("cuGetErrorName") - p_cuGetErrorString = _get_driver_fn("cuGetErrorString") - - # Context - p_cuDevicePrimaryCtxRetain = _get_driver_fn("cuDevicePrimaryCtxRetain") - p_cuDevicePrimaryCtxRelease = _get_driver_fn("cuDevicePrimaryCtxRelease") - p_cuCtxGetCurrent = _get_driver_fn("cuCtxGetCurrent") - p_cuCtxSetCurrent = _get_driver_fn("cuCtxSetCurrent") - p_cuCtxSynchronize = _get_driver_fn("cuCtxSynchronize") - p_cuCtxGetStreamPriorityRange = _get_driver_fn("cuCtxGetStreamPriorityRange") - p_cuCtxGetDevice = _get_driver_fn("cuCtxGetDevice") - # Graph node parameter updates need CUDA 12.2+ (checked again at the call site). - p_cuGraphNodeSetParams = _get_optional_driver_fn("cuGraphNodeSetParams") - p_cuGreenCtxCreate = _get_optional_driver_fn("cuGreenCtxCreate") - p_cuGreenCtxDestroy = _get_optional_driver_fn("cuGreenCtxDestroy") - p_cuCtxFromGreenCtx = _get_optional_driver_fn("cuCtxFromGreenCtx") - p_cuDevResourceGenerateDesc = _get_optional_driver_fn("cuDevResourceGenerateDesc") - p_cuGreenCtxStreamCreate = _get_optional_driver_fn("cuGreenCtxStreamCreate") - - # Stream - p_cuStreamCreateWithPriority = _get_driver_fn("cuStreamCreateWithPriority") - p_cuStreamDestroy = _get_driver_fn("cuStreamDestroy") - p_cuStreamGetCtx = _get_driver_fn("cuStreamGetCtx") - - # Event - p_cuEventCreate = _get_driver_fn("cuEventCreate") - p_cuEventDestroy = _get_driver_fn("cuEventDestroy") - p_cuIpcOpenEventHandle = _get_driver_fn("cuIpcOpenEventHandle") - - # Device - p_cuDeviceGetCount = _get_driver_fn("cuDeviceGetCount") - - # Memory pool - p_cuMemPoolSetAccess = _get_driver_fn("cuMemPoolSetAccess") - p_cuMemPoolDestroy = _get_driver_fn("cuMemPoolDestroy") - p_cuMemPoolCreate = _get_driver_fn("cuMemPoolCreate") - p_cuDeviceGetMemPool = _get_driver_fn("cuDeviceGetMemPool") - p_cuMemPoolImportFromShareableHandle = _get_driver_fn("cuMemPoolImportFromShareableHandle") - - # Memory allocation - p_cuMemAllocFromPoolAsync = _get_driver_fn("cuMemAllocFromPoolAsync") - p_cuMemAllocAsync = _get_driver_fn("cuMemAllocAsync") - p_cuMemAlloc = _get_driver_fn("cuMemAlloc") - p_cuMemAllocHost = _get_driver_fn("cuMemAllocHost") - - # Memory deallocation - p_cuMemFreeAsync = _get_driver_fn("cuMemFreeAsync") - p_cuMemFree = _get_driver_fn("cuMemFree") - p_cuMemFreeHost = _get_driver_fn("cuMemFreeHost") - - # IPC - p_cuMemPoolImportPointer = _get_driver_fn("cuMemPoolImportPointer") - - # Library - p_cuLibraryLoadFromFile = _get_driver_fn("cuLibraryLoadFromFile") - p_cuLibraryLoadData = _get_driver_fn("cuLibraryLoadData") - p_cuLibraryUnload = _get_driver_fn("cuLibraryUnload") - p_cuLibraryGetKernel = _get_driver_fn("cuLibraryGetKernel") - - # Graph - p_cuGraphDestroy = _get_driver_fn("cuGraphDestroy") - p_cuGraphInstantiateWithParams = _get_driver_fn("cuGraphInstantiateWithParams") - p_cuGraphExecUpdate = _get_driver_fn("cuGraphExecUpdate") - p_cuGraphExecDestroy = _get_driver_fn("cuGraphExecDestroy") - p_cuUserObjectCreate = _get_driver_fn("cuUserObjectCreate") - p_cuUserObjectRelease = _get_driver_fn("cuUserObjectRelease") - p_cuGraphRetainUserObject = _get_driver_fn("cuGraphRetainUserObject") - p_cuGraphReleaseUserObject = _get_driver_fn("cuGraphReleaseUserObject") - p_cuGraphNodeFindInClone = _get_driver_fn("cuGraphNodeFindInClone") - p_cuGraphChildGraphNodeGetGraph = _get_driver_fn("cuGraphChildGraphNodeGetGraph") - - # Linker - p_cuLinkDestroy = _get_driver_fn("cuLinkDestroy") - - # Graphics interop - p_cuGraphicsUnmapResources = _get_driver_fn("cuGraphicsUnmapResources") - p_cuGraphicsUnregisterResource = _get_driver_fn("cuGraphicsUnregisterResource") - - # Texture / surface / array (PR #467) - p_cuArray3DCreate = _get_driver_fn("cuArray3DCreate") - p_cuArrayDestroy = _get_driver_fn("cuArrayDestroy") - p_cuMipmappedArrayCreate = _get_driver_fn("cuMipmappedArrayCreate") - p_cuMipmappedArrayDestroy = _get_driver_fn("cuMipmappedArrayDestroy") - p_cuMipmappedArrayGetLevel = _get_driver_fn("cuMipmappedArrayGetLevel") - p_cuTexObjectCreate = _get_driver_fn("cuTexObjectCreate") - p_cuTexObjectDestroy = _get_driver_fn("cuTexObjectDestroy") - p_cuSurfObjectCreate = _get_driver_fn("cuSurfObjectCreate") - p_cuSurfObjectDestroy = _get_driver_fn("cuSurfObjectDestroy") - - -_init_driver_fn_pointers() initialize_deferred_cleanup() @@ -569,51 +318,3 @@ def _attach_rollback_failure_for_testing(int status): """ _attach_rollback_failure_local( b"cuTestOperation", status, b"failed while testing") - -# ============================================================================= -# NVRTC function pointer initialization -# ============================================================================= - -cdef void* _get_nvrtc_fn(str name): - capsule = cynvrtc.__pyx_capi__[name] - return PyCapsule_GetPointer(capsule, PyCapsule_GetName(capsule)) - -cdef void _init_nvrtc_fn_pointers() noexcept: - global p_nvrtcDestroyProgram - p_nvrtcDestroyProgram = _get_nvrtc_fn("nvrtcDestroyProgram") - -_init_nvrtc_fn_pointers() - -# ============================================================================= -# NVVM function pointer initialization -# -# NVVM may not be available at runtime, so we handle missing function pointers -# gracefully. The C++ deleter checks for null before calling. -# ============================================================================= - -cdef void* _get_nvvm_fn(str name): - capsule = cynvvm.__pyx_capi__[name] - return PyCapsule_GetPointer(capsule, PyCapsule_GetName(capsule)) - -cdef void _init_nvvm_fn_pointers() noexcept: - global p_nvvmDestroyProgram - p_nvvmDestroyProgram = _get_nvvm_fn("nvvmDestroyProgram") - -_init_nvvm_fn_pointers() - -# ============================================================================= -# nvJitLink function pointer initialization -# -# nvJitLink may not be available at runtime, so we handle missing function -# pointers gracefully. The C++ deleter checks for null before calling. -# ============================================================================= - -cdef void* _get_nvjitlink_fn(str name): - capsule = cynvjitlink.__pyx_capi__[name] - return PyCapsule_GetPointer(capsule, PyCapsule_GetName(capsule)) - -cdef void _init_nvjitlink_fn_pointers() noexcept: - global p_nvJitLinkDestroy - p_nvJitLinkDestroy = _get_nvjitlink_fn("nvJitLinkDestroy") - -_init_nvjitlink_fn_pointers() diff --git a/cuda_core/cuda/core/_stream.pyx b/cuda_core/cuda/core/_stream.pyx index 4c51b32f4f0..2ed1e402be4 100644 --- a/cuda_core/cuda/core/_stream.pyx +++ b/cuda_core/cuda/core/_stream.pyx @@ -14,6 +14,7 @@ from cuda.core._utils.cuda_utils cimport ( check_or_create_options, HANDLE_RETURN, ) +from cuda.core._utils.version cimport cy_driver_version import cython import warnings @@ -171,7 +172,14 @@ cdef class Stream: prio = high # C++ creates the stream and returns owning handle with context dependency. - # For green contexts, the C++ layer auto-dispatches to cuGreenCtxStreamCreate. + # For green contexts, the C++ layer auto-dispatches to cuGreenCtxStreamCreate, + # a 12.5 driver API (cuGreenCtxCreate itself is 12.4); the driver alone + # decides availability, and the gate lives here, not in C++. + if context.is_green and cy_driver_version() < (12, 5, 0): + raise RuntimeError( + "Green context stream creation requires CUDA driver 12.5 or newer " + f"(current driver: {'.'.join(map(str, cy_driver_version()))})" + ) h_stream = create_stream_handle(h_context, flags, prio) if not h_stream: res_code = get_last_error() @@ -182,11 +190,6 @@ cdef class Stream: "Green context streams must be non-blocking. " "Use StreamOptions(nonblocking=True) or omit the option (True is the default)." ) - elif res_code == cydriver.CUresult.CUDA_ERROR_NOT_SUPPORTED: - raise RuntimeError( - "cuGreenCtxStreamCreate is not available. " - "Green context stream creation requires CUDA 12.5 or newer." - ) else: HANDLE_RETURN(res_code) cdef Stream self = Stream._from_handle(cls, h_stream) diff --git a/cuda_core/tests/test_rt_layout.py b/cuda_core/tests/test_rt_layout.py index d70d24ee374..228ddfd433a 100644 --- a/cuda_core/tests/test_rt_layout.py +++ b/cuda_core/tests/test_rt_layout.py @@ -112,6 +112,53 @@ def test_cuda_version_is_named_only_in_versions_hpp(): assert "CUDA_CORE_BUILD_MAJOR" in read(RT / "versions.hpp") +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_driver_function_table_matches_the_cuda_bindings_loader(): + """driver_api.hpp lists each driver function with the CUDA version cuda-bindings + requests it at; that number decides which functions every supported driver + must provide. Check it against the loader cuda-bindings generates.""" + loader = CORE.parents[2] / "cuda_bindings" / "cuda" / "bindings" / "_internal" / "driver_linux.pyx" + if not loader.is_file(): + pytest.skip("cuda-bindings source is not next to cuda_core") + entries = re.findall(r"^\s*X\((cu\w+), (\d+)\)", read(RT / "driver_api.hpp"), re.M) + assert len(entries) >= 60 + assert len({name for name, _ in entries}) == len(entries), "duplicate table entry" + requested = {} + for name, version in re.findall(r"cuGetProcAddress_v2\('(\w+)', &__\w+, (\d+)", read(loader)): + requested.setdefault(name, set()).add(int(version)) + mismatched = { + name: (int(introduced), sorted(requested.get(name, ()))) + for name, introduced in entries + if int(introduced) not in requested.get(name, ()) + } + assert mismatched == {} + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_driver_calls_go_through_the_table(): + """Every driver call uses DRIVER_CALL (or a pw_ wrapper), which resolves the + table on first use and never dereferences null. The only raw p_ calls are + the table's own machinery and the sites under ipc_import_mutex, where the + table is resolved before the lock and marked `// raw:`.""" + machinery = {"driver_api.hpp", "driver_api.cpp", "py_driver_fns.cpp", "internal.hpp"} + raw_call = re.compile(r"\bp_(cu|nv)\w+\(") + offenders = [] + for path in HEADERS + SOURCES: + if path.name in machinery: + continue + for number, line in enumerate(read(path).splitlines(), 1): + if raw_call.search(line) and "// raw:" not in line and not line.lstrip().startswith("//"): + offenders.append(f"{path.name}:{number}") + assert offenders == [] + marked = [ + f"{path.name}:{number}" + for path in SOURCES + for number, line in enumerate(read(path).splitlines(), 1) + if "// raw:" in line + ] + assert {name.split(":")[0] for name in marked} == {"memory.cpp"}, marked + + @pytest.mark.agent_authored(model="claude-fable-5-1") def test_pxd_functions_are_not_called_by_name_inside_the_module(): """Cython emits a static prototype for each cdef function the .pxd declares, so From 49553936e9840ad5ef3ebb6783acd5947370405b Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Fri, 18 Sep 2026 13:10:54 -0700 Subject: [PATCH 04/18] cuda.core: gate features on the driver alone now that cuda-bindings has a floor (#2783) Every cuda-bindings cuda.core accepts (12.9.8+, 13.4.1+) has the full API surface of its CUDA major, so a run-time check of the cuda-bindings version says nothing a build-major fence does not. Double checks (driver and bindings) become driver-only checks; checks of the bindings minor become `IF CUDA_CORE_BUILD_MAJOR` fences in Cython or a comparison with the new `cuda.core._utils.version.BUILD_CUDA_MAJOR` in Python; checks that every accepted cuda-bindings satisfies are deleted, along with the fallbacks they guarded. - Kernel argument info, green contexts, workqueues, graph node updates, conditional graph nodes: driver-only gates. - cuGraphNodeGetParams (13.2): a driver gate on the CUDA 13 build, and the call goes through cydriver instead of the Python driver layer. - Managed memory NUMA locations, virtual memory MANAGED type: build fence (plus the 13.0 driver where a driver entry point is involved). - Copy attribute enums (CUDA 12.8): unconditional. - NVVM, nvJitLink, NVRTC PCH bindings: always present; only the library load is probed. - Checkpoint: the CUDA 13 build only (CUcheckpointGpuPair is a CUDA 13 type; the CUDA 12 bindings never had it), stated plainly. - cuda.core.system: NVML through cuda.bindings unconditionally; the driver/runtime fallbacks are gone and CUDA_BINDINGS_NVML_IS_COMPATIBLE is a deprecated constant True. system.typing exports DeviceArch and FieldId unconditionally. - Error-enum explanations come from cuda-bindings' enum docstrings; the frozen 13.1.1 tables and their loader are removed. - cuda.bindings.utils.warn_if_cuda_major_version_mismatch (13.3+) is called on the CUDA 13 build only, instead of try/except ImportError. cy_binding_version() has no callers left and is removed; binding_version() stays. Tests follow the same rule set. --- cuda_core/cuda/core/_device.pyx | 13 +- cuda_core/cuda/core/_device_resources.pyx | 18 +- cuda_core/cuda/core/_linker.pyx | 37 +- cuda_core/cuda/core/_memory/_copy_enums.py | 53 +- .../cuda/core/_memory/_managed_buffer.py | 16 +- .../cuda/core/_memory/_managed_location.py | 11 +- .../cuda/core/_memory/_managed_memory_ops.pyx | 2 +- .../core/_memory/_virtual_memory_resource.py | 6 +- cuda_core/cuda/core/_module.pyx | 7 +- cuda_core/cuda/core/_program.pyx | 22 +- .../_utils/driver_cu_result_explanations.py | 11 +- .../driver_cu_result_explanations_frozen.py | 357 ------------ .../core/_utils/enum_explanations_helpers.py | 71 +-- .../_utils/runtime_cuda_error_explanations.py | 11 +- .../runtime_cuda_error_explanations_frozen.py | 538 ------------------ cuda_core/cuda/core/_utils/version.pxd | 1 - cuda_core/cuda/core/_utils/version.pyi | 2 + cuda_core/cuda/core/_utils/version.pyx | 15 +- cuda_core/cuda/core/checkpoint.py | 35 +- cuda_core/cuda/core/graph/_graph_builder.pyx | 17 +- cuda_core/cuda/core/graph/_subclasses.pxd | 3 + cuda_core/cuda/core/graph/_subclasses.pyx | 151 ++--- cuda_core/cuda/core/system/__init__.py | 33 +- cuda_core/cuda/core/system/_nvlink.pxi | 5 +- cuda_core/cuda/core/system/_system.pyx | 55 +- cuda_core/cuda/core/system/typing.py | 57 +- cuda_core/docs/source/release/1.3.0-notes.rst | 12 + cuda_core/tests/graph/test_graph_builder.py | 15 +- .../tests/graph/test_graph_definition.py | 10 +- .../tests/memory/test_copy_batch_options.py | 4 +- .../tests/memory/test_copy_single_options.py | 8 +- cuda_core/tests/memory/test_managed_ops.py | 39 +- cuda_core/tests/system/test_system_device.py | 11 +- cuda_core/tests/system/test_system_events.py | 6 +- cuda_core/tests/system/test_system_system.py | 7 - cuda_core/tests/test_checkpoint.py | 11 +- cuda_core/tests/test_cuda_utils.py | 11 - cuda_core/tests/test_device.py | 6 - cuda_core/tests/test_enum_coverage.py | 340 ++++++----- cuda_core/tests/test_error_handling.py | 4 +- cuda_core/tests/test_green_context.py | 8 +- cuda_core/tests/test_launcher.py | 3 - cuda_core/tests/test_linker.py | 12 - cuda_core/tests/test_module.py | 4 +- .../tests/test_optional_dependency_imports.py | 87 +-- cuda_core/tests/test_program.py | 14 +- .../test_utils_enum_explanations_helpers.py | 83 --- 47 files changed, 452 insertions(+), 1790 deletions(-) delete mode 100644 cuda_core/cuda/core/_utils/driver_cu_result_explanations_frozen.py delete mode 100644 cuda_core/cuda/core/_utils/runtime_cuda_error_explanations_frozen.py diff --git a/cuda_core/cuda/core/_device.pyx b/cuda_core/cuda/core/_device.pyx index 40b09d6c761..f7901081c98 100644 --- a/cuda_core/cuda/core/_device.pyx +++ b/cuda_core/cuda/core/_device.pyx @@ -1057,13 +1057,6 @@ class Device: cuda.core.system.Device The corresponding system-level device instance used for NVML access. """ - from cuda.core.system._system import CUDA_BINDINGS_NVML_IS_COMPATIBLE - - if not CUDA_BINDINGS_NVML_IS_COMPATIBLE: - raise RuntimeError( - "cuda.core.system.Device requires cuda-bindings 12.9.6+ for CUDA 12.x, or cuda-bindings 13.2.0+ for CUDA 13.x" - ) - from cuda.core.system import Device as SystemDevice return SystemDevice(uuid=self.uuid) @@ -1647,11 +1640,9 @@ cdef inline int Device_ensure_cuda_initialized() except? -1: with _lock, nogil: HANDLE_RETURN(cydriver.cuInit(0)) _is_cuInit = True - try: + IF CUDA_CORE_BUILD_MAJOR >= 13: + # Added in cuda-bindings 13.3; absent from the 12.x line. from cuda.bindings.utils import warn_if_cuda_major_version_mismatch - except ImportError: - pass - else: warn_if_cuda_major_version_mismatch() return 0 diff --git a/cuda_core/cuda/core/_device_resources.pyx b/cuda_core/cuda/core/_device_resources.pyx index 1d697a0c6d6..0f6e94bf6b3 100644 --- a/cuda_core/cuda/core/_device_resources.pyx +++ b/cuda_core/cuda/core/_device_resources.pyx @@ -20,7 +20,7 @@ from cuda.bindings cimport cydriver from cuda.core._rt cimport ContextHandle, GreenCtxHandle, as_cu, get_context_green_ctx from cuda.core._utils.cuda_utils cimport check_or_create_options, HANDLE_RETURN from cuda.core._utils.cuda_utils import is_sequence -from cuda.core._utils.version cimport cy_binding_version, cy_driver_version +from cuda.core._utils.version cimport cy_driver_version from cuda.core._utils.validators import check_str_enum @@ -47,7 +47,6 @@ cdef inline int _check_green_ctx_support() except?-1: if _green_ctx_checked == -1: raise RuntimeError(_green_ctx_err_msg) cdef tuple drv = cy_driver_version() - cdef tuple bind = cy_binding_version() if drv < (12, 4, 0): _green_ctx_err_msg = ( "Green context support requires CUDA driver 12.4 or newer " @@ -55,13 +54,6 @@ cdef inline int _check_green_ctx_support() except?-1: ) _green_ctx_checked = -1 raise RuntimeError(_green_ctx_err_msg) - if bind < (12, 4, 0): - _green_ctx_err_msg = ( - "Green context support requires cuda.bindings 12.4 or newer " - f"(current bindings: {'.'.join(map(str, bind))})" - ) - _green_ctx_checked = -1 - raise RuntimeError(_green_ctx_err_msg) _green_ctx_checked = 1 return 0 @@ -73,7 +65,6 @@ cdef inline int _check_workqueue_support() except?-1: if _workqueue_checked == -1: raise RuntimeError(_workqueue_err_msg) cdef tuple drv = cy_driver_version() - cdef tuple bind = cy_binding_version() if drv < (13, 1, 0): _workqueue_err_msg = ( "WorkqueueResource requires CUDA driver 13.1 or newer " @@ -81,13 +72,6 @@ cdef inline int _check_workqueue_support() except?-1: ) _workqueue_checked = -1 raise RuntimeError(_workqueue_err_msg) - if bind < (13, 1, 0): - _workqueue_err_msg = ( - "WorkqueueResource requires cuda.bindings 13.1 or newer " - f"(current bindings: {'.'.join(map(str, bind))})" - ) - _workqueue_checked = -1 - raise RuntimeError(_workqueue_err_msg) _workqueue_checked = 1 return 0 diff --git a/cuda_core/cuda/core/_linker.pyx b/cuda_core/cuda/core/_linker.pyx index 3e68d24afa3..cdd4373d478 100644 --- a/cuda_core/cuda/core/_linker.pyx +++ b/cuda_core/cuda/core/_linker.pyx @@ -30,7 +30,6 @@ from typing import TYPE_CHECKING, Union from warnings import warn from cuda.pathfinder import DynamicLibNotFoundError -from cuda.pathfinder._optional_cuda_import import _optional_cuda_import from cuda.core._device import Device from cuda.core._module import ObjectCode from cuda.core._utils.clear_error_support import assert_type @@ -728,26 +727,24 @@ def _decide_nvjitlink_or_driver() -> bool: " For best results, consider upgrading to a recent version of" ) - nvjitlink_module = _optional_cuda_import("cuda.bindings.nvjitlink") - if nvjitlink_module is None: - warn_txt = f"cuda.bindings.nvjitlink is not available, therefore {warn_txt_common} cuda-bindings." - else: - from cuda.bindings._internal import nvjitlink + # cuda.bindings.nvjitlink is present in every cuda-bindings cuda.core accepts; + # only the nvJitLink library itself can be missing or too old. + from cuda.bindings._internal import nvjitlink - try: - has_version_symbol = _nvjitlink_has_version_symbol(nvjitlink) - except DynamicLibNotFoundError: - warn_txt = ( - f"cuda.bindings.nvjitlink is not available, therefore {warn_txt_common} cuda-bindings." - ) - else: - if has_version_symbol: - _use_nvjitlink_backend = True - return False # Use nvjitlink - warn_txt = ( - f"{'nvJitLink*.dll' if sys.platform == 'win32' else 'libnvJitLink.so*'} is too old (<12.3)." - f" Therefore cuda.bindings.nvjitlink is not usable and {warn_txt_common} nvJitLink." - ) + try: + has_version_symbol = _nvjitlink_has_version_symbol(nvjitlink) + except DynamicLibNotFoundError: + warn_txt = ( + f"cuda.bindings.nvjitlink is not available, therefore {warn_txt_common} cuda-bindings." + ) + else: + if has_version_symbol: + _use_nvjitlink_backend = True + return False # Use nvjitlink + warn_txt = ( + f"{'nvJitLink*.dll' if sys.platform == 'win32' else 'libnvJitLink.so*'} is too old (<12.3)." + f" Therefore cuda.bindings.nvjitlink is not usable and {warn_txt_common} nvJitLink." + ) warn(warn_txt, stacklevel=2, category=RuntimeWarning) _driver = driver diff --git a/cuda_core/cuda/core/_memory/_copy_enums.py b/cuda_core/cuda/core/_memory/_copy_enums.py index 84c72e71110..8b53708ee26 100644 --- a/cuda_core/cuda/core/_memory/_copy_enums.py +++ b/cuda_core/cuda/core/_memory/_copy_enums.py @@ -11,7 +11,6 @@ from cuda.core._host import Host from cuda.core._utils.cuda_utils import driver from cuda.core._utils.pycompat import StrEnum -from cuda.core._utils.version import binding_version __all__ = ["CopyOptions", "MemcpyOverlapMode", "MemcpySrcAccessOrder"] @@ -117,47 +116,31 @@ def __post_init__(self): def _to_driver_enum(self) -> int: """Return the driver CUmemcpySrcAccessOrder value.""" - if not _SRC_ACCESS_ORDER_TO_DRIVER: - raise NotImplementedError(_CUDA13_REQUIRED) return _SRC_ACCESS_ORDER_TO_DRIVER[MemcpySrcAccessOrder(self.src_access_order)] def _to_driver_flags(self) -> int: """Return the driver CUmemcpyFlags value.""" - if not _OVERLAP_MODE_TO_DRIVER: - raise NotImplementedError(_CUDA13_REQUIRED) return _OVERLAP_MODE_TO_DRIVER[MemcpyOverlapMode(self.overlap_mode)] -_CUDA13_REQUIRED = "copy attributes require cuda.bindings 13.0 or newer" - -# CUmemcpySrcAccessOrder and CUmemcpyFlags are exposed by cuda.bindings 13.0+, -# so these maps are empty when it is older. Nothing reaches them there: -# copy_batch refuses non-default CopyOptions when the batched entry point is -# unavailable. -# -# Keyed by ``str``: under ``python_version = "3.10"`` mypy resolves StrEnum to -# the unstubbed backports shim and so infers the members as plain ``str``. -# StrEnum members are ``str`` instances, so this holds on every version. The -# values are wrapped in ``int()`` because the driver enums are untyped. -_SRC_ACCESS_ORDER_TO_DRIVER: dict[str, int] -_OVERLAP_MODE_TO_DRIVER: dict[str, int] - -if binding_version() >= (13, 0, 0): - _src_order = driver.CUmemcpySrcAccessOrder - _flags = driver.CUmemcpyFlags - _SRC_ACCESS_ORDER_TO_DRIVER = { - MemcpySrcAccessOrder.STREAM: int(_src_order.CU_MEMCPY_SRC_ACCESS_ORDER_STREAM), - MemcpySrcAccessOrder.DURING_API_CALL: int(_src_order.CU_MEMCPY_SRC_ACCESS_ORDER_DURING_API_CALL), - MemcpySrcAccessOrder.ANY: int(_src_order.CU_MEMCPY_SRC_ACCESS_ORDER_ANY), - } - _OVERLAP_MODE_TO_DRIVER = { - MemcpyOverlapMode.DEFAULT: int(_flags.CU_MEMCPY_FLAG_DEFAULT), - MemcpyOverlapMode.PREFER_OVERLAP_WITH_COMPUTE: int(_flags.CU_MEMCPY_FLAG_PREFER_OVERLAP_WITH_COMPUTE), - } - del _src_order, _flags -else: - _SRC_ACCESS_ORDER_TO_DRIVER = {} - _OVERLAP_MODE_TO_DRIVER = {} +# CUmemcpySrcAccessOrder and CUmemcpyFlags were added in CUDA 12.8; every +# cuda-bindings cuda.core accepts has them. Keyed by ``str``: under +# ``python_version = "3.10"`` mypy resolves StrEnum to the unstubbed backports +# shim and so infers the members as plain ``str``. StrEnum members are ``str`` +# instances, so this holds on every version. The values are wrapped in +# ``int()`` because the driver enums are untyped. +_src_order = driver.CUmemcpySrcAccessOrder +_flags = driver.CUmemcpyFlags +_SRC_ACCESS_ORDER_TO_DRIVER: dict[str, int] = { + MemcpySrcAccessOrder.STREAM: int(_src_order.CU_MEMCPY_SRC_ACCESS_ORDER_STREAM), + MemcpySrcAccessOrder.DURING_API_CALL: int(_src_order.CU_MEMCPY_SRC_ACCESS_ORDER_DURING_API_CALL), + MemcpySrcAccessOrder.ANY: int(_src_order.CU_MEMCPY_SRC_ACCESS_ORDER_ANY), +} +_OVERLAP_MODE_TO_DRIVER: dict[str, int] = { + MemcpyOverlapMode.DEFAULT: int(_flags.CU_MEMCPY_FLAG_DEFAULT), + MemcpyOverlapMode.PREFER_OVERLAP_WITH_COMPUTE: int(_flags.CU_MEMCPY_FLAG_PREFER_OVERLAP_WITH_COMPUTE), +} +del _src_order, _flags def _reject_unsupported_during_api_call( diff --git a/cuda_core/cuda/core/_memory/_managed_buffer.py b/cuda_core/cuda/core/_memory/_managed_buffer.py index 44b84ff241e..4353fb7975d 100644 --- a/cuda_core/cuda/core/_memory/_managed_buffer.py +++ b/cuda_core/cuda/core/_memory/_managed_buffer.py @@ -19,7 +19,7 @@ _read_preferred_location_v2, ) from cuda.core._utils.cuda_utils import driver, handle_return -from cuda.core._utils.version import binding_version, driver_version +from cuda.core._utils.version import BUILD_CUDA_MAJOR, driver_version if TYPE_CHECKING: from cuda.core._memory._buffer import MemoryResource @@ -215,13 +215,13 @@ def preferred_location(self) -> Device | Host | None: as ``Host()``. """ # The v2 path uses CU_MEM_RANGE_ATTRIBUTE_PREFERRED_LOCATION_{TYPE,ID}, - # both added in CUDA 13. Require both bindings and the runtime driver - # to be 13.0+; otherwise fall back to the legacy device-ordinal path. - # See PR #2054 / #2064 for prior bindings-only-check regressions. - if binding_version() >= (13, 0, 0) and driver_version() >= (13, 0, 0): + # both added in CUDA 13: it exists in the CUDA 13 build only, and the + # runtime driver must be 13.0+ too; otherwise fall back to the legacy + # device-ordinal path. See PR #2054 / #2064 for prior regressions. + if BUILD_CUDA_MAJOR >= 13 and driver_version() >= (13, 0, 0): return _read_preferred_location_v2(self) - # CUDA 12 legacy path (no NUMA info available; also taken when - # bindings are 13.x but the runtime driver is still 12.x). + # CUDA 12 legacy path (no NUMA info available; also taken by a CUDA 13 + # build when the runtime driver is still 12.x). loc_id = _get_int_attr(self, _ATTR_PREFERRED) if loc_id == -2: return None @@ -249,7 +249,7 @@ def last_prefetch_location(self) -> Device | Host | None: the legacy attribute carries only a device ordinal (or ``-1`` for host), so host NUMA details are unavailable. """ - if binding_version() >= (13, 0, 0) and driver_version() >= (13, 0, 0): + if BUILD_CUDA_MAJOR >= 13 and driver_version() >= (13, 0, 0): return _read_last_prefetch_location_v2(self) loc_id = _get_int_attr(self, _ATTR_LAST_PREFETCH) if loc_id == -2: diff --git a/cuda_core/cuda/core/_memory/_managed_location.py b/cuda_core/cuda/core/_memory/_managed_location.py index 336f09f2eee..e77f242de7f 100644 --- a/cuda_core/cuda/core/_memory/_managed_location.py +++ b/cuda_core/cuda/core/_memory/_managed_location.py @@ -6,7 +6,7 @@ from dataclasses import dataclass from typing import TYPE_CHECKING, Literal -from cuda.core._utils.version import binding_version, driver_version +from cuda.core._utils.version import BUILD_CUDA_MAJOR, driver_version if TYPE_CHECKING: from cuda.core._device import Device @@ -39,13 +39,14 @@ def _reject_numa_host_on_cuda12(spec: _LocSpec) -> None: ``TypeError`` at the call boundary with actionable wording. """ # The host-NUMA kinds map to CU_MEM_LOCATION_TYPE_HOST_NUMA{,_CURRENT}, - # both added in CUDA 13. Require both bindings and the runtime driver to - # be 13.0+; bindings-only is insufficient (PR #2054 / #2064 precedent). - if binding_version() >= (13, 0, 0) and driver_version() >= (13, 0, 0): + # both added in CUDA 13: the CUDA 13 build passes them to the v2 driver + # entry points, which need a 13.0+ runtime driver as well (a build check + # alone is insufficient; PR #2054 / #2064 precedent). + if BUILD_CUDA_MAJOR >= 13 and driver_version() >= (13, 0, 0): return if spec.kind in ("host_numa", "host_numa_current"): raise TypeError( - "Host(numa_id=...) / Host.numa_current() require both cuda-bindings 13.0+ " + "Host(numa_id=...) / Host.numa_current() require the CUDA 13 build of cuda.core " "and a CUDA 13+ runtime driver; use Host() instead" ) diff --git a/cuda_core/cuda/core/_memory/_managed_memory_ops.pyx b/cuda_core/cuda/core/_memory/_managed_memory_ops.pyx index 61caa59ce1f..77cf423fab5 100644 --- a/cuda_core/cuda/core/_memory/_managed_memory_ops.pyx +++ b/cuda_core/cuda/core/_memory/_managed_memory_ops.pyx @@ -91,7 +91,7 @@ IF CUDA_CORE_BUILD_MAJOR < 13: if kind == "host": return -1 raise RuntimeError( - "Host(numa_id=...) / Host.numa_current() require both cuda-bindings 13.0+ " + "Host(numa_id=...) / Host.numa_current() require the CUDA 13 build of cuda.core " "and a CUDA 13+ runtime driver; use Host() instead" ) diff --git a/cuda_core/cuda/core/_memory/_virtual_memory_resource.py b/cuda_core/cuda/core/_memory/_virtual_memory_resource.py index ea1e2455c6f..bdc57595f53 100644 --- a/cuda_core/cuda/core/_memory/_virtual_memory_resource.py +++ b/cuda_core/cuda/core/_memory/_virtual_memory_resource.py @@ -21,7 +21,7 @@ from cuda.core._utils.cuda_utils import ( _check_driver_error as raise_if_driver_error, ) -from cuda.core._utils.version import binding_version +from cuda.core._utils.version import BUILD_CUDA_MAJOR from cuda.core.typing import ( DevicePointerType, VirtualMemoryAccessType, @@ -112,9 +112,9 @@ class VirtualMemoryResourceOptions: VirtualMemoryLocationType.HOST_NUMA_CURRENT: _l.CU_MEM_LOCATION_TYPE_HOST_NUMA_CURRENT, } _t = driver.CUmemAllocationType - # CUDA 13+ exposes MANAGED in CUmemAllocationType; older 12.x does not + # CUDA 13 added MANAGED to CUmemAllocationType; the CUDA 12 build has no such member. _allocation_type = {VirtualMemoryAllocationType.PINNED: _t.CU_MEM_ALLOCATION_TYPE_PINNED} # noqa: RUF012 - if binding_version() >= (13, 0, 0): + if BUILD_CUDA_MAJOR >= 13: _allocation_type[VirtualMemoryAllocationType.MANAGED] = _t.CU_MEM_ALLOCATION_TYPE_MANAGED @staticmethod diff --git a/cuda_core/cuda/core/_module.pyx b/cuda_core/cuda/core/_module.pyx index fbaaef9e1e8..0264ebad7e9 100644 --- a/cuda_core/cuda/core/_module.pyx +++ b/cuda_core/cuda/core/_module.pyx @@ -35,7 +35,7 @@ from cuda.core._utils.clear_error_support import ( raise_code_path_meant_to_be_unreachable, ) from cuda.core._utils.cuda_utils cimport HANDLE_RETURN -from cuda.core._utils.version cimport cy_binding_version, cy_driver_version +from cuda.core._utils.version cimport cy_driver_version from cuda.core._utils.cuda_utils import driver from cuda.bindings cimport cydriver @@ -472,11 +472,6 @@ cdef class Kernel: "Driver version 12.4 or newer is required for this function. " f"Using driver version {'.'.join(map(str, cy_driver_version()))}" ) - if cy_binding_version() < (12, 4, 0): - raise NotImplementedError( - "cuda.bindings 12.4 or newer is required for this function. " - f"Using binding version {'.'.join(map(str, cy_binding_version()))}" - ) cdef size_t arg_pos = 0 cdef list param_info_data = [] cdef cydriver.CUkernel cu_kernel = as_cu(self._h_kernel) diff --git a/cuda_core/cuda/core/_program.pyx b/cuda_core/cuda/core/_program.pyx index fd4a31ba1e8..f28f63ff067 100644 --- a/cuda_core/cuda/core/_program.pyx +++ b/cuda_core/cuda/core/_program.pyx @@ -46,7 +46,7 @@ from cuda.core._utils.cuda_utils import ( is_nested_sequence, is_sequence, ) -from cuda.core._utils.version import binding_version, driver_version +from cuda.core._utils.version import driver_version from cuda.core.utils._cache_dir import _default_cache_dir from cuda.core.typing import ObjectCodeFormatType, CompilerBackendType, PCHStatusType, SourceCodeType @@ -758,13 +758,8 @@ def _get_nvvm_module() -> object: raise RuntimeError("NVVM module is not available (previous import attempt failed)") try: - version = binding_version() - if version < (12, 9, 0): - raise RuntimeError( - f"NVVM bindings require cuda-bindings >= 12.9.0, but found {'.'.join(map(str, version))}. " - "Please update cuda-bindings to use NVVM features." - ) - + # cuda.bindings.nvvm is present in every cuda-bindings cuda.core accepts; + # the probe checks that libnvvm itself can be loaded. nvvm = _optional_cuda_import( "cuda.bindings.nvvm", probe_function=lambda module: module.version(), # probe triggers libnvvm load @@ -1046,15 +1041,6 @@ cdef object _nvrtc_compile_and_extract( return ObjectCode._init(bytes(data), target_type, symbol_mapping=symbol_mapping, name=name) -cdef int _nvrtc_pch_apis_cached = -1 # -1 = unchecked - -cdef bint _has_nvrtc_pch_apis(): - global _nvrtc_pch_apis_cached - if _nvrtc_pch_apis_cached < 0: - _nvrtc_pch_apis_cached = hasattr(nvrtc, "nvrtcGetPCHCreateStatus") - return _nvrtc_pch_apis_cached - - cdef object _read_pch_status(cynvrtc.nvrtcProgram prog): """Query nvrtcGetPCHCreateStatus and translate to a high-level string.""" cdef cynvrtc.nvrtcResult err @@ -1079,7 +1065,7 @@ cdef object Program_compile_nvrtc(Program self, str target_type, object name_exp ) cdef bint pch_creation_possible = self._options.create_pch or self._options.pch - if not pch_creation_possible or not _has_nvrtc_pch_apis(): + if not pch_creation_possible: self._pch_status = None return result diff --git a/cuda_core/cuda/core/_utils/driver_cu_result_explanations.py b/cuda_core/cuda/core/_utils/driver_cu_result_explanations.py index d1e0a53bb00..a8c9f0e9b69 100644 --- a/cuda_core/cuda/core/_utils/driver_cu_result_explanations.py +++ b/cuda_core/cuda/core/_utils/driver_cu_result_explanations.py @@ -4,13 +4,6 @@ from __future__ import annotations from cuda.bindings import driver -from cuda.core._utils.enum_explanations_helpers import get_best_available_explanations +from cuda.core._utils.enum_explanations_helpers import DocstringBackedExplanations - -def _load_fallback_explanations() -> dict[int, str | tuple[str, ...]]: - from cuda.core._utils.driver_cu_result_explanations_frozen import _FALLBACK_EXPLANATIONS - - return _FALLBACK_EXPLANATIONS # type: ignore[return-value] - - -DRIVER_CU_RESULT_EXPLANATIONS = get_best_available_explanations(driver.CUresult, _load_fallback_explanations) +DRIVER_CU_RESULT_EXPLANATIONS = DocstringBackedExplanations(driver.CUresult) diff --git a/cuda_core/cuda/core/_utils/driver_cu_result_explanations_frozen.py b/cuda_core/cuda/core/_utils/driver_cu_result_explanations_frozen.py deleted file mode 100644 index 41eb158f4e2..00000000000 --- a/cuda_core/cuda/core/_utils/driver_cu_result_explanations_frozen.py +++ /dev/null @@ -1,357 +0,0 @@ -# SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 - -# Like the runtime counterpart, this fallback is a deliberately frozen -# compatibility snapshot, not a release-maintained mirror of CUDA's enums. -# Do not update it past CUDA Toolkit v13.1.1. Bindings releases new enough to -# define later codes provide explanations through enum-member docstrings; if an -# older binding receives one from a newer driver, it falls through to -# cuGetErrorString(). Synchronizing this table with later Toolkit releases would -# restore the duplicate maintenance burden removed by PR #1860. -# CUDA Toolkit v13.1.1 -_FALLBACK_EXPLANATIONS = { - 0: ( - "The API call returned with no errors. In the case of query calls, this" - " also means that the operation being queried is complete (see" - " ::cuEventQuery() and ::cuStreamQuery())." - ), - 1: ( - "This indicates that one or more of the parameters passed to the API call" - " is not within an acceptable range of values." - ), - 2: ( - "The API call failed because it was unable to allocate enough memory or" - " other resources to perform the requested operation." - ), - 3: ( - "This indicates that the CUDA driver has not been initialized with" - " ::cuInit() or that initialization has failed." - ), - 4: "This indicates that the CUDA driver is in the process of shutting down.", - 5: ( - "This indicates profiler is not initialized for this run. This can" - " happen when the application is running with external profiling tools" - " like visual profiler." - ), - 6: ( - "This error return is deprecated as of CUDA 5.0. It is no longer an error" - " to attempt to enable/disable the profiling via ::cuProfilerStart or" - " ::cuProfilerStop without initialization." - ), - 7: ( - "This error return is deprecated as of CUDA 5.0. It is no longer an error" - " to call cuProfilerStart() when profiling is already enabled." - ), - 8: ( - "This error return is deprecated as of CUDA 5.0. It is no longer an error" - " to call cuProfilerStop() when profiling is already disabled." - ), - 34: ( - "This indicates that the CUDA driver that the application has loaded is a" - " stub library. Applications that run with the stub rather than a real" - " driver loaded will result in CUDA API returning this error." - ), - 36: ( - "This indicates that the API call requires a newer CUDA driver than the one" - " currently installed. Users should install an updated NVIDIA CUDA driver" - " to allow the API call to succeed." - ), - 46: ( - "This indicates that requested CUDA device is unavailable at the current" - " time. Devices are often unavailable due to use of" - " ::CU_COMPUTEMODE_EXCLUSIVE_PROCESS or ::CU_COMPUTEMODE_PROHIBITED." - ), - 100: ("This indicates that no CUDA-capable devices were detected by the installed CUDA driver."), - 101: ( - "This indicates that the device ordinal supplied by the user does not" - " correspond to a valid CUDA device or that the action requested is" - " invalid for the specified device." - ), - 102: "This error indicates that the Grid license is not applied.", - 200: ("This indicates that the device kernel image is invalid. This can also indicate an invalid CUDA module."), - 201: ( - "This most frequently indicates that there is no context bound to the" - " current thread. This can also be returned if the context passed to an" - " API call is not a valid handle (such as a context that has had" - " ::cuCtxDestroy() invoked on it). This can also be returned if a user" - " mixes different API versions (i.e. 3010 context with 3020 API calls)." - " See ::cuCtxGetApiVersion() for more details." - " This can also be returned if the green context passed to an API call" - " was not converted to a ::CUcontext using ::cuCtxFromGreenCtx API." - ), - 202: ( - "This indicated that the context being supplied as a parameter to the" - " API call was already the active context." - " This error return is deprecated as of CUDA 3.2. It is no longer an" - " error to attempt to push the active context via ::cuCtxPushCurrent()." - ), - 205: "This indicates that a map or register operation has failed.", - 206: "This indicates that an unmap or unregister operation has failed.", - 207: ("This indicates that the specified array is currently mapped and thus cannot be destroyed."), - 208: "This indicates that the resource is already mapped.", - 209: ( - "This indicates that there is no kernel image available that is suitable" - " for the device. This can occur when a user specifies code generation" - " options for a particular CUDA source file that do not include the" - " corresponding device configuration." - ), - 210: "This indicates that a resource has already been acquired.", - 211: "This indicates that a resource is not mapped.", - 212: ("This indicates that a mapped resource is not available for access as an array."), - 213: ("This indicates that a mapped resource is not available for access as a pointer."), - 214: ("This indicates that an uncorrectable ECC error was detected during execution."), - 215: ("This indicates that the ::CUlimit passed to the API call is not supported by the active device."), - 216: ( - "This indicates that the ::CUcontext passed to the API call can" - " only be bound to a single CPU thread at a time but is already" - " bound to a CPU thread." - ), - 217: ("This indicates that peer access is not supported across the given devices."), - 218: "This indicates that a PTX JIT compilation failed.", - 219: "This indicates an error with OpenGL or DirectX context.", - 220: ("This indicates that an uncorrectable NVLink error was detected during the execution."), - 221: "This indicates that the PTX JIT compiler library was not found.", - 222: "This indicates that the provided PTX was compiled with an unsupported toolchain.", - 223: "This indicates that the PTX JIT compilation was disabled.", - 224: ("This indicates that the ::CUexecAffinityType passed to the API call is not supported by the active device."), - 225: ( - "This indicates that the code to be compiled by the PTX JIT contains unsupported call to cudaDeviceSynchronize." - ), - 226: ( - "This indicates that an exception occurred on the device that is now" - " contained by the GPU's error containment capability. Common causes are -" - " a. Certain types of invalid accesses of peer GPU memory over nvlink" - " b. Certain classes of hardware errors" - " This leaves the process in an inconsistent state and any further CUDA" - " work will return the same error. To continue using CUDA, the process must" - " be terminated and relaunched." - ), - 300: ( - "This indicates that the device kernel source is invalid. This includes" - " compilation/linker errors encountered in device code or user error." - ), - 301: "This indicates that the file specified was not found.", - 302: "This indicates that a link to a shared object failed to resolve.", - 303: "This indicates that initialization of a shared object failed.", - 304: "This indicates that an OS call failed.", - 400: ( - "This indicates that a resource handle passed to the API call was not" - " valid. Resource handles are opaque types like ::CUstream and ::CUevent." - ), - 401: ( - "This indicates that a resource required by the API call is not in a" - " valid state to perform the requested operation." - ), - 402: ( - "This indicates an attempt was made to introspect an object in a way that" - " would discard semantically important information. This is either due to" - " the object using funtionality newer than the API version used to" - " introspect it or omission of optional return arguments." - ), - 500: ( - "This indicates that a named symbol was not found. Examples of symbols" - " are global/constant variable names, driver function names, texture names," - " and surface names." - ), - 600: ( - "This indicates that asynchronous operations issued previously have not" - " completed yet. This result is not actually an error, but must be indicated" - " differently than ::CUDA_SUCCESS (which indicates completion). Calls that" - " may return this value include ::cuEventQuery() and ::cuStreamQuery()." - ), - 700: ( - "While executing a kernel, the device encountered a" - " load or store instruction on an invalid memory address." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 701: ( - "This indicates that a launch did not occur because it did not have" - " appropriate resources. This error usually indicates that the user has" - " attempted to pass too many arguments to the device kernel, or the" - " kernel launch specifies too many threads for the kernel's register" - " count. Passing arguments of the wrong size (i.e. a 64-bit pointer" - " when a 32-bit int is expected) is equivalent to passing too many" - " arguments and can also result in this error." - ), - 702: ( - "This indicates that the device kernel took too long to execute. This can" - " only occur if timeouts are enabled - see the device attribute" - " ::CU_DEVICE_ATTRIBUTE_KERNEL_EXEC_TIMEOUT for more information." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 703: ("This error indicates a kernel launch that uses an incompatible texturing mode."), - 704: ( - "This error indicates that a call to ::cuCtxEnablePeerAccess() is" - " trying to re-enable peer access to a context which has already" - " had peer access to it enabled." - ), - 705: ( - "This error indicates that ::cuCtxDisablePeerAccess() is" - " trying to disable peer access which has not been enabled yet" - " via ::cuCtxEnablePeerAccess()." - ), - 708: ("This error indicates that the primary context for the specified device has already been initialized."), - 709: ( - "This error indicates that the context current to the calling thread" - " has been destroyed using ::cuCtxDestroy, or is a primary context which" - " has not yet been initialized." - ), - 710: ( - "A device-side assert triggered during kernel execution. The context" - " cannot be used anymore, and must be destroyed. All existing device" - " memory allocations from this context are invalid and must be" - " reconstructed if the program is to continue using CUDA." - ), - 711: ( - "This error indicates that the hardware resources required to enable" - " peer access have been exhausted for one or more of the devices" - " passed to ::cuCtxEnablePeerAccess()." - ), - 712: ("This error indicates that the memory range passed to ::cuMemHostRegister() has already been registered."), - 713: ( - "This error indicates that the pointer passed to ::cuMemHostUnregister()" - " does not correspond to any currently registered memory region." - ), - 714: ( - "While executing a kernel, the device encountered a stack error." - " This can be due to stack corruption or exceeding the stack size limit." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 715: ( - "While executing a kernel, the device encountered an illegal instruction." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 716: ( - "While executing a kernel, the device encountered a load or store instruction" - " on a memory address which is not aligned." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 717: ( - "While executing a kernel, the device encountered an instruction" - " which can only operate on memory locations in certain address spaces" - " (global, shared, or local), but was supplied a memory address not" - " belonging to an allowed address space." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 718: ( - "While executing a kernel, the device program counter wrapped its address space." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 719: ( - "An exception occurred on the device while executing a kernel. Common" - " causes include dereferencing an invalid device pointer and accessing" - " out of bounds shared memory. Less common cases can be system specific - more" - " information about these cases can be found in the system specific user guide." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 720: ( - "This error indicates that the number of blocks launched per grid for a kernel that was" - " launched via either ::cuLaunchCooperativeKernel or ::cuLaunchCooperativeKernelMultiDevice" - " exceeds the maximum number of blocks as allowed by ::cuOccupancyMaxActiveBlocksPerMultiprocessor" - " or ::cuOccupancyMaxActiveBlocksPerMultiprocessorWithFlags times the number of multiprocessors" - " as specified by the device attribute ::CU_DEVICE_ATTRIBUTE_MULTIPROCESSOR_COUNT." - ), - 721: ( - "An exception occurred on the device while exiting a kernel using tensor memory: the" - " tensor memory was not completely deallocated. This leaves the process in an inconsistent" - " state and any further CUDA work will return the same error. To continue using CUDA, the" - " process must be terminated and relaunched." - ), - 800: "This error indicates that the attempted operation is not permitted.", - 801: ("This error indicates that the attempted operation is not supported on the current system or device."), - 802: ( - "This error indicates that the system is not yet ready to start any CUDA" - " work. To continue using CUDA, verify the system configuration is in a" - " valid state and all required driver daemons are actively running." - " More information about this error can be found in the system specific" - " user guide." - ), - 803: ( - "This error indicates that there is a mismatch between the versions of" - " the display driver and the CUDA driver. Refer to the compatibility documentation" - " for supported versions." - ), - 804: ( - "This error indicates that the system was upgraded to run with forward compatibility" - " but the visible hardware detected by CUDA does not support this configuration." - " Refer to the compatibility documentation for the supported hardware matrix or ensure" - " that only supported hardware is visible during initialization via the CUDA_VISIBLE_DEVICES" - " environment variable." - ), - 805: "This error indicates that the MPS client failed to connect to the MPS control daemon or the MPS server.", - 806: "This error indicates that the remote procedural call between the MPS server and the MPS client failed.", - 807: ( - "This error indicates that the MPS server is not ready to accept new MPS client requests." - " This error can be returned when the MPS server is in the process of recovering from a fatal failure." - ), - 808: "This error indicates that the hardware resources required to create MPS client have been exhausted.", - 809: "This error indicates the the hardware resources required to support device connections have been exhausted.", - 810: "This error indicates that the MPS client has been terminated by the server. To continue using CUDA, the process must be terminated and relaunched.", - 811: "This error indicates that the module is using CUDA Dynamic Parallelism, but the current configuration, like MPS, does not support it.", - 812: "This error indicates that a module contains an unsupported interaction between different versions of CUDA Dynamic Parallelism.", - 900: ("This error indicates that the operation is not permitted when the stream is capturing."), - 901: ( - "This error indicates that the current capture sequence on the stream" - " has been invalidated due to a previous error." - ), - 902: ( - "This error indicates that the operation would have resulted in a merge of two independent capture sequences." - ), - 903: "This error indicates that the capture was not initiated in this stream.", - 904: ("This error indicates that the capture sequence contains a fork that was not joined to the primary stream."), - 905: ( - "This error indicates that a dependency would have been created which" - " crosses the capture sequence boundary. Only implicit in-stream ordering" - " dependencies are allowed to cross the boundary." - ), - 906: ("This error indicates a disallowed implicit dependency on a current capture sequence from cudaStreamLegacy."), - 907: ( - "This error indicates that the operation is not permitted on an event which" - " was last recorded in a capturing stream." - ), - 908: ( - "A stream capture sequence not initiated with the ::CU_STREAM_CAPTURE_MODE_RELAXED" - " argument to ::cuStreamBeginCapture was passed to ::cuStreamEndCapture in a" - " different thread." - ), - 909: "This error indicates that the timeout specified for the wait operation has lapsed.", - 910: ( - "This error indicates that the graph update was not performed because it included" - " changes which violated constraints specific to instantiated graph update." - ), - 911: ( - "This indicates that an async error has occurred in a device outside of CUDA." - " If CUDA was waiting for an external device's signal before consuming shared data," - " the external device signaled an error indicating that the data is not valid for" - " consumption. This leaves the process in an inconsistent state and any further CUDA" - " work will return the same error. To continue using CUDA, the process must be" - " terminated and relaunched." - ), - 912: "Indicates a kernel launch error due to cluster misconfiguration.", - 913: ("Indiciates a function handle is not loaded when calling an API that requires a loaded function."), - 914: ("This error indicates one or more resources passed in are not valid resource types for the operation."), - 915: ("This error indicates one or more resources are insufficient or non-applicable for the operation."), - 916: ("This error indicates that an error happened during the key rotation sequence."), - 917: ( - "This error indicates that the requested operation is not permitted because the" - " stream is in a detached state. This can occur if the green context associated" - " with the stream has been destroyed, limiting the stream's operational capabilities." - ), - 999: "This indicates that an unknown internal error has occurred.", -} diff --git a/cuda_core/cuda/core/_utils/enum_explanations_helpers.py b/cuda_core/cuda/core/_utils/enum_explanations_helpers.py index 6b666f4536c..e5e46145a83 100644 --- a/cuda_core/cuda/core/_utils/enum_explanations_helpers.py +++ b/cuda_core/cuda/core/_utils/enum_explanations_helpers.py @@ -3,62 +3,24 @@ """Internal support for error-enum explanations. -``cuda_core`` keeps frozen 13.1.1 fallback tables for older ``cuda-bindings`` -releases. Driver/runtime error enums carry usable ``__doc__`` text starting in -the 12.x backport line at ``cuda-bindings`` 12.9.6, and in the mainline 13.x -series at ``cuda-bindings`` 13.2.0. This module decides which source to use -and normalizes generated docstrings so user-facing ``CUDAError`` messages stay -presentable. +Driver and runtime error enums in ``cuda-bindings`` carry per-member +``__doc__`` text (since 12.9.6 in the 12.x line and 13.2.0 in the 13.x line; +every ``cuda-bindings`` that ``cuda.core`` accepts has it). This module +normalizes those generated docstrings so user-facing ``CUDAError`` messages +stay presentable. The cleanup rules here were derived while validating generated enum docstrings -in PR #1805. Keep them narrow and remove them when codegen quirks or fallback -support are no longer needed. +in PR #1805. Keep them narrow and remove them when the codegen quirks are gone. """ from __future__ import annotations -import importlib.metadata import re -from collections.abc import Callable from typing import Any -_MIN_12X_BINDING_VERSION_FOR_ENUM_DOCSTRINGS = (12, 9, 6) -_MIN_13X_BINDING_VERSION_FOR_ENUM_DOCSTRINGS = (13, 2, 0) _RST_INLINE_ROLE_RE = re.compile(r":(?:[a-z]+:)?[a-z]+:`([^`]+)`") _WORDWRAP_HYPHEN_AFTER_RE = re.compile(r"(?<=[0-9A-Za-z_])- (?=[0-9A-Za-z_])") _WORDWRAP_HYPHEN_BEFORE_RE = re.compile(r"(?<=[0-9A-Za-z_]) -(?=[0-9A-Za-z_])") -_ExplanationTable = dict[int, str | tuple[str, ...]] -_ExplanationTableLoader = Callable[[], _ExplanationTable] - - -def _parse_version_triple(version_str: str) -> tuple[int, int, int]: - """Parse a PEP 440 version string into a (major, minor, patch) triple. - - Strips local-version identifiers and handles pre-release suffixes such as - ``0b1`` or ``0rc1`` by extracting only the leading integer from each - release segment. - """ - parts = version_str.partition("+")[0].split(".")[:3] - ints = ([int(m.group(1)) if (m := re.match(r"(\d+)", v)) else 0 for v in parts] + [0, 0, 0])[:3] - return (ints[0], ints[1], ints[2]) - - -# ``version.pyx`` cannot be reused here (circular import via ``cuda_utils``). -def _binding_version() -> tuple[int, int, int]: - """Return the installed ``cuda-bindings`` version, or a conservative old value.""" - try: - version = importlib.metadata.version("cuda-bindings") - except importlib.metadata.PackageNotFoundError: - return (0, 0, 0) # For very old versions of cuda-python - return _parse_version_triple(version) - - -def _binding_version_has_usable_enum_docstrings(version: tuple[int, int, int]) -> bool: - """Whether released bindings are known to carry usable error-enum ``__doc__`` text.""" - return ( - _MIN_12X_BINDING_VERSION_FOR_ENUM_DOCSTRINGS <= version < (13, 0, 0) - or version >= _MIN_13X_BINDING_VERSION_FOR_ENUM_DOCSTRINGS - ) def _fix_hyphenation_wordwrap_spacing(s: str) -> str: @@ -102,9 +64,9 @@ def clean_enum_member_docstring(doc: str | None) -> str | None: class DocstringBackedExplanations: - """Compatibility shim exposing enum-member ``__doc__`` text via ``dict.get``. + """Expose enum-member ``__doc__`` text via ``dict.get``. - Keeps the existing ``.get(int(error))`` lookup shape used by ``cuda_utils.pyx``. + Keeps the ``.get(int(error))`` lookup shape used by ``cuda_utils.pyx``. """ __slots__ = ("_enum_type",) @@ -123,20 +85,3 @@ def get(self, code: int, default: str | None = None) -> str | None: return default return clean_enum_member_docstring(raw_doc) - - -def get_best_available_explanations( - enum_type: Any, - fallback: _ExplanationTable | _ExplanationTableLoader, -) -> DocstringBackedExplanations | _ExplanationTable: - """Pick one explanation source per bindings version. - - Use enum-member ``__doc__`` only for bindings versions known to expose - usable per-member text (12.9.6+ in the 12.x backport line, 13.2.0+ in the - 13.x mainline). Otherwise keep using the frozen 13.1.1 fallback tables. - """ - if not _binding_version_has_usable_enum_docstrings(_binding_version()): - if callable(fallback): - return fallback() - return fallback - return DocstringBackedExplanations(enum_type) diff --git a/cuda_core/cuda/core/_utils/runtime_cuda_error_explanations.py b/cuda_core/cuda/core/_utils/runtime_cuda_error_explanations.py index 12a49d2ec96..3bff19207e1 100644 --- a/cuda_core/cuda/core/_utils/runtime_cuda_error_explanations.py +++ b/cuda_core/cuda/core/_utils/runtime_cuda_error_explanations.py @@ -4,13 +4,6 @@ from __future__ import annotations from cuda.bindings import runtime -from cuda.core._utils.enum_explanations_helpers import get_best_available_explanations +from cuda.core._utils.enum_explanations_helpers import DocstringBackedExplanations - -def _load_fallback_explanations() -> dict[int, str | tuple[str, ...]]: - from cuda.core._utils.runtime_cuda_error_explanations_frozen import _FALLBACK_EXPLANATIONS - - return _FALLBACK_EXPLANATIONS # type: ignore[return-value] - - -RUNTIME_CUDA_ERROR_EXPLANATIONS = get_best_available_explanations(runtime.cudaError_t, _load_fallback_explanations) +RUNTIME_CUDA_ERROR_EXPLANATIONS = DocstringBackedExplanations(runtime.cudaError_t) diff --git a/cuda_core/cuda/core/_utils/runtime_cuda_error_explanations_frozen.py b/cuda_core/cuda/core/_utils/runtime_cuda_error_explanations_frozen.py deleted file mode 100644 index 017c4087400..00000000000 --- a/cuda_core/cuda/core/_utils/runtime_cuda_error_explanations_frozen.py +++ /dev/null @@ -1,538 +0,0 @@ -# SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 - -# CUDA Toolkit v13.1.1 -_FALLBACK_EXPLANATIONS = { - 0: ( - "The API call returned with no errors. In the case of query calls, this" - " also means that the operation being queried is complete (see" - " ::cudaEventQuery() and ::cudaStreamQuery())." - ), - 1: ( - "This indicates that one or more of the parameters passed to the API call" - " is not within an acceptable range of values." - ), - 2: ( - "The API call failed because it was unable to allocate enough memory or" - " other resources to perform the requested operation." - ), - 3: ("The API call failed because the CUDA driver and runtime could not be initialized."), - 4: ( - "This indicates that a CUDA Runtime API call cannot be executed because" - " it is being called during process shut down, at a point in time after" - " CUDA driver has been unloaded." - ), - 5: ( - "This indicates profiler is not initialized for this run. This can" - " happen when the application is running with external profiling tools" - " like visual profiler." - ), - 6: ( - "This error return is deprecated as of CUDA 5.0. It is no longer an error" - " to attempt to enable/disable the profiling via ::cudaProfilerStart or" - " ::cudaProfilerStop without initialization." - ), - 7: ( - "This error return is deprecated as of CUDA 5.0. It is no longer an error" - " to call cudaProfilerStart() when profiling is already enabled." - ), - 8: ( - "This error return is deprecated as of CUDA 5.0. It is no longer an error" - " to call cudaProfilerStop() when profiling is already disabled." - ), - 9: ( - "This indicates that a kernel launch is requesting resources that can" - " never be satisfied by the current device. Requesting more shared memory" - " per block than the device supports will trigger this error, as will" - " requesting too many threads or blocks. See ::cudaDeviceProp for more" - " device limitations." - ), - 12: ( - "This indicates that one or more of the pitch-related parameters passed" - " to the API call is not within the acceptable range for pitch." - ), - 13: ("This indicates that the symbol name/identifier passed to the API call is not a valid name or identifier."), - 16: ( - "This indicates that at least one host pointer passed to the API call is" - " not a valid host pointer." - " This error return is deprecated as of CUDA 10.1." - ), - 17: ( - "This indicates that at least one device pointer passed to the API call is" - " not a valid device pointer." - " This error return is deprecated as of CUDA 10.1." - ), - 18: ("This indicates that the texture passed to the API call is not a valid texture."), - 19: ( - "This indicates that the texture binding is not valid. This occurs if you" - " call ::cudaGetTextureAlignmentOffset() with an unbound texture." - ), - 20: ( - "This indicates that the channel descriptor passed to the API call is not" - " valid. This occurs if the format is not one of the formats specified by" - " ::cudaChannelFormatKind, or if one of the dimensions is invalid." - ), - 21: ( - "This indicates that the direction of the memcpy passed to the API call is" - " not one of the types specified by ::cudaMemcpyKind." - ), - 22: ( - "This indicated that the user has taken the address of a constant variable," - " which was forbidden up until the CUDA 3.1 release." - " This error return is deprecated as of CUDA 3.1. Variables in constant" - " memory may now have their address taken by the runtime via" - " ::cudaGetSymbolAddress()." - ), - 23: ( - "This indicated that a texture fetch was not able to be performed." - " This was previously used for device emulation of texture operations." - " This error return is deprecated as of CUDA 3.1. Device emulation mode was" - " removed with the CUDA 3.1 release." - ), - 24: ( - "This indicated that a texture was not bound for access." - " This was previously used for device emulation of texture operations." - " This error return is deprecated as of CUDA 3.1. Device emulation mode was" - " removed with the CUDA 3.1 release." - ), - 25: ( - "This indicated that a synchronization operation had failed." - " This was previously used for some device emulation functions." - " This error return is deprecated as of CUDA 3.1. Device emulation mode was" - " removed with the CUDA 3.1 release." - ), - 26: ( - "This indicates that a non-float texture was being accessed with linear" - " filtering. This is not supported by CUDA." - ), - 27: ( - "This indicates that an attempt was made to read an unsupported data type as a" - " normalized float. This is not supported by CUDA." - ), - 28: ( - "Mixing of device and device emulation code was not allowed." - " This error return is deprecated as of CUDA 3.1. Device emulation mode was" - " removed with the CUDA 3.1 release." - ), - 31: ( - "This indicates that the API call is not yet implemented. Production" - " releases of CUDA will never return this error." - " This error return is deprecated as of CUDA 4.1." - ), - 32: ( - "This indicated that an emulated device pointer exceeded the 32-bit address" - " range." - " This error return is deprecated as of CUDA 3.1. Device emulation mode was" - " removed with the CUDA 3.1 release." - ), - 34: ( - "This indicates that the CUDA driver that the application has loaded is a" - " stub library. Applications that run with the stub rather than a real" - " driver loaded will result in CUDA API returning this error." - ), - 35: ( - "This indicates that the installed NVIDIA CUDA driver is older than the" - " CUDA runtime library. This is not a supported configuration. Users should" - " install an updated NVIDIA display driver to allow the application to run." - ), - 36: ( - "This indicates that the API call requires a newer CUDA driver than the one" - " currently installed. Users should install an updated NVIDIA CUDA driver" - " to allow the API call to succeed." - ), - 37: ("This indicates that the surface passed to the API call is not a valid surface."), - 43: ( - "This indicates that multiple global or constant variables (across separate" - " CUDA source files in the application) share the same string name." - ), - 44: ( - "This indicates that multiple textures (across separate CUDA source" - " files in the application) share the same string name." - ), - 45: ( - "This indicates that multiple surfaces (across separate CUDA source" - " files in the application) share the same string name." - ), - 46: ( - "This indicates that all CUDA devices are busy or unavailable at the current" - " time. Devices are often busy/unavailable due to use of" - " ::cudaComputeModeProhibited, ::cudaComputeModeExclusiveProcess, or when long" - " running CUDA kernels have filled up the GPU and are blocking new work" - " from starting. They can also be unavailable due to memory constraints" - " on a device that already has active CUDA work being performed." - ), - 49: ( - "This indicates that the current context is not compatible with this" - " the CUDA Runtime. This can only occur if you are using CUDA" - " Runtime/Driver interoperability and have created an existing Driver" - " context using the driver API. The Driver context may be incompatible" - " either because the Driver context was created using an older version" - " of the API, because the Runtime API call expects a primary driver" - " context and the Driver context is not primary, or because the Driver" - ' context has been destroyed. Please see CUDART_DRIVER "Interactions' - ' with the CUDA Driver API" for more information.' - ), - 52: ( - "The device function being invoked (usually via ::cudaLaunchKernel()) was not" - " previously configured via the ::cudaConfigureCall() function." - ), - 53: ( - "This indicated that a previous kernel launch failed. This was previously" - " used for device emulation of kernel launches." - " This error return is deprecated as of CUDA 3.1. Device emulation mode was" - " removed with the CUDA 3.1 release." - ), - 65: ( - "This error indicates that a device runtime grid launch did not occur" - " because the depth of the child grid would exceed the maximum supported" - " number of nested grid launches." - ), - 66: ( - "This error indicates that a grid launch did not occur because the kernel" - " uses file-scoped textures which are unsupported by the device runtime." - " Kernels launched via the device runtime only support textures created with" - " the Texture Object API's." - ), - 67: ( - "This error indicates that a grid launch did not occur because the kernel" - " uses file-scoped surfaces which are unsupported by the device runtime." - " Kernels launched via the device runtime only support surfaces created with" - " the Surface Object API's." - ), - 68: ( - "This error indicates that a call to ::cudaDeviceSynchronize made from" - " the device runtime failed because the call was made at grid depth greater" - " than than either the default (2 levels of grids) or user specified device" - " limit ::cudaLimitDevRuntimeSyncDepth. To be able to synchronize on" - " launched grids at a greater depth successfully, the maximum nested" - " depth at which ::cudaDeviceSynchronize will be called must be specified" - " with the ::cudaLimitDevRuntimeSyncDepth limit to the ::cudaDeviceSetLimit" - " api before the host-side launch of a kernel using the device runtime." - " Keep in mind that additional levels of sync depth require the runtime" - " to reserve large amounts of device memory that cannot be used for" - " user allocations. Note that ::cudaDeviceSynchronize made from device" - " runtime is only supported on devices of compute capability < 9.0." - ), - 69: ( - "This error indicates that a device runtime grid launch failed because" - " the launch would exceed the limit ::cudaLimitDevRuntimePendingLaunchCount." - " For this launch to proceed successfully, ::cudaDeviceSetLimit must be" - " called to set the ::cudaLimitDevRuntimePendingLaunchCount to be higher" - " than the upper bound of outstanding launches that can be issued to the" - " device runtime. Keep in mind that raising the limit of pending device" - " runtime launches will require the runtime to reserve device memory that" - " cannot be used for user allocations." - ), - 98: ("The requested device function does not exist or is not compiled for the proper device architecture."), - 100: ("This indicates that no CUDA-capable devices were detected by the installed CUDA driver."), - 101: ( - "This indicates that the device ordinal supplied by the user does not" - " correspond to a valid CUDA device or that the action requested is" - " invalid for the specified device." - ), - 102: "This indicates that the device doesn't have a valid Grid License.", - 103: ( - "By default, the CUDA runtime may perform a minimal set of self-tests," - " as well as CUDA driver tests, to establish the validity of both." - " Introduced in CUDA 11.2, this error return indicates that at least one" - " of these tests has failed and the validity of either the runtime" - " or the driver could not be established." - ), - 127: "This indicates an internal startup failure in the CUDA runtime.", - 200: "This indicates that the device kernel image is invalid.", - 201: ( - "This most frequently indicates that there is no context bound to the" - " current thread. This can also be returned if the context passed to an" - " API call is not a valid handle (such as a context that has had" - " ::cuCtxDestroy() invoked on it). This can also be returned if a user" - " mixes different API versions (i.e. 3010 context with 3020 API calls)." - " See ::cuCtxGetApiVersion() for more details." - ), - 205: "This indicates that the buffer object could not be mapped.", - 206: "This indicates that the buffer object could not be unmapped.", - 207: ("This indicates that the specified array is currently mapped and thus cannot be destroyed."), - 208: "This indicates that the resource is already mapped.", - 209: ( - "This indicates that there is no kernel image available that is suitable" - " for the device. This can occur when a user specifies code generation" - " options for a particular CUDA source file that do not include the" - " corresponding device configuration." - ), - 210: "This indicates that a resource has already been acquired.", - 211: "This indicates that a resource is not mapped.", - 212: ("This indicates that a mapped resource is not available for access as an array."), - 213: ("This indicates that a mapped resource is not available for access as a pointer."), - 214: ("This indicates that an uncorrectable ECC error was detected during execution."), - 215: ("This indicates that the ::cudaLimit passed to the API call is not supported by the active device."), - 216: ( - "This indicates that a call tried to access an exclusive-thread device that" - " is already in use by a different thread." - ), - 217: ("This error indicates that P2P access is not supported across the given devices."), - 218: ( - "A PTX compilation failed. The runtime may fall back to compiling PTX if" - " an application does not contain a suitable binary for the current device." - ), - 219: "This indicates an error with the OpenGL or DirectX context.", - 220: ("This indicates that an uncorrectable NVLink error was detected during the execution."), - 221: ( - "This indicates that the PTX JIT compiler library was not found. The JIT Compiler" - " library is used for PTX compilation. The runtime may fall back to compiling PTX" - " if an application does not contain a suitable binary for the current device." - ), - 222: ( - "This indicates that the provided PTX was compiled with an unsupported toolchain." - " The most common reason for this, is the PTX was generated by a compiler newer" - " than what is supported by the CUDA driver and PTX JIT compiler." - ), - 223: ( - "This indicates that the JIT compilation was disabled. The JIT compilation compiles" - " PTX. The runtime may fall back to compiling PTX if an application does not contain" - " a suitable binary for the current device." - ), - 224: "This indicates that the provided execution affinity is not supported by the device.", - 225: ( - "This indicates that the code to be compiled by the PTX JIT contains unsupported call to cudaDeviceSynchronize." - ), - 226: ( - "This indicates that an exception occurred on the device that is now" - " contained by the GPU's error containment capability. Common causes are -" - " a. Certain types of invalid accesses of peer GPU memory over nvlink" - " b. Certain classes of hardware errors" - " This leaves the process in an inconsistent state and any further CUDA" - " work will return the same error. To continue using CUDA, the process must" - " be terminated and relaunched." - ), - 300: "This indicates that the device kernel source is invalid.", - 301: "This indicates that the file specified was not found.", - 302: "This indicates that a link to a shared object failed to resolve.", - 303: "This indicates that initialization of a shared object failed.", - 304: "This error indicates that an OS call failed.", - 400: ( - "This indicates that a resource handle passed to the API call was not" - " valid. Resource handles are opaque types like ::cudaStream_t and" - " ::cudaEvent_t." - ), - 401: ( - "This indicates that a resource required by the API call is not in a" - " valid state to perform the requested operation." - ), - 402: ( - "This indicates an attempt was made to introspect an object in a way that" - " would discard semantically important information. This is either due to" - " the object using funtionality newer than the API version used to" - " introspect it or omission of optional return arguments." - ), - 500: ( - "This indicates that a named symbol was not found. Examples of symbols" - " are global/constant variable names, driver function names, texture names," - " and surface names." - ), - 600: ( - "This indicates that asynchronous operations issued previously have not" - " completed yet. This result is not actually an error, but must be indicated" - " differently than ::cudaSuccess (which indicates completion). Calls that" - " may return this value include ::cudaEventQuery() and ::cudaStreamQuery()." - ), - 700: ( - "The device encountered a load or store instruction on an invalid memory address." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 701: ( - "This indicates that a launch did not occur because it did not have" - " appropriate resources. Although this error is similar to" - " ::cudaErrorInvalidConfiguration, this error usually indicates that the" - " user has attempted to pass too many arguments to the device kernel, or the" - " kernel launch specifies too many threads for the kernel's register count." - ), - 702: ( - "This indicates that the device kernel took too long to execute. This can" - " only occur if timeouts are enabled - see the device attribute" - ' ::cudaDeviceAttr::cudaDevAttrKernelExecTimeout "cudaDevAttrKernelExecTimeout"' - " for more information." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 703: ("This error indicates a kernel launch that uses an incompatible texturing mode."), - 704: ( - "This error indicates that a call to ::cudaDeviceEnablePeerAccess() is" - " trying to re-enable peer addressing on from a context which has already" - " had peer addressing enabled." - ), - 705: ( - "This error indicates that ::cudaDeviceDisablePeerAccess() is trying to" - " disable peer addressing which has not been enabled yet via" - " ::cudaDeviceEnablePeerAccess()." - ), - 708: ( - "This indicates that the user has called ::cudaSetValidDevices()," - " ::cudaSetDeviceFlags(), ::cudaD3D9SetDirect3DDevice()," - " ::cudaD3D10SetDirect3DDevice, ::cudaD3D11SetDirect3DDevice(), or" - " ::cudaVDPAUSetVDPAUDevice() after initializing the CUDA runtime by" - " calling non-device management operations (allocating memory and" - " launching kernels are examples of non-device management operations)." - " This error can also be returned if using runtime/driver" - " interoperability and there is an existing ::CUcontext active on the" - " host thread." - ), - 709: ( - "This error indicates that the context current to the calling thread" - " has been destroyed using ::cuCtxDestroy, or is a primary context which" - " has not yet been initialized." - ), - 710: ( - "An assert triggered in device code during kernel execution. The device" - " cannot be used again. All existing allocations are invalid. To continue" - " using CUDA, the process must be terminated and relaunched." - ), - 711: ( - "This error indicates that the hardware resources required to enable" - " peer access have been exhausted for one or more of the devices" - " passed to ::cudaEnablePeerAccess()." - ), - 712: ("This error indicates that the memory range passed to ::cudaHostRegister() has already been registered."), - 713: ( - "This error indicates that the pointer passed to ::cudaHostUnregister()" - " does not correspond to any currently registered memory region." - ), - 714: ( - "Device encountered an error in the call stack during kernel execution," - " possibly due to stack corruption or exceeding the stack size limit." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 715: ( - "The device encountered an illegal instruction during kernel execution" - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 716: ( - "The device encountered a load or store instruction" - " on a memory address which is not aligned." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 717: ( - "While executing a kernel, the device encountered an instruction" - " which can only operate on memory locations in certain address spaces" - " (global, shared, or local), but was supplied a memory address not" - " belonging to an allowed address space." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 718: ( - "The device encountered an invalid program counter." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 719: ( - "An exception occurred on the device while executing a kernel. Common" - " causes include dereferencing an invalid device pointer and accessing" - " out of bounds shared memory. Less common cases can be system specific - more" - " information about these cases can be found in the system specific user guide." - " This leaves the process in an inconsistent state and any further CUDA work" - " will return the same error. To continue using CUDA, the process must be terminated" - " and relaunched." - ), - 720: ( - "This error indicates that the number of blocks launched per grid for a kernel that was" - " launched via either ::cudaLaunchCooperativeKernel" - " exceeds the maximum number of blocks as allowed by ::cudaOccupancyMaxActiveBlocksPerMultiprocessor" - " or ::cudaOccupancyMaxActiveBlocksPerMultiprocessorWithFlags times the number of multiprocessors" - " as specified by the device attribute ::cudaDevAttrMultiProcessorCount." - ), - 721: ( - "An exception occurred on the device while exiting a kernel using tensor memory: the" - " tensor memory was not completely deallocated. This leaves the process in an inconsistent" - " state and any further CUDA work will return the same error. To continue using CUDA, the" - " process must be terminated and relaunched." - ), - 800: "This error indicates the attempted operation is not permitted.", - 801: ("This error indicates the attempted operation is not supported on the current system or device."), - 802: ( - "This error indicates that the system is not yet ready to start any CUDA" - " work. To continue using CUDA, verify the system configuration is in a" - " valid state and all required driver daemons are actively running." - " More information about this error can be found in the system specific" - " user guide." - ), - 803: ( - "This error indicates that there is a mismatch between the versions of" - " the display driver and the CUDA driver. Refer to the compatibility documentation" - " for supported versions." - ), - 804: ( - "This error indicates that the system was upgraded to run with forward compatibility" - " but the visible hardware detected by CUDA does not support this configuration." - " Refer to the compatibility documentation for the supported hardware matrix or ensure" - " that only supported hardware is visible during initialization via the CUDA_VISIBLE_DEVICES" - " environment variable." - ), - 805: "This error indicates that the MPS client failed to connect to the MPS control daemon or the MPS server.", - 806: "This error indicates that the remote procedural call between the MPS server and the MPS client failed.", - 807: ( - "This error indicates that the MPS server is not ready to accept new MPS client requests." - " This error can be returned when the MPS server is in the process of recovering from a fatal failure." - ), - 808: "This error indicates that the hardware resources required to create MPS client have been exhausted.", - 809: "This error indicates the the hardware resources required to device connections have been exhausted.", - 810: "This error indicates that the MPS client has been terminated by the server. To continue using CUDA, the process must be terminated and relaunched.", - 811: "This error indicates, that the program is using CUDA Dynamic Parallelism, but the current configuration, like MPS, does not support it.", - 812: "This error indicates, that the program contains an unsupported interaction between different versions of CUDA Dynamic Parallelism.", - 900: "The operation is not permitted when the stream is capturing.", - 901: ("The current capture sequence on the stream has been invalidated due to a previous error."), - 902: ("The operation would have resulted in a merge of two independent capture sequences."), - 903: "The capture was not initiated in this stream.", - 904: ("The capture sequence contains a fork that was not joined to the primary stream."), - 905: ( - "A dependency would have been created which crosses the capture sequence" - " boundary. Only implicit in-stream ordering dependencies are allowed to" - " cross the boundary." - ), - 906: ( - "The operation would have resulted in a disallowed implicit dependency on" - " a current capture sequence from cudaStreamLegacy." - ), - 907: ("The operation is not permitted on an event which was last recorded in a capturing stream."), - 908: ( - "A stream capture sequence not initiated with the ::cudaStreamCaptureModeRelaxed" - " argument to ::cudaStreamBeginCapture was passed to ::cudaStreamEndCapture in a" - " different thread." - ), - 909: "This indicates that the wait operation has timed out.", - 910: ( - "This error indicates that the graph update was not performed because it included" - " changes which violated constraints specific to instantiated graph update." - ), - 911: ( - "This indicates that an async error has occurred in a device outside of CUDA." - " If CUDA was waiting for an external device's signal before consuming shared data," - " the external device signaled an error indicating that the data is not valid for" - " consumption. This leaves the process in an inconsistent state and any further CUDA" - " work will return the same error. To continue using CUDA, the process must be" - " terminated and relaunched." - ), - 912: ("This indicates that a kernel launch error has occurred due to cluster misconfiguration."), - 913: ("Indiciates a function handle is not loaded when calling an API that requires a loaded function."), - 914: ("This error indicates one or more resources passed in are not valid resource types for the operation."), - 915: ("This error indicates one or more resources are insufficient or non-applicable for the operation."), - 917: ( - "This error indicates that the requested operation is not permitted because the" - " stream is in a detached state. This can occur if the green context associated" - " with the stream has been destroyed, limiting the stream's operational capabilities." - ), - 999: "This indicates that an unknown internal error has occurred.", - 10000: ( - "Any unhandled CUDA driver error is added to this value and returned via" - " the runtime. Production releases of CUDA should not return such errors." - " This error return is deprecated as of CUDA 4.1." - ), -} diff --git a/cuda_core/cuda/core/_utils/version.pxd b/cuda_core/cuda/core/_utils/version.pxd index 2746d463dba..075ce846add 100644 --- a/cuda_core/cuda/core/_utils/version.pxd +++ b/cuda_core/cuda/core/_utils/version.pxd @@ -2,5 +2,4 @@ # # SPDX-License-Identifier: Apache-2.0 -cdef tuple cy_binding_version() cdef tuple cy_driver_version() diff --git a/cuda_core/cuda/core/_utils/version.pyi b/cuda_core/cuda/core/_utils/version.pyi index db2e27a57d0..bd8defe7eab 100644 --- a/cuda_core/cuda/core/_utils/version.pyi +++ b/cuda_core/cuda/core/_utils/version.pyi @@ -10,6 +10,8 @@ def _parse_version_triple(version_str: str) -> tuple[int, int, int]: ``0b1`` or ``0rc1`` by extracting only the leading integer from each release segment. """ +BUILD_CUDA_MAJOR: int + @functools.cache def binding_version() -> tuple[int, int, int]: """Return the cuda-bindings version as a (major, minor, patch) triple.""" diff --git a/cuda_core/cuda/core/_utils/version.pyx b/cuda_core/cuda/core/_utils/version.pyx index ed4c93c0262..ae396af940d 100644 --- a/cuda_core/cuda/core/_utils/version.pyx +++ b/cuda_core/cuda/core/_utils/version.pyx @@ -8,6 +8,13 @@ import re from cuda.core._utils.cuda_utils import driver, handle_return +# The CUDA major series this build of cuda.core targets (12 or 13), from the +# compile-time environment build_hooks.py sets. The installed cuda-bindings has +# the same major (cuda/core/__init__.py enforces it at import). Python modules +# that must branch on the series, where `IF CUDA_CORE_BUILD_MAJOR` is not +# available, read this instead of comparing binding_version(). +BUILD_CUDA_MAJOR: int = CUDA_CORE_BUILD_MAJOR + def _parse_version_triple(version_str: str) -> tuple[int, int, int]: """Parse a PEP 440 version string into a (major, minor, patch) triple. @@ -38,17 +45,9 @@ def driver_version() -> tuple[int, int, int]: return (ver // 1000, (ver // 10) % 100, ver % 10) -cdef tuple _cached_binding_version = None cdef tuple _cached_driver_version = None -cdef tuple cy_binding_version(): - global _cached_binding_version - if _cached_binding_version is None: - _cached_binding_version = binding_version() - return _cached_binding_version - - cdef tuple cy_driver_version(): global _cached_driver_version if _cached_driver_version is None: diff --git a/cuda_core/cuda/core/checkpoint.py b/cuda_core/cuda/core/checkpoint.py index 32fe7e26d83..cf9e09ed14c 100644 --- a/cuda_core/cuda/core/checkpoint.py +++ b/cuda_core/cuda/core/checkpoint.py @@ -8,7 +8,7 @@ from cuda.bindings import driver as _driver from cuda.core._utils.cuda_utils import handle_return as _handle_cuda_return -from cuda.core._utils.version import binding_version as _binding_version +from cuda.core._utils.version import BUILD_CUDA_MAJOR as _BUILD_CUDA_MAJOR from cuda.core._utils.version import driver_version as _driver_version from cuda.core.typing import ProcessStateType as _ProcessStateType @@ -19,18 +19,6 @@ ("CU_PROCESS_STATE_FAILED", "failed"), ) -_REQUIRED_BINDING_ATTRS = ( - "cuCheckpointProcessCheckpoint", - "cuCheckpointProcessGetRestoreThreadId", - "cuCheckpointProcessGetState", - "cuCheckpointProcessLock", - "cuCheckpointProcessRestore", - "cuCheckpointProcessUnlock", - "CUcheckpointGpuPair", - "CUcheckpointLockArgs", - "CUprocessState", - "CUcheckpointRestoreArgs", -) _REQUIRED_DRIVER_VERSION = (12, 8, 0) _driver_capability_checked = False @@ -130,18 +118,10 @@ def _get_driver() -> Any: if _driver_capability_checked: return _driver - binding_ver = _binding_version() - if not _binding_version_supports_checkpoint(binding_ver): - raise RuntimeError( - "CUDA checkpointing requires cuda.bindings with CUDA checkpoint API support. " - f"Found cuda.bindings {'.'.join(str(part) for part in binding_ver[:3])}." - ) - - missing = [name for name in _REQUIRED_BINDING_ATTRS if not hasattr(_driver, name)] - if missing: - raise RuntimeError( - f"CUDA checkpointing requires cuda.bindings with CUDA checkpoint API support. Missing: {', '.join(missing)}" - ) + # Restoring onto other GPUs uses CUcheckpointGpuPair, a CUDA 13 type that + # the CUDA 12 build's cuda-bindings does not have. + if _BUILD_CUDA_MAJOR < 13: + raise RuntimeError("CUDA checkpointing requires the CUDA 13 build of cuda.core (cuda-core[cu13]).") driver_ver = _driver_version() if driver_ver < _REQUIRED_DRIVER_VERSION: @@ -154,11 +134,6 @@ def _get_driver() -> Any: return _driver -def _binding_version_supports_checkpoint(version: tuple[int, ...]) -> bool: - major, minor, patch = version[:3] - return (major == 12 and (minor, patch) >= (8, 0)) or (major == 13 and (minor, patch) >= (0, 2)) or major > 13 - - def _get_process_state_names(driver: Any) -> dict[Any, _ProcessStateType]: return {getattr(driver.CUprocessState, attr): state_name for attr, state_name in _PROCESS_STATE_NAME_ATTRS} diff --git a/cuda_core/cuda/core/graph/_graph_builder.pyx b/cuda_core/cuda/core/graph/_graph_builder.pyx index 03672316bef..8dd3ac6d3a4 100644 --- a/cuda_core/cuda/core/graph/_graph_builder.pyx +++ b/cuda_core/cuda/core/graph/_graph_builder.pyx @@ -52,7 +52,7 @@ from cuda.core._rt cimport ( ) from cuda.core._stream cimport Stream, Stream_accept from cuda.core._utils.cuda_utils cimport HANDLE_RETURN -from cuda.core._utils.version cimport cy_binding_version, cy_driver_version +from cuda.core._utils.version cimport cy_driver_version from cuda.core._utils.cuda_utils import ( CUDAError, @@ -251,10 +251,7 @@ def _instantiate_graph(source, options: GraphCompleteOptions | None = None) -> G ) elif params.result_out == driver.CUgraphInstantiateResult.CUDA_GRAPH_INSTANTIATE_MULTIPLE_CTXS_NOT_SUPPORTED: raise RuntimeError("Instantiation for device launch failed due to the nodes belonging to different contexts.") - elif ( - cy_binding_version() >= (12, 8, 0) - and params.result_out == driver.CUgraphInstantiateResult.CUDA_GRAPH_INSTANTIATE_CONDITIONAL_HANDLE_UNUSED - ): + elif params.result_out == driver.CUgraphInstantiateResult.CUDA_GRAPH_INSTANTIATE_CONDITIONAL_HANDLE_UNUSED: raise RuntimeError("One or more conditional handles are not associated with conditional builders.") elif params.result_out != driver.CUgraphInstantiateResult.CUDA_GRAPH_INSTANTIATE_SUCCESS: raise RuntimeError(f"Graph instantiation failed with unexpected error code: {params.result_out}") @@ -666,8 +663,6 @@ cdef class GraphBuilder: GB_check_open(self) if cy_driver_version() < (12, 3, 0): raise RuntimeError(f"Driver version {'.'.join(map(str, cy_driver_version()))} does not support conditional handles") - if cy_binding_version() < (12, 3, 0): - raise RuntimeError(f"Binding version {'.'.join(map(str, cy_binding_version()))} does not support conditional handles") if default_value is not None: flags = driver.CU_GRAPH_COND_ASSIGN_DEFAULT else: @@ -706,8 +701,6 @@ cdef class GraphBuilder: GB_check_open(self) if cy_driver_version() < (12, 3, 0): raise RuntimeError(f"Driver version {'.'.join(map(str, cy_driver_version()))} does not support conditional if") - if cy_binding_version() < (12, 3, 0): - raise RuntimeError(f"Binding version {'.'.join(map(str, cy_binding_version()))} does not support conditional if") if not isinstance(condition, GraphCondition): raise TypeError( f"condition must be a GraphCondition object (from " @@ -743,8 +736,6 @@ cdef class GraphBuilder: GB_check_open(self) if cy_driver_version() < (12, 8, 0): raise RuntimeError(f"Driver version {'.'.join(map(str, cy_driver_version()))} does not support conditional if-else") - if cy_binding_version() < (12, 8, 0): - raise RuntimeError(f"Binding version {'.'.join(map(str, cy_binding_version()))} does not support conditional if-else") if not isinstance(condition, GraphCondition): raise TypeError( f"condition must be a GraphCondition object (from " @@ -783,8 +774,6 @@ cdef class GraphBuilder: GB_check_open(self) if cy_driver_version() < (12, 8, 0): raise RuntimeError(f"Driver version {'.'.join(map(str, cy_driver_version()))} does not support conditional switch") - if cy_binding_version() < (12, 8, 0): - raise RuntimeError(f"Binding version {'.'.join(map(str, cy_binding_version()))} does not support conditional switch") if not isinstance(condition, GraphCondition): raise TypeError( f"condition must be a GraphCondition object (from " @@ -820,8 +809,6 @@ cdef class GraphBuilder: GB_check_open(self) if cy_driver_version() < (12, 3, 0): raise RuntimeError(f"Driver version {'.'.join(map(str, cy_driver_version()))} does not support conditional while loop") - if cy_binding_version() < (12, 3, 0): - raise RuntimeError(f"Binding version {'.'.join(map(str, cy_binding_version()))} does not support conditional while loop") if not isinstance(condition, GraphCondition): raise TypeError( f"condition must be a GraphCondition object (from " diff --git a/cuda_core/cuda/core/graph/_subclasses.pxd b/cuda_core/cuda/core/graph/_subclasses.pxd index c85f5e3f201..b7c37c64bcb 100644 --- a/cuda_core/cuda/core/graph/_subclasses.pxd +++ b/cuda_core/cuda/core/graph/_subclasses.pxd @@ -162,6 +162,9 @@ cdef class ConditionalNode(GraphNode): @staticmethod cdef ConditionalNode _create_from_driver(GraphNodeHandle h_node) + IF CUDA_CORE_BUILD_MAJOR >= 13: + @staticmethod + cdef ConditionalNode _create_from_driver_params(GraphNodeHandle h_node) cdef class IfNode(ConditionalNode): diff --git a/cuda_core/cuda/core/graph/_subclasses.pyx b/cuda_core/cuda/core/graph/_subclasses.pyx index 0f77cabfb39..1e1d2c1953c 100644 --- a/cuda_core/cuda/core/graph/_subclasses.pyx +++ b/cuda_core/cuda/core/graph/_subclasses.pyx @@ -60,14 +60,13 @@ from cuda.core._rt cimport ( make_opaque_py, ) from cuda.core._utils.cuda_utils cimport HANDLE_RETURN, _parse_fill_value -from cuda.core._utils.version cimport cy_binding_version, cy_driver_version +from cuda.core._utils.version cimport cy_driver_version from cuda.core.graph._host_callback cimport ( _is_py_host_trampoline, _resolve_host_callback, ) -from cuda.core._utils.cuda_utils import driver, handle_return from cuda.core.typing import GraphConditionalType __all__ = [ @@ -97,10 +96,6 @@ __all__ = [ ] -cdef bint _has_cuGraphNodeGetParams = False -cdef bint _version_checked = False - - cdef void _require_graph_node_update_support() except *: cdef tuple version = cy_driver_version() if version < (12, 2, 0): @@ -108,12 +103,6 @@ cdef void _require_graph_node_update_support() except *: "Graph node mutation requires CUDA driver 12.2 or newer; " f"using driver version {'.'.join(map(str, version))}" ) - version = cy_binding_version() - if version < (12, 2, 0): - raise RuntimeError( - "Graph node mutation requires cuda.bindings 12.2 or newer; " - f"using cuda.bindings version {'.'.join(map(str, version))}" - ) cdef void _set_definition_node_params( @@ -214,14 +203,22 @@ cdef void _set_executable_node_enabled( cdef bint _check_node_get_params(): - global _has_cuGraphNodeGetParams, _version_checked - if not _version_checked: - from cuda.core._utils.version import binding_version, driver_version - _has_cuGraphNodeGetParams = ( - driver_version() >= (13, 2, 0) and binding_version() >= (13, 2, 0) - ) - _version_checked = True - return _has_cuGraphNodeGetParams + """Whether cuGraphNodeGetParams (CUDA 13.2) can be called. + + The CUDA 13 build always has the binding; only the driver can lack it.""" + IF CUDA_CORE_BUILD_MAJOR >= 13: + return cy_driver_version() >= (13, 2, 0) + ELSE: + return False + + +IF CUDA_CORE_BUILD_MAJOR >= 13: + cdef void _node_get_params( + cydriver.CUgraphNode node, + cydriver.CUgraphNodeParams* params) except *: + c_memset(params, 0, sizeof(params[0])) + with nogil: + HANDLE_RETURN(cydriver.cuGraphNodeGetParams(node, params)) cdef void _reject_unsupported_kernel_node( @@ -671,7 +668,7 @@ cdef class MemsetNode(GraphNode): cdef cydriver.CUcontext ctx = NULL cdef cydriver.CUDA_MEMSET_NODE_PARAMS current cdef cydriver.CUgraphNodeParams params - cdef object queried + cdef cydriver.CUgraphNodeParams queried # no-cython-lint if dst is None and dst_owner is not None: raise ValueError("dst_owner requires dst") @@ -684,11 +681,14 @@ cdef class MemsetNode(GraphNode): with nogil: HANDLE_RETURN(cydriver.cuGraphMemsetNodeGetParams( node, ¤t)) - if _check_node_get_params(): - queried = handle_return(driver.cuGraphNodeGetParams( - node)) - ctx = int(queried.memset.ctx) - else: + IF CUDA_CORE_BUILD_MAJOR >= 13: + if _check_node_get_params(): + _node_get_params(node, &queried) + ctx = queried.memset.ctx + else: + with nogil: + HANDLE_RETURN(cydriver.cuCtxGetCurrent(&ctx)) + ELSE: with nogil: HANDLE_RETURN(cydriver.cuCtxGetCurrent(&ctx)) @@ -861,7 +861,7 @@ cdef class MemcpyNode(GraphNode): cdef cydriver.CUgraphNodeParams params cdef cydriver.CUmemorytype c_dst_type cdef cydriver.CUmemorytype c_src_type - cdef object queried + cdef cydriver.CUgraphNodeParams queried # no-cython-lint if dst is None and dst_owner is not None: raise ValueError("dst_owner requires dst") @@ -875,12 +875,14 @@ cdef class MemcpyNode(GraphNode): with nogil: HANDLE_RETURN(cydriver.cuGraphMemcpyNodeGetParams( node, ¶ms.memcpy.copyParams)) - if _check_node_get_params(): - queried = handle_return(driver.cuGraphNodeGetParams( - node)) - ctx = int( - queried.memcpy.copyCtx) - else: + IF CUDA_CORE_BUILD_MAJOR >= 13: + if _check_node_get_params(): + _node_get_params(node, &queried) + ctx = queried.memcpy.copyCtx + else: + with nogil: + HANDLE_RETURN(cydriver.cuCtxGetCurrent(&ctx)) + ELSE: with nogil: HANDLE_RETURN(cydriver.cuCtxGetCurrent(&ctx)) params.memcpy.copyCtx = ctx @@ -1262,47 +1264,52 @@ cdef class ConditionalNode(GraphNode): n._cond_type = cydriver.CU_GRAPH_COND_TYPE_IF n._branches = () return n - - cdef cydriver.CUgraphNode node = as_cu(h_node) - params = handle_return(driver.cuGraphNodeGetParams( - node)) - cond_params = params.conditional - cdef int cond_type_int = int(cond_params.type) - cdef unsigned int size = int(cond_params.size) - - cdef GraphCondition condition = GraphCondition.__new__(GraphCondition) - condition._c_handle = ( - int(cond_params.handle)) - - cdef GraphHandle h_graph = graph_node_get_graph(h_node) - cdef list branch_list = [] - cdef unsigned int i - cdef GraphHandle h_branch - if cond_params.phGraph_out is not None: - for i in range(size): - h_branch = create_child_graph_handle( - int(cond_params.phGraph_out[i]), - h_graph, node) - branch_list.append(GraphDefinition._from_handle(h_branch)) - cdef tuple branches = tuple(branch_list) - - cdef type cls - if cond_type_int == cydriver.CU_GRAPH_COND_TYPE_IF: - if size == 1: - cls = IfNode + IF CUDA_CORE_BUILD_MAJOR >= 13: + return ConditionalNode._create_from_driver_params(h_node) + ELSE: + raise AssertionError("unreachable: cuGraphNodeGetParams needs the CUDA 13 build") + + IF CUDA_CORE_BUILD_MAJOR >= 13: + @staticmethod + cdef ConditionalNode _create_from_driver_params(GraphNodeHandle h_node): + cdef ConditionalNode n + cdef cydriver.CUgraphNode node = as_cu(h_node) + cdef cydriver.CUgraphNodeParams params + _node_get_params(node, ¶ms) + cdef int cond_type_int = params.conditional.type + cdef unsigned int size = params.conditional.size + + cdef GraphCondition condition = GraphCondition.__new__(GraphCondition) + condition._c_handle = params.conditional.handle + + cdef GraphHandle h_graph = graph_node_get_graph(h_node) + cdef list branch_list = [] + cdef unsigned int i + cdef GraphHandle h_branch + if params.conditional.phGraph_out != NULL: + for i in range(size): + h_branch = create_child_graph_handle( + params.conditional.phGraph_out[i], h_graph, node) + branch_list.append(GraphDefinition._from_handle(h_branch)) + cdef tuple branches = tuple(branch_list) + + cdef type cls + if cond_type_int == cydriver.CU_GRAPH_COND_TYPE_IF: + if size == 1: + cls = IfNode + else: + cls = IfElseNode + elif cond_type_int == cydriver.CU_GRAPH_COND_TYPE_WHILE: + cls = WhileNode else: - cls = IfElseNode - elif cond_type_int == cydriver.CU_GRAPH_COND_TYPE_WHILE: - cls = WhileNode - else: - cls = SwitchNode + cls = SwitchNode - n = cls.__new__(cls) - n._h_node = h_node - n._condition = condition - n._cond_type = cond_type_int - n._branches = branches - return n + n = cls.__new__(cls) + n._h_node = h_node + n._condition = condition + n._cond_type = cond_type_int + n._branches = branches + return n def __repr__(self) -> str: return f"" diff --git a/cuda_core/cuda/core/system/__init__.py b/cuda_core/cuda/core/system/__init__.py index acb648549bc..b73927f0f6f 100644 --- a/cuda_core/cuda/core/system/__init__.py +++ b/cuda_core/cuda/core/system/__init__.py @@ -8,8 +8,6 @@ # contexts created, so that a user can use NVML to explore things about their # system without loading CUDA. -from typing import TYPE_CHECKING - __all__ = [ "CUDA_BINDINGS_NVML_IS_COMPATIBLE", "get_driver_branch", @@ -23,25 +21,14 @@ from cuda.core.system import typing +from ._device import * +from ._device import __all__ as _device_all from ._system import * - -# The TYPE_CHECKING branch is split out from the runtime branch so that -# stubgen-pyx, which only recognizes the literal `if TYPE_CHECKING:` form, -# preserves these imports in the generated .pyi. When -# CUDA_BINDINGS_NVML_IS_COMPATIBLE is no longer necessary, this complexity can -# be removed. -if TYPE_CHECKING: - from ._device import * - from ._system_events import * - from .exceptions import * -elif CUDA_BINDINGS_NVML_IS_COMPATIBLE: - from ._device import * - from ._device import __all__ as _device_all - from ._system_events import * - from ._system_events import __all__ as _system_events_all - from .exceptions import * - from .exceptions import __all__ as _exceptions_all - - __all__.extend(_device_all) - __all__.extend(_system_events_all) - __all__.extend(_exceptions_all) +from ._system_events import * +from ._system_events import __all__ as _system_events_all +from .exceptions import * +from .exceptions import __all__ as _exceptions_all + +__all__.extend(_device_all) +__all__.extend(_system_events_all) +__all__.extend(_exceptions_all) diff --git a/cuda_core/cuda/core/system/_nvlink.pxi b/cuda_core/cuda/core/system/_nvlink.pxi index 49ac1b75ba1..acd64c63fed 100644 --- a/cuda_core/cuda/core/system/_nvlink.pxi +++ b/cuda_core/cuda/core/system/_nvlink.pxi @@ -11,12 +11,9 @@ _NVLINK_VERSION_MAPPING = { nvml.NvlinkVersion.VERSION_3_1: (3, 1), nvml.NvlinkVersion.VERSION_4_0: (4, 0), nvml.NvlinkVersion.VERSION_5_0: (5, 0), + nvml.NvlinkVersion.VERSION_6_0: (6, 0), } -_NVLINK_VERSION_6_0 = getattr(nvml.NvlinkVersion, "VERSION_6_0", None) -if _NVLINK_VERSION_6_0 is not None: - _NVLINK_VERSION_MAPPING[_NVLINK_VERSION_6_0] = (6, 0) - class _NvlinkInfoMeta(type): @property diff --git a/cuda_core/cuda/core/system/_system.pyx b/cuda_core/cuda/core/system/_system.pyx index 2a6c8ffc23d..c7414f46e1a 100644 --- a/cuda_core/cuda/core/system/_system.pyx +++ b/cuda_core/cuda/core/system/_system.pyx @@ -3,12 +3,15 @@ # SPDX-License-Identifier: Apache-2.0 -# This file needs to either use NVML exclusively, or when `cuda.bindings.nvml` -# isn't available, fall back to non-NVML-based methods for backward -# compatibility. +# cuda.core.system uses NVML through cuda.bindings.nvml, which every +# cuda-bindings cuda.core accepts provides (cuda/core/_bindings_floor.py). +# Loading the NVML library itself happens in initialize(), on first use, so +# this module stays importable without CUDA or NVML installed. -CUDA_BINDINGS_NVML_IS_COMPATIBLE: bool +# Always True: kept for callers that read it before the cuda-bindings floor +# made NVML support unconditional. Deprecated. +CUDA_BINDINGS_NVML_IS_COMPATIBLE: bool = True # Please keep in sync with the equivalent implementation in @@ -36,23 +39,8 @@ else: c_locale_guard = None -try: - from cuda.bindings._version import __version_tuple__ as _BINDINGS_VERSION -except ImportError: - CUDA_BINDINGS_NVML_IS_COMPATIBLE = False -else: - CUDA_BINDINGS_NVML_IS_COMPATIBLE = _BINDINGS_VERSION >= (13, 2, 0) or (_BINDINGS_VERSION[0] == 12 and _BINDINGS_VERSION[1:3] >= (9, 6)) - - -if CUDA_BINDINGS_NVML_IS_COMPATIBLE: - try: - from cuda.bindings import nvml - except ImportError: - CUDA_BINDINGS_NVML_IS_COMPATIBLE = False - - from cuda.core.system._nvml_context import initialize -else: - from cuda.core._utils.cuda_utils import driver, handle_return, runtime +from cuda.bindings import nvml +from cuda.core.system._nvml_context import initialize def get_user_mode_driver_version() -> tuple[int, ...]: @@ -60,7 +48,7 @@ def get_user_mode_driver_version() -> tuple[int, ...]: Get the user-mode (UMD / CUDA) driver version. This is the most commonly needed version when checking CUDA driver - compatibility. It works with all ``cuda-bindings`` versions. + compatibility. Returns ------- @@ -68,11 +56,8 @@ def get_user_mode_driver_version() -> tuple[int, ...]: A 2-tuple ``(MAJOR, MINOR)``, e.g. ``(13, 0)`` for CUDA 13.0. """ cdef int v - if CUDA_BINDINGS_NVML_IS_COMPATIBLE: - initialize() - v = nvml.system_get_cuda_driver_version() - else: - v = handle_return(driver.cuDriverGetVersion()) + initialize() + v = nvml.system_get_cuda_driver_version() return (v // 1000, (v // 10) % 100) @@ -91,10 +76,6 @@ def get_kernel_mode_driver_version() -> tuple[int, ...]: RuntimeError If the NVML library is not available. """ - if not CUDA_BINDINGS_NVML_IS_COMPATIBLE: - raise RuntimeError( - "get_kernel_mode_driver_version requires NVML support" - ) initialize() return tuple(int(x) for x in nvml.system_get_driver_version().split(".")) @@ -108,8 +89,7 @@ def get_nvml_version() -> tuple[int, ...]: version: tuple[int, ...] Tuple of integers representing the NVML version components. """ - if not CUDA_BINDINGS_NVML_IS_COMPATIBLE: - raise RuntimeError("NVML library is not available") + initialize() return tuple(int(v) for v in nvml.system_get_nvml_version().split(".")) @@ -122,8 +102,6 @@ def get_driver_branch() -> str: branch: str The driver branch string (e.g., ``"560"``, ``"open"``, etc.). """ - if not CUDA_BINDINGS_NVML_IS_COMPATIBLE: - raise RuntimeError("NVML library is not available") initialize() return nvml.system_get_driver_branch() @@ -132,11 +110,8 @@ def get_num_devices() -> int: """ Return the number of devices in the system. """ - if CUDA_BINDINGS_NVML_IS_COMPATIBLE: - initialize() - return nvml.device_get_count_v2() - else: - return handle_return(runtime.cudaGetDeviceCount()) + initialize() + return nvml.device_get_count_v2() def get_process_name(pid: int) -> str: diff --git a/cuda_core/cuda/core/system/typing.py b/cuda_core/cuda/core/system/typing.py index 6ef9bcb2bc6..50e3356f8ff 100644 --- a/cuda_core/cuda/core/system/typing.py +++ b/cuda_core/cuda/core/system/typing.py @@ -2,6 +2,8 @@ # # SPDX-License-Identifier: Apache-2.0 +from cuda.bindings import nvml as _nvml +from cuda.bindings._internal._fast_enum import FastEnum as _FastEnum from cuda.core._utils.pycompat import StrEnum __all__ = [ @@ -12,8 +14,10 @@ "ClocksEventReasons", "CoolerControl", "CoolerTarget", + "DeviceArch", "EventType", "FanControlPolicy", + "FieldId", "GpuP2PCapsIndex", "GpuP2PStatus", "GpuTopologyLevel", @@ -320,44 +324,29 @@ class ThermalTarget(StrEnum): ThermalTarget.VCD_OUTLET.__doc__ = "Visual Computing Device Outlet temperature requires visual computing device handle." -# DeviceArch values are derived from cuda.bindings.nvml at definition time, so -# the class can only be defined when nvml is importable. -try: - from cuda.bindings import nvml as _nvml - - try: - from cuda.bindings._internal._fast_enum import FastEnum as _FastEnum - except ImportError: - from enum import IntEnum as _FastEnum - - # This uses FastEnum instead of StrEnum because the ordering of the values is - # meaningful, e.g. Kepler "or later" - class DeviceArch(_FastEnum): - """ - Device architecture. - """ - - KEPLER = int(_nvml.DeviceArch.KEPLER) - MAXWELL = int(_nvml.DeviceArch.MAXWELL) - PASCAL = int(_nvml.DeviceArch.PASCAL) - VOLTA = int(_nvml.DeviceArch.VOLTA) - TURING = int(_nvml.DeviceArch.TURING) - AMPERE = int(_nvml.DeviceArch.AMPERE) - ADA = int(_nvml.DeviceArch.ADA) - HOPPER = int(_nvml.DeviceArch.HOPPER) - BLACKWELL = int(_nvml.DeviceArch.BLACKWELL) - UNKNOWN = int(_nvml.DeviceArch.UNKNOWN) - - __all__.append("DeviceArch") +# DeviceArch values are derived from cuda.bindings.nvml at definition time. +# This uses FastEnum instead of StrEnum because the ordering of the values is +# meaningful, e.g. Kepler "or later" +class DeviceArch(_FastEnum): + """ + Device architecture. + """ - FieldId = _nvml.FieldId + KEPLER = int(_nvml.DeviceArch.KEPLER) + MAXWELL = int(_nvml.DeviceArch.MAXWELL) + PASCAL = int(_nvml.DeviceArch.PASCAL) + VOLTA = int(_nvml.DeviceArch.VOLTA) + TURING = int(_nvml.DeviceArch.TURING) + AMPERE = int(_nvml.DeviceArch.AMPERE) + ADA = int(_nvml.DeviceArch.ADA) + HOPPER = int(_nvml.DeviceArch.HOPPER) + BLACKWELL = int(_nvml.DeviceArch.BLACKWELL) + UNKNOWN = int(_nvml.DeviceArch.UNKNOWN) - __all__.append("FieldId") - del _nvml, _FastEnum +FieldId = _nvml.FieldId -except ImportError: - pass +del _nvml, _FastEnum del StrEnum diff --git a/cuda_core/docs/source/release/1.3.0-notes.rst b/cuda_core/docs/source/release/1.3.0-notes.rst index b0fa9aa4624..9d47317c4f7 100644 --- a/cuda_core/docs/source/release/1.3.0-notes.rst +++ b/cuda_core/docs/source/release/1.3.0-notes.rst @@ -20,6 +20,18 @@ Breaking Changes missing C functions, silently disabled features, or crashes. (https://github.com/NVIDIA/cuda-python/issues/2783) +- With the ``cuda-bindings`` floor in place, whether a feature is available now depends on the + CUDA driver alone (and on the CUDA major of the ``cuda.core`` build). Checks that also inspected + the ``cuda-bindings`` version are gone, and the error messages they produced with them; error + messages that name a minimum now name a driver version. ``cuda.core.system`` always uses NVML + through ``cuda-bindings``, so ``cuda.core.system.CUDA_BINDINGS_NVML_IS_COMPATIBLE`` is always + ``True`` and is deprecated. :mod:`cuda.core.checkpoint` requires the CUDA 13 build of + ``cuda.core`` (it did in effect before: the CUDA 12 ``cuda-bindings`` lack a type it uses) and + now says so. The C++ layer calls the driver through the entry points ``cuda-bindings`` resolves + rather than through its Cython wrappers, so a driver function that the installed driver lacks + can no longer surface as ``SystemError`` from a C++ call. + (https://github.com/NVIDIA/cuda-python/issues/2783) + New features ------------ diff --git a/cuda_core/tests/graph/test_graph_builder.py b/cuda_core/tests/graph/test_graph_builder.py index b681e19de8a..8754823cac1 100644 --- a/cuda_core/tests/graph/test_graph_builder.py +++ b/cuda_core/tests/graph/test_graph_builder.py @@ -13,9 +13,7 @@ from cuda_python_test_helpers.marks import requires_module, skipif_need_cuda_headers from helpers.graph_kernels import compile_common_kernels, compile_conditional_kernels from helpers.misc import try_create_condition -from packaging.version import Version -import cuda.bindings from cuda.core import Device, LaunchConfig, LegacyPinnedMemoryResource, Program, ProgramOptions, StreamOptions, launch from cuda.core.graph import Graph, GraphBuilder, GraphCompleteOptions, GraphDefinition from cuda.core.graph._graph_builder import ( @@ -33,10 +31,10 @@ def _wait_until(predicate, timeout=5.0): def _skip_if_conditional_handles_unsupported(): - from cuda.core._utils.version import binding_version, driver_version + from cuda.core._utils.version import driver_version - if driver_version() < (12, 3, 0) or binding_version() < (12, 3, 0): - pytest.skip("conditional handles require CUDA driver and bindings 12.3+") + if driver_version() < (12, 3, 0): + pytest.skip("conditional handles require CUDA driver 12.3+") def test_graph_is_building(init_cuda): @@ -781,13 +779,6 @@ def _assert_programmatic_dependency_edge(graph_definition): """ from cuda.bindings import driver - # cuda.bindings before 13.3.0 (before 12.9.7 on the 12.x branch) returned - # CUgraphEdgeData wrappers backed by a scratch buffer that was freed before the - # call returned, so every field reads back as freed heap memory (#1804). - version = Version(cuda.bindings.__version__) - if version < Version("13.3.0" if version.major >= 13 else "12.9.7"): - pytest.skip(f"cuda.bindings {version} returns dangling graph edge data (#1804)") - h_graph = graph_definition.handle if driver.CUDA_VERSION >= 13000: get_edges = driver.cuGraphGetEdges diff --git a/cuda_core/tests/graph/test_graph_definition.py b/cuda_core/tests/graph/test_graph_definition.py index 4444bafc5de..551b2fa5a3e 100644 --- a/cuda_core/tests/graph/test_graph_definition.py +++ b/cuda_core/tests/graph/test_graph_definition.py @@ -54,18 +54,18 @@ def _skip_if_no_managed_mempool(): def _has_node_get_params(): - from cuda.core._utils.version import binding_version, driver_version + from cuda.core._utils.version import BUILD_CUDA_MAJOR, driver_version - return driver_version() >= (13, 2, 0) and binding_version() >= (13, 2, 0) + return BUILD_CUDA_MAJOR >= 13 and driver_version() >= (13, 2, 0) _HAS_NODE_GET_PARAMS = _has_node_get_params() def _bindings_major_version(): - from cuda.core._utils.version import binding_version + from cuda.core._utils.version import BUILD_CUDA_MAJOR - return binding_version()[0] + return BUILD_CUDA_MAJOR _BINDINGS_MAJOR = _bindings_major_version() @@ -471,7 +471,7 @@ def _build_switch_node(g): pytest.param( NodeSpec("alloc_managed", AllocNode, "CU_GRAPH_NODE_TYPE_MEM_ALLOC", _build_alloc_managed_node), id="alloc_managed", - marks=pytest.mark.skipif(_BINDINGS_MAJOR < 13, reason="managed alloc requires CUDA 13.0+ bindings"), + marks=pytest.mark.skipif(_BINDINGS_MAJOR < 13, reason="managed alloc requires the CUDA 13 build"), ), pytest.param(NodeSpec("free", FreeNode, "CU_GRAPH_NODE_TYPE_MEM_FREE", _build_free_node), id="free"), pytest.param(NodeSpec("memset", MemsetNode, "CU_GRAPH_NODE_TYPE_MEMSET", _build_memset_node), id="memset"), diff --git a/cuda_core/tests/memory/test_copy_batch_options.py b/cuda_core/tests/memory/test_copy_batch_options.py index 85dbe65e3c4..b6fbdc8b0df 100644 --- a/cuda_core/tests/memory/test_copy_batch_options.py +++ b/cuda_core/tests/memory/test_copy_batch_options.py @@ -25,7 +25,7 @@ _normalize_copy_options, ) from cuda.core._stream import PER_THREAD_DEFAULT_STREAM -from cuda.core._utils.version import binding_version, driver_version +from cuda.core._utils.version import BUILD_CUDA_MAJOR, driver_version from cuda.core.utils import ( CopyOptions, MemcpyOverlapMode, @@ -36,7 +36,7 @@ def _batch_native_available(): """True when copy_batch will actually use cuMemcpyBatchAsync.""" - return binding_version() >= (13, 0, 0) and driver_version() >= (13, 0, 0) + return BUILD_CUDA_MAJOR >= 13 and driver_version() >= (13, 0, 0) class TestOptionsEncoding: diff --git a/cuda_core/tests/memory/test_copy_single_options.py b/cuda_core/tests/memory/test_copy_single_options.py index c1ce9024291..3dc17ebd4a6 100644 --- a/cuda_core/tests/memory/test_copy_single_options.py +++ b/cuda_core/tests/memory/test_copy_single_options.py @@ -10,7 +10,7 @@ from cuda.core import Device, Host, LegacyPinnedMemoryResource from cuda.core._stream import LEGACY_DEFAULT_STREAM, PER_THREAD_DEFAULT_STREAM -from cuda.core._utils.version import binding_version, driver_version +from cuda.core._utils.version import BUILD_CUDA_MAJOR, driver_version from cuda.core.utils import CopyOptions, MemcpyOverlapMode, MemcpySrcAccessOrder SIZE = 4096 @@ -19,12 +19,12 @@ def _options_honored(): """True when cuMemcpyWithAttributesAsync will actually be used for options. - Mirrors _with_attributes_available() in _buffer.pyx. CI runs a matrix - that includes pre-CUDA-13.2 driver/bindings combinations (see + Mirrors _with_attributes_available() in _copy_attributes.pxd. CI runs a + matrix that includes CUDA 12 builds and pre-13.2 drivers (see ci/test-matrix.yml), where this is False and the DURING_API_CALL tests below must expect a RuntimeError instead of a successful copy. """ - return driver_version() >= (13, 2, 0) and binding_version() >= (13, 2, 0) + return BUILD_CUDA_MAJOR >= 13 and driver_version() >= (13, 2, 0) @pytest.fixture diff --git a/cuda_core/tests/memory/test_managed_ops.py b/cuda_core/tests/memory/test_managed_ops.py index 16e6a8f181f..5fa44f3f43a 100644 --- a/cuda_core/tests/memory/test_managed_ops.py +++ b/cuda_core/tests/memory/test_managed_ops.py @@ -7,9 +7,8 @@ from helpers.buffers import DummyDeviceMemoryResource, DummyUnifiedMemoryResource from helpers.memory import create_managed_memory_resource_or_skip -from cuda.bindings import driver from cuda.core import Device, Host, ManagedBuffer -from cuda.core._utils.version import binding_version, driver_version +from cuda.core._utils.version import BUILD_CUDA_MAJOR, driver_version # Managed-memory prefetch and CU_MEM_RANGE_ATTRIBUTE_LAST_PREFETCH_LOCATION # operate at physical-page granularity. Test buffers must each occupy a full @@ -52,8 +51,8 @@ def _skip_if_managed_location_ops_unsupported(device): def _skip_if_managed_discard_prefetch_unsupported(device): _skip_if_managed_location_ops_unsupported(device) - if not hasattr(driver, "cuMemDiscardAndPrefetchBatchAsync"): - pytest.skip("discard-prefetch requires cuda.bindings support") + if BUILD_CUDA_MAJOR < 13: + pytest.skip("discard-prefetch requires the CUDA 13 build") visible_devices = Device.get_all_devices() if not all(dev.properties.concurrent_managed_access for dev in visible_devices): @@ -161,20 +160,18 @@ def test_host_passthrough(self): def test_host_numa_passthrough(self): from cuda.core._memory._managed_location import _coerce_location - from cuda.core._utils.version import binding_version - if binding_version() < (13, 0, 0): - pytest.skip("Host(numa_id=N) requires CUDA 13 bindings") + if BUILD_CUDA_MAJOR < 13: + pytest.skip("Host(numa_id=N) requires the CUDA 13 build") spec = _coerce_location(Host(numa_id=3)) assert spec.kind == "host_numa" assert spec.id == 3 def test_host_numa_current_passthrough(self): from cuda.core._memory._managed_location import _coerce_location - from cuda.core._utils.version import binding_version - if binding_version() < (13, 0, 0): - pytest.skip("Host.numa_current() requires CUDA 13 bindings") + if BUILD_CUDA_MAJOR < 13: + pytest.skip("Host.numa_current() requires the CUDA 13 build") spec = _coerce_location(Host.numa_current()) assert spec.kind == "host_numa_current" @@ -246,8 +243,8 @@ class TestDiscardBatch: def test_basic(self, location_ops_device, location_ops_mr): from cuda.core.utils import discard_batch, prefetch_batch - if not hasattr(driver, "cuMemDiscardBatchAsync"): - pytest.skip("cuMemDiscardBatchAsync unavailable") + if BUILD_CUDA_MAJOR < 13: + pytest.skip("cuMemDiscardBatchAsync requires the CUDA 13 build") device = location_ops_device stream = device.create_stream() bufs = [location_ops_mr.allocate(_MANAGED_TEST_ALLOCATION_SIZE, stream=stream) for _ in range(3)] @@ -265,8 +262,8 @@ class TestDiscardPrefetchBatch: def test_same_location(self, location_ops_device, location_ops_mr): from cuda.core.utils import discard_prefetch_batch, prefetch_batch - if not hasattr(driver, "cuMemDiscardAndPrefetchBatchAsync"): - pytest.skip("cuMemDiscardAndPrefetchBatchAsync unavailable") + if BUILD_CUDA_MAJOR < 13: + pytest.skip("cuMemDiscardAndPrefetchBatchAsync requires the CUDA 13 build") device = location_ops_device stream = device.create_stream() bufs = [location_ops_mr.allocate(_MANAGED_TEST_ALLOCATION_SIZE, stream=stream) for _ in range(2)] @@ -360,8 +357,8 @@ def test_last_prefetch_location_initially_none(self, external_managed_buffer): assert external_managed_buffer.last_prefetch_location is None @pytest.mark.skipif( - binding_version() < (13, 0, 0) or driver_version() < (13, 0, 0), - reason="Host NUMA last-prefetch location requires CUDA 13", + BUILD_CUDA_MAJOR < 13 or driver_version() < (13, 0, 0), + reason="Host NUMA last-prefetch location requires the CUDA 13 build and driver", ) @pytest.mark.agent_authored(model="gpt-5") def test_last_prefetch_location_roundtrip_host_numa(self, location_ops_device, managed_buffer): @@ -398,10 +395,8 @@ def test_preferred_location_roundtrip(self, location_ops_device, external_manage @pytest.mark.thread_unsafe(reason="external_managed_buffer is shared between threads") def test_preferred_location_roundtrip_host_numa(self, location_ops_device): """Host(numa_id=N) round-trips correctly on CUDA 13 builds.""" - from cuda.core._utils.version import binding_version - - if binding_version() < (13, 0, 0): - pytest.skip("Host(numa_id=N) round-trip requires CUDA 13 bindings") + if BUILD_CUDA_MAJOR < 13: + pytest.skip("Host(numa_id=N) round-trip requires the CUDA 13 build") plain = DummyUnifiedMemoryResource(location_ops_device).allocate(_MANAGED_TEST_ALLOCATION_SIZE) try: buf = ManagedBuffer.from_handle(plain.handle, plain.size, owner=plain) @@ -486,8 +481,8 @@ def test_instance_prefetch(self, location_ops_device, managed_buffer): assert buf.last_prefetch_location == device def test_instance_discard(self, location_ops_device, managed_buffer): - if not hasattr(driver, "cuMemDiscardBatchAsync"): - pytest.skip("cuMemDiscardBatchAsync unavailable") + if BUILD_CUDA_MAJOR < 13: + pytest.skip("cuMemDiscardBatchAsync requires the CUDA 13 build") device = location_ops_device buf = managed_buffer stream = device.create_stream() diff --git a/cuda_core/tests/system/test_system_device.py b/cuda_core/tests/system/test_system_device.py index 03d92a98a20..4fb97472986 100644 --- a/cuda_core/tests/system/test_system_device.py +++ b/cuda_core/tests/system/test_system_device.py @@ -15,19 +15,16 @@ import helpers import pytest +from cuda.bindings import nvml +from cuda.bindings.nvml import DeviceArch from cuda.core import Device as CudaDevice from cuda.core import system -from cuda.core.system import typing - -if system.CUDA_BINDINGS_NVML_IS_COMPATIBLE: - from cuda.bindings import nvml - from cuda.bindings.nvml import DeviceArch - from cuda.core.system import _device +from cuda.core.system import _device, typing @pytest.fixture(autouse=True, scope="module") def check_gpu_available(): - if not system.CUDA_BINDINGS_NVML_IS_COMPATIBLE or system.get_num_devices() == 0: + if system.get_num_devices() == 0: pytest.skip("No GPUs available to run device tests", allow_module_level=True) diff --git a/cuda_core/tests/system/test_system_events.py b/cuda_core/tests/system/test_system_events.py index 9d02c470188..e7a41a1da59 100644 --- a/cuda_core/tests/system/test_system_events.py +++ b/cuda_core/tests/system/test_system_events.py @@ -10,12 +10,10 @@ import helpers import pytest +from cuda.bindings import nvml from cuda.core import system from cuda.core.system import typing - -if system.CUDA_BINDINGS_NVML_IS_COMPATIBLE: - from cuda.bindings import nvml - from cuda.core.system._system_events import SystemEvent, SystemEvents, _pci_bus_id_from_gpu_id +from cuda.core.system._system_events import SystemEvent, SystemEvents, _pci_bus_id_from_gpu_id @pytest.mark.agent_authored(model="claude-opus-4.7") diff --git a/cuda_core/tests/system/test_system_system.py b/cuda_core/tests/system/test_system_system.py index 9fc600b05da..fca171a28b0 100644 --- a/cuda_core/tests/system/test_system_system.py +++ b/cuda_core/tests/system/test_system_system.py @@ -35,13 +35,6 @@ def test_kernel_mode_driver_version(): assert 0 <= ver_patch[0] <= 99 -def test_kernel_mode_driver_version_requires_nvml(): - if system.CUDA_BINDINGS_NVML_IS_COMPATIBLE: - pytest.skip("NVML is available, cannot test the error path") - with pytest.raises(RuntimeError, match="requires NVML support"): - system.get_kernel_mode_driver_version() - - @skip_if_nvml_unsupported def test_nvml_version(): nvml_version = system.get_nvml_version() diff --git a/cuda_core/tests/test_checkpoint.py b/cuda_core/tests/test_checkpoint.py index ff727eb9fff..a9aa1726f42 100644 --- a/cuda_core/tests/test_checkpoint.py +++ b/cuda_core/tests/test_checkpoint.py @@ -40,18 +40,11 @@ def _checkpoint_available(): def _checkpoint_unavailable_can_skip(message): - if message.startswith( + return message.startswith( ( "CUDA checkpointing is not supported by the installed NVIDIA driver.", - "CUDA checkpointing requires cuda.bindings with CUDA checkpoint API support. Found cuda.bindings ", + "CUDA checkpointing requires the CUDA 13 build of cuda.core", ) - ): - return True - - return ( - checkpoint._binding_version()[0] == 12 - and message - == "CUDA checkpointing requires cuda.bindings with CUDA checkpoint API support. Missing: CUcheckpointGpuPair" ) diff --git a/cuda_core/tests/test_cuda_utils.py b/cuda_core/tests/test_cuda_utils.py index 32ea504248d..3f8268b71f3 100644 --- a/cuda_core/tests/test_cuda_utils.py +++ b/cuda_core/tests/test_cuda_utils.py @@ -11,14 +11,6 @@ from cuda.core._utils.clear_error_support import assert_type_str_or_bytes_like, raise_code_path_meant_to_be_unreachable -def _skip_if_bindings_pre_enum_docstrings(): - from cuda.core._utils.enum_explanations_helpers import _binding_version_has_usable_enum_docstrings - from cuda.core._utils.version import binding_version - - if not _binding_version_has_usable_enum_docstrings(binding_version()): - pytest.skip("cuda-bindings version does not expose usable enum __doc__ strings") - - def _assert_cleanup_example_matches_or_xfail(actual, expected): # Pin a few real cleanup-sensitive enum docs. If one starts failing, review # the raw ``__doc__`` and today's cleaned output: either update the expected @@ -59,7 +51,6 @@ def test_check_runtime_error(): def test_driver_error_enum_has_non_empty_docstring(): - _skip_if_bindings_pre_enum_docstrings() doc = driver.CUresult.CUDA_ERROR_INVALID_VALUE.__doc__ assert doc is not None @@ -67,7 +58,6 @@ def test_driver_error_enum_has_non_empty_docstring(): def test_runtime_error_enum_has_non_empty_docstring(): - _skip_if_bindings_pre_enum_docstrings() doc = runtime.cudaError_t.cudaErrorInvalidValue.__doc__ assert doc is not None @@ -131,7 +121,6 @@ def test_runtime_error_enum_has_non_empty_docstring(): ], ) def test_enum_doc_cleanup_examples_are_reviewed_on_change(explanations, error, expected): - _skip_if_bindings_pre_enum_docstrings() actual = explanations.get(int(error)) _assert_cleanup_example_matches_or_xfail(actual, expected) diff --git a/cuda_core/tests/test_device.py b/cuda_core/tests/test_device.py index 6b3aca9dc73..de72888f757 100644 --- a/cuda_core/tests/test_device.py +++ b/cuda_core/tests/test_device.py @@ -27,15 +27,9 @@ def test_device_init_disabled(): def test_to_system_device(deinit_cuda): - from cuda.core.system import _system device = Device() - if not _system.CUDA_BINDINGS_NVML_IS_COMPATIBLE: - with pytest.raises(RuntimeError): - device.to_system_device() - pytest.skip("NVML support requires cuda.bindings version 12.9.6+ for CUDA 12.x or 13.2.0+ for CUDA 13.x") - from cuda_python_test_helpers.arch_check import hardware_supports_nvml if not hardware_supports_nvml(): diff --git a/cuda_core/tests/test_enum_coverage.py b/cuda_core/tests/test_enum_coverage.py index 119554c226f..d38a5b15d00 100644 --- a/cuda_core/tests/test_enum_coverage.py +++ b/cuda_core/tests/test_enum_coverage.py @@ -14,9 +14,11 @@ import pytest import cuda.core +import cuda.core.system.typing as system_typing import cuda.core.typing -from cuda.bindings import driver -from cuda.core import system +from cuda.bindings import driver, nvml +from cuda.core._utils.version import BUILD_CUDA_MAJOR +from cuda.core.system import _device, _system_events if sys.version_info >= (3, 11): from enum import StrEnum @@ -123,175 +125,168 @@ ), ] -if system.CUDA_BINDINGS_NVML_IS_COMPATIBLE: - # Populated below only when NVML bindings are compatible, so that importing - # this module on an incompatible host does not raise ImportError. - import cuda.core.system.typing as system_typing - from cuda.bindings import nvml - from cuda.core.system import _device, _system_events +_MODULES.append(system_typing) - _MODULES.append(system_typing) - - _CLOCKS_EVENT_REASONS_STR_UNMAPPED = { - core_member - for binding_member, core_member in ( - ("EVENT_REASON_BOARD_LIMIT", "BOARD_LIMIT"), - ("EVENT_REASON_RELIABILITY", "RELIABILITY"), - ) - if binding_member not in nvml.ClocksEventReasons.__members__ - } - - _CASES.extend( - [ - ( - nvml.DeviceAddressingModeType, - system_typing.AddressingMode, - _device._ADDRESSING_MODE_MAPPING, - # NONE means "no special addressing mode is active"; not a valid target - {"DEVICE_ADDRESSING_MODE_NONE"}, - set(), - ), - ( - nvml.BrandType, - None, # maps to plain str, not a StrEnum - _device._BRAND_TYPE_MAPPING, - # COUNT is a sentinel, not a real brand - {"BRAND_COUNT"}, - set(), - ), - ( - nvml.GpuP2PStatus, - system_typing.GpuP2PStatus, - _device._GPU_P2P_STATUS_MAPPING, - # Both the typo'd (SUPPORED) and corrected (SUPPORTED) spellings - # share the same integer value; the mapping covers both via aliases - {"P2P_STATUS_CHIPSET_NOT_SUPPORED"}, - set(), - ), - ( - nvml.ClocksEventReasons, - system_typing.ClocksEventReasons, - _device._CLOCKS_EVENT_REASONS_MAPPING, - set(), - _CLOCKS_EVENT_REASONS_STR_UNMAPPED, - ), - ( - nvml.EventType, - system_typing.EventType, - _device._EVENT_TYPE_MAPPING, - set(), - set(), - ), - ( - nvml.FanControlPolicy, - system_typing.FanControlPolicy, - _device._FAN_CONTROL_POLICY_MAPPING, - set(), - set(), - ), - ( - nvml.CoolerControl, - system_typing.CoolerControl, - _device._COOLER_CONTROL_MAPPING, - # NONE means no signal; COUNT is a sentinel - {"THERMAL_COOLER_SIGNAL_NONE", "THERMAL_COOLER_SIGNAL_COUNT"}, - set(), - ), - ( - nvml.CoolerTarget, - system_typing.CoolerTarget, - _device._COOLER_TARGET_MAPPING, - # GPU_RELATED is a composite bitmask (GPU | MEMORY | POWER_SUPPLY); - # the wrapper expands it into individual targets instead of mapping - # it as a single entry - {"THERMAL_GPU_RELATED"}, - set(), - ), - ( - nvml.ThermalController, - system_typing.ThermalController, - _device._THERMAL_CONTROLLER_MAPPING, - {"NONE"}, - {"NONE"}, - ), - ( - nvml.ThermalTarget, - system_typing.ThermalTarget, - _device._THERMAL_TARGET_MAPPING, - # UNKNOWN is a fallback sentinel; handled by .get() - {"UNKNOWN"}, - set(), - ), - ( - nvml.NvlinkVersion, - None, # maps to tuple, not a StrEnum - _device._NVLINK_VERSION_MAPPING, - # VERSION_INVALID is a sentinel for "no NvLink present" - {"VERSION_INVALID"}, - set(), - ), - ( - nvml.SystemEventType, - system_typing.SystemEventType, - _system_events._SYSTEM_EVENT_TYPE_MAPPING, - set(), - set(), - ), - ( - nvml.AffinityScope, - system_typing.AffinityScope, - _device._AFFINITY_SCOPE_MAPPING, - set(), - set(), - ), - ( - nvml.GpuP2PCapsIndex, - system_typing.GpuP2PCapsIndex, - _device._GPU_P2P_CAPS_INDEX_MAPPING, - set(), - set(), - ), - ( - nvml.GpuTopologyLevel, - system_typing.GpuTopologyLevel, - _device._GPU_TOPOLOGY_LEVEL_MAPPING, - set(), - set(), - ), - ( - nvml.ClockId, - system_typing.ClockId, - _device._CLOCK_ID_MAPPING, - # APP_CLOCK_TARGET and APP_CLOCK_DEFAULT are deprecated; COUNT is a sentinel - {"APP_CLOCK_TARGET", "APP_CLOCK_DEFAULT", "COUNT"}, - set(), - ), - ( - nvml.ClockType, - system_typing.ClockType, - _device._CLOCK_TYPE_MAPPING, - # COUNT is a sentinel - {"CLOCK_COUNT"}, - set(), - ), - ( - nvml.InforomObject, - system_typing.InforomObject, - _device._INFOROM_OBJECT_MAPPING, - # COUNT is a sentinel - {"INFOROM_COUNT"}, - set(), - ), - ( - nvml.TemperatureThresholds, - system_typing.TemperatureThresholds, - _device._TEMPERATURE_THRESHOLD_MAPPING, - # COUNT is a sentinel - {"TEMPERATURE_THRESHOLD_COUNT"}, - set(), - ), - ] +_CLOCKS_EVENT_REASONS_STR_UNMAPPED = { + core_member + for binding_member, core_member in ( + ("EVENT_REASON_BOARD_LIMIT", "BOARD_LIMIT"), + ("EVENT_REASON_RELIABILITY", "RELIABILITY"), ) + if binding_member not in nvml.ClocksEventReasons.__members__ +} + +_CASES.extend( + [ + ( + nvml.DeviceAddressingModeType, + system_typing.AddressingMode, + _device._ADDRESSING_MODE_MAPPING, + # NONE means "no special addressing mode is active"; not a valid target + {"DEVICE_ADDRESSING_MODE_NONE"}, + set(), + ), + ( + nvml.BrandType, + None, # maps to plain str, not a StrEnum + _device._BRAND_TYPE_MAPPING, + # COUNT is a sentinel, not a real brand + {"BRAND_COUNT"}, + set(), + ), + ( + nvml.GpuP2PStatus, + system_typing.GpuP2PStatus, + _device._GPU_P2P_STATUS_MAPPING, + # Both the typo'd (SUPPORED) and corrected (SUPPORTED) spellings + # share the same integer value; the mapping covers both via aliases + {"P2P_STATUS_CHIPSET_NOT_SUPPORED"}, + set(), + ), + ( + nvml.ClocksEventReasons, + system_typing.ClocksEventReasons, + _device._CLOCKS_EVENT_REASONS_MAPPING, + set(), + _CLOCKS_EVENT_REASONS_STR_UNMAPPED, + ), + ( + nvml.EventType, + system_typing.EventType, + _device._EVENT_TYPE_MAPPING, + set(), + set(), + ), + ( + nvml.FanControlPolicy, + system_typing.FanControlPolicy, + _device._FAN_CONTROL_POLICY_MAPPING, + set(), + set(), + ), + ( + nvml.CoolerControl, + system_typing.CoolerControl, + _device._COOLER_CONTROL_MAPPING, + # NONE means no signal; COUNT is a sentinel + {"THERMAL_COOLER_SIGNAL_NONE", "THERMAL_COOLER_SIGNAL_COUNT"}, + set(), + ), + ( + nvml.CoolerTarget, + system_typing.CoolerTarget, + _device._COOLER_TARGET_MAPPING, + # GPU_RELATED is a composite bitmask (GPU | MEMORY | POWER_SUPPLY); + # the wrapper expands it into individual targets instead of mapping + # it as a single entry + {"THERMAL_GPU_RELATED"}, + set(), + ), + ( + nvml.ThermalController, + system_typing.ThermalController, + _device._THERMAL_CONTROLLER_MAPPING, + {"NONE"}, + {"NONE"}, + ), + ( + nvml.ThermalTarget, + system_typing.ThermalTarget, + _device._THERMAL_TARGET_MAPPING, + # UNKNOWN is a fallback sentinel; handled by .get() + {"UNKNOWN"}, + set(), + ), + ( + nvml.NvlinkVersion, + None, # maps to tuple, not a StrEnum + _device._NVLINK_VERSION_MAPPING, + # VERSION_INVALID is a sentinel for "no NvLink present" + {"VERSION_INVALID"}, + set(), + ), + ( + nvml.SystemEventType, + system_typing.SystemEventType, + _system_events._SYSTEM_EVENT_TYPE_MAPPING, + set(), + set(), + ), + ( + nvml.AffinityScope, + system_typing.AffinityScope, + _device._AFFINITY_SCOPE_MAPPING, + set(), + set(), + ), + ( + nvml.GpuP2PCapsIndex, + system_typing.GpuP2PCapsIndex, + _device._GPU_P2P_CAPS_INDEX_MAPPING, + set(), + set(), + ), + ( + nvml.GpuTopologyLevel, + system_typing.GpuTopologyLevel, + _device._GPU_TOPOLOGY_LEVEL_MAPPING, + set(), + set(), + ), + ( + nvml.ClockId, + system_typing.ClockId, + _device._CLOCK_ID_MAPPING, + # APP_CLOCK_TARGET and APP_CLOCK_DEFAULT are deprecated; COUNT is a sentinel + {"APP_CLOCK_TARGET", "APP_CLOCK_DEFAULT", "COUNT"}, + set(), + ), + ( + nvml.ClockType, + system_typing.ClockType, + _device._CLOCK_TYPE_MAPPING, + # COUNT is a sentinel + {"CLOCK_COUNT"}, + set(), + ), + ( + nvml.InforomObject, + system_typing.InforomObject, + _device._INFOROM_OBJECT_MAPPING, + # COUNT is a sentinel + {"INFOROM_COUNT"}, + set(), + ), + ( + nvml.TemperatureThresholds, + system_typing.TemperatureThresholds, + _device._TEMPERATURE_THRESHOLD_MAPPING, + # COUNT is a sentinel + {"TEMPERATURE_THRESHOLD_COUNT"}, + set(), + ), + ] +) # StrEnum subclasses that intentionally have no associated cuda_binding. @@ -323,11 +318,10 @@ } -# CUdevWorkqueueConfigScope was added to the CUDA driver in 13.1 (missing -# from the 13.0.0 cuda.h and earlier); on cuda-bindings for CUDA 12.x or -# 13.0.x, WorkqueueSharingScopeType has no driver-side counterpart to +# CUdevWorkqueueConfigScope was added to the CUDA driver in 13.1; on the +# CUDA 12 build, WorkqueueSharingScopeType has no driver-side counterpart to # check against. -if hasattr(driver, "CUdevWorkqueueConfigScope"): +if BUILD_CUDA_MAJOR >= 13: _CASES.append( ( driver.CUdevWorkqueueConfigScope, diff --git a/cuda_core/tests/test_error_handling.py b/cuda_core/tests/test_error_handling.py index c718a0f1086..1ff3ea2b461 100644 --- a/cuda_core/tests/test_error_handling.py +++ b/cuda_core/tests/test_error_handling.py @@ -36,7 +36,7 @@ ) from cuda.core._stream import default_stream from cuda.core._utils.cuda_utils import CUDAError, driver, handle_return -from cuda.core._utils.version import binding_version, driver_version +from cuda.core._utils.version import BUILD_CUDA_MAJOR, driver_version from cuda.core.graph import GraphDefinition INVALID_CONTEXT = int(driver.CUresult.CUDA_ERROR_INVALID_CONTEXT) @@ -254,7 +254,7 @@ def test_set_current_with_context_works_without_a_current_context(init_cuda): def test_memset_update_keeps_new_owners_alive_when_context_cannot_be_restored(device_x2): """The node's new parameters stay valid: the attachment is published before the restoration failure is raised, so the updated graph instantiates and runs.""" - if driver_version() < (13, 2, 0) or binding_version() < (13, 2, 0): + if BUILD_CUDA_MAJOR < 13 or driver_version() < (13, 2, 0): pytest.skip("node contexts are only recorded by cuGraphNodeGetParams on CUDA 13.2+") node_dev, other_dev = device_x2 node_dev.set_current() diff --git a/cuda_core/tests/test_green_context.py b/cuda_core/tests/test_green_context.py index 5f1954c6b58..bf336f8e0f2 100644 --- a/cuda_core/tests/test_green_context.py +++ b/cuda_core/tests/test_green_context.py @@ -20,7 +20,7 @@ launch, ) from cuda.core._utils.cuda_utils import CUDAError, driver, handle_return -from cuda.core._utils.version import binding_version, driver_version +from cuda.core._utils.version import BUILD_CUDA_MAJOR, driver_version from cuda.core.graph import GraphDefinition from cuda.core.typing import WorkqueueSharingScopeType @@ -155,8 +155,8 @@ def test_memory_node_updates_preserve_green_context( init_cuda, green_ctx, ): - if driver_version() < (13, 2, 0) or binding_version() < (13, 2, 0): - pytest.skip("generic graph node parameter queries require CUDA 13.2+") + if BUILD_CUDA_MAJOR < 13 or driver_version() < (13, 2, 0): + pytest.skip("generic graph node parameter queries require the CUDA 13 build and driver 13.2+") memory_resource = LegacyPinnedMemoryResource() src = memory_resource.allocate(4) @@ -408,7 +408,7 @@ def test_discovery_mode(self, sm_resource): @pytest.mark.agent_authored(model="gpt-5.6-sol") def test_by_count_discovery_respects_alignment(self, sm_resource): """CUDA 12 SplitByCount discovery returns an aligned SM count.""" - if binding_version()[0] != 12: + if BUILD_CUDA_MAJOR != 12: pytest.skip("test covers the CUDA 12 SplitByCount path") groups, _ = sm_resource.split(SMResourceOptions(count=None)) diff --git a/cuda_core/tests/test_launcher.py b/cuda_core/tests/test_launcher.py index bd798424f70..d5ee2262883 100644 --- a/cuda_core/tests/test_launcher.py +++ b/cuda_core/tests/test_launcher.py @@ -1013,9 +1013,6 @@ def test_launch_graph_conditional_handle_as_kernel_arg(init_cuda, use_subclass): """CUgraphConditionalHandle is packed as its uint64 value (readback).""" from cuda.bindings import driver - if not hasattr(driver, "CUgraphConditionalHandle"): - pytest.skip("CUgraphConditionalHandle requires cuda-bindings 12.3+") - class SubclassedHandle(driver.CUgraphConditionalHandle): pass diff --git a/cuda_core/tests/test_linker.py b/cuda_core/tests/test_linker.py index c80071fa405..45444ab4b4d 100644 --- a/cuda_core/tests/test_linker.py +++ b/cuda_core/tests/test_linker.py @@ -326,12 +326,6 @@ def test_which_backend_falls_back_when_nvjitlink_too_old(self, monkeypatch): monkeypatch.setattr(_linker, "_use_nvjitlink_backend", None) monkeypatch.setattr(_linker, "_driver", None) - def fake__optional_cuda_import(modname, probe_function=None): - assert modname == "cuda.bindings.nvjitlink" - assert probe_function is None - return object() - - monkeypatch.setattr(_linker, "_optional_cuda_import", fake__optional_cuda_import) monkeypatch.setattr(_linker, "_nvjitlink_has_version_symbol", lambda _nvjitlink: False) with pytest.warns(RuntimeWarning, match="too old \\(<12.3\\)"): @@ -350,13 +344,7 @@ def test_which_backend_falls_back_when_dylib_missing(self, monkeypatch): def raise_missing(_nvjitlink): raise DynamicLibNotFoundError("missing") - def fake__optional_cuda_import(modname, probe_function=None): - assert modname == "cuda.bindings.nvjitlink" - assert probe_function is None - return object() - monkeypatch.setattr(_linker, "_nvjitlink_has_version_symbol", raise_missing) - monkeypatch.setattr(_linker, "_optional_cuda_import", fake__optional_cuda_import) with pytest.warns(RuntimeWarning, match="cuda.bindings.nvjitlink is not available"): assert Linker.which_backend() == "driver" diff --git a/cuda_core/tests/test_module.py b/cuda_core/tests/test_module.py index 3e85101bd85..3b1fe9f7a9d 100644 --- a/cuda_core/tests/test_module.py +++ b/cuda_core/tests/test_module.py @@ -16,7 +16,7 @@ from cuda.core import Device, Kernel, Linker, LinkerOptions, ObjectCode, Program, ProgramOptions from cuda.core._program import _can_load_generated_ptx from cuda.core._utils.cuda_utils import CUDAError, driver, handle_return -from cuda.core._utils.version import binding_version, driver_version +from cuda.core._utils.version import driver_version try: import numba @@ -65,7 +65,7 @@ def _is_nvfatbin_available(): @pytest.fixture(scope="module") def cuda12_4_prerequisite_check(): - return binding_version() >= (12, 0, 0) and driver_version() >= (12, 4, 0) + return driver_version() >= (12, 4, 0) @pytest.fixture(name="convert_path", params=[str, lambda p: p], ids=["str", "path"]) diff --git a/cuda_core/tests/test_optional_dependency_imports.py b/cuda_core/tests/test_optional_dependency_imports.py index b08b7d344d9..68a1779a260 100644 --- a/cuda_core/tests/test_optional_dependency_imports.py +++ b/cuda_core/tests/test_optional_dependency_imports.py @@ -35,28 +35,7 @@ def restore_optional_import_state(): _linker._use_nvjitlink_backend = saved_use_nvjitlink -@pytest.mark.agent_authored(model="gpt-5.6-sol") -def test_get_nvvm_module_rejects_old_bindings(monkeypatch): - """NVVM import requires cuda-bindings >= 12.9.0 and caches a failed attempt.""" - calls = 0 - - def old_binding_version(): - nonlocal calls - calls += 1 - return (12, 8, 0) - - monkeypatch.setattr(_program, "binding_version", old_binding_version) - - with pytest.raises(RuntimeError, match="cuda-bindings >= 12.9.0"): - _program._get_nvvm_module() - with pytest.raises(RuntimeError, match="previous import attempt failed"): - _program._get_nvvm_module() - assert calls == 1 - - def test_get_nvvm_module_reraises_nested_module_not_found(monkeypatch): - monkeypatch.setattr(_program, "binding_version", lambda: (12, 9, 0)) - def fake__optional_cuda_import(modname, probe_function=None): assert modname == "cuda.bindings.nvvm" assert probe_function is not None @@ -72,8 +51,6 @@ def fake__optional_cuda_import(modname, probe_function=None): def test_get_nvvm_module_reports_missing_nvvm_module(monkeypatch): - monkeypatch.setattr(_program, "binding_version", lambda: (12, 9, 0)) - def fake__optional_cuda_import(modname, probe_function=None): assert modname == "cuda.bindings.nvvm" assert probe_function is not None @@ -86,8 +63,6 @@ def fake__optional_cuda_import(modname, probe_function=None): def test_get_nvvm_module_handles_missing_libnvvm(monkeypatch): - monkeypatch.setattr(_program, "binding_version", lambda: (12, 9, 0)) - def fake__optional_cuda_import(modname, probe_function=None): assert modname == "cuda.bindings.nvvm" assert probe_function is not None @@ -99,36 +74,6 @@ def fake__optional_cuda_import(modname, probe_function=None): _program._get_nvvm_module() -def test_decide_nvjitlink_or_driver_reraises_nested_module_not_found(monkeypatch): - def fake__optional_cuda_import(modname, probe_function=None): - assert modname == "cuda.bindings.nvjitlink" - assert probe_function is None - err = ModuleNotFoundError("No module named 'not_a_real_dependency'") - err.name = "not_a_real_dependency" - raise err - - monkeypatch.setattr(_linker, "_optional_cuda_import", fake__optional_cuda_import) - - with pytest.raises(ModuleNotFoundError, match="not_a_real_dependency") as excinfo: - _linker._decide_nvjitlink_or_driver() - assert excinfo.value.name == "not_a_real_dependency" - - -def test_decide_nvjitlink_or_driver_falls_back_when_module_missing(monkeypatch): - def fake__optional_cuda_import(modname, probe_function=None): - assert modname == "cuda.bindings.nvjitlink" - assert probe_function is None - return None - - monkeypatch.setattr(_linker, "_optional_cuda_import", fake__optional_cuda_import) - - with pytest.warns(RuntimeWarning, match="cuda.bindings.nvjitlink is not available"): - use_driver_backend = _linker._decide_nvjitlink_or_driver() - - assert use_driver_backend is True - assert _linker._use_nvjitlink_backend is False - - @pytest.mark.agent_authored(model="grok-4.5") def test_decide_nvjitlink_or_driver_falls_back_when_dylib_missing(monkeypatch): """Missing nvJitLink dylib must fall back via DynamicLibNotFoundError.""" @@ -136,12 +81,6 @@ def test_decide_nvjitlink_or_driver_falls_back_when_dylib_missing(monkeypatch): def raise_missing(_nvjitlink): raise DynamicLibNotFoundError("libnvJitLink missing") - def fake__optional_cuda_import(modname, probe_function=None): - assert modname == "cuda.bindings.nvjitlink" - assert probe_function is None - return object() - - monkeypatch.setattr(_linker, "_optional_cuda_import", fake__optional_cuda_import) monkeypatch.setattr(_linker, "_nvjitlink_has_version_symbol", raise_missing) with pytest.warns(RuntimeWarning, match="cuda.bindings.nvjitlink is not available"): @@ -153,12 +92,6 @@ def fake__optional_cuda_import(modname, probe_function=None): @pytest.mark.agent_authored(model="grok-4.5") def test_decide_nvjitlink_or_driver_falls_back_when_nvjitlink_too_old(monkeypatch): - def fake__optional_cuda_import(modname, probe_function=None): - assert modname == "cuda.bindings.nvjitlink" - assert probe_function is None - return object() - - monkeypatch.setattr(_linker, "_optional_cuda_import", fake__optional_cuda_import) monkeypatch.setattr(_linker, "_nvjitlink_has_version_symbol", lambda _nvjitlink: False) with pytest.warns(RuntimeWarning, match="too old \\(<12.3\\)"): @@ -170,12 +103,6 @@ def fake__optional_cuda_import(modname, probe_function=None): @pytest.mark.agent_authored(model="grok-4.5") def test_decide_nvjitlink_or_driver_selects_nvjitlink_when_version_symbol_present(monkeypatch): - def fake__optional_cuda_import(modname, probe_function=None): - assert modname == "cuda.bindings.nvjitlink" - assert probe_function is None - return object() - - monkeypatch.setattr(_linker, "_optional_cuda_import", fake__optional_cuda_import) monkeypatch.setattr(_linker, "_nvjitlink_has_version_symbol", lambda _nvjitlink: True) use_driver_backend = _linker._decide_nvjitlink_or_driver() @@ -189,22 +116,16 @@ def test_decide_nvjitlink_or_driver_does_not_call_version(monkeypatch): """Regression guard for #2408: must not call module.version().""" called = {"version": False, "inspect": False} - class FakeModule: - def version(self): - called["version"] = True - raise AssertionError("module.version() must not be used for nvJitLink probing") + def fake_version(): + called["version"] = True + raise AssertionError("module.version() must not be used for nvJitLink probing") def fake_has_version(_nvjitlink): called["inspect"] = True return True - def fake__optional_cuda_import(modname, probe_function=None): - assert modname == "cuda.bindings.nvjitlink" - assert probe_function is None - return FakeModule() - - monkeypatch.setattr(_linker, "_optional_cuda_import", fake__optional_cuda_import) monkeypatch.setattr(_linker, "_nvjitlink_has_version_symbol", fake_has_version) + monkeypatch.setattr("cuda.bindings.nvjitlink.version", fake_version) assert _linker._decide_nvjitlink_or_driver() is False assert called["inspect"] is True diff --git a/cuda_core/tests/test_program.py b/cuda_core/tests/test_program.py index 311bbf1875f..b14c313cf20 100644 --- a/cuda_core/tests/test_program.py +++ b/cuda_core/tests/test_program.py @@ -60,19 +60,9 @@ def _get_nvrtc_version_for_tests(): # CUDAError from a successfully loaded library propagates (real bug). -def _has_nvrtc_pch_apis_for_tests(): - required = ( - "nvrtcGetPCHHeapSize", - "nvrtcSetPCHHeapSize", - "nvrtcGetPCHCreateStatus", - "nvrtcGetPCHHeapSizeRequired", - ) - return all(hasattr(nvrtc, name) for name in required) - - nvrtc_pch_available = pytest.mark.skipif( - (_get_nvrtc_version_for_tests() or 0) < 12800 or not _has_nvrtc_pch_apis_for_tests(), - reason="PCH runtime APIs require NVRTC >= 12.8 bindings", + (_get_nvrtc_version_for_tests() or 0) < 12800, + reason="PCH runtime APIs require NVRTC >= 12.8", ) bundled_headers_available = pytest.mark.skipif( diff --git a/cuda_core/tests/test_utils_enum_explanations_helpers.py b/cuda_core/tests/test_utils_enum_explanations_helpers.py index 6d4c9e32b82..0cbb2672de0 100644 --- a/cuda_core/tests/test_utils_enum_explanations_helpers.py +++ b/cuda_core/tests/test_utils_enum_explanations_helpers.py @@ -2,15 +2,10 @@ # # SPDX-License-Identifier: Apache-2.0 -import importlib -import sys - import pytest -from cuda.core._utils import enum_explanations_helpers from cuda.core._utils.enum_explanations_helpers import ( DocstringBackedExplanations, - _binding_version_has_usable_enum_docstrings, clean_enum_member_docstring, ) @@ -90,81 +85,3 @@ def test_docstring_backed_get_returns_default_for_missing_docstring(): lut = DocstringBackedExplanations(_FakeEnumType({7: _FakeEnumMember(None)})) assert lut.get(7) is None assert lut.get(7, default="sentinel") == "sentinel" - - -@pytest.mark.parametrize( - ("version", "expected"), - [ - pytest.param((12, 9, 5), False, id="before_12_9_6"), - pytest.param((12, 9, 6), True, id="from_12_9_6"), - pytest.param((13, 0, 0), False, id="13_0_mainline_gap"), - pytest.param((13, 1, 1), False, id="13_1_1"), - pytest.param((13, 2, 0), True, id="from_13_2_0"), - ], -) -def test_binding_version_has_usable_enum_docstrings(version, expected): - assert _binding_version_has_usable_enum_docstrings(version) is expected - - -@pytest.mark.parametrize( - ("version", "expects_docstrings"), - [ - pytest.param((12, 9, 5), False, id="before_12_9_6"), - pytest.param((12, 9, 6), True, id="from_12_9_6"), - pytest.param((13, 0, 0), False, id="13_0_mainline_gap"), - pytest.param((13, 2, 0), True, id="from_13_2_0"), - ], -) -def test_get_best_available_explanations_switches_by_version(monkeypatch, version, expects_docstrings): - fallback = {7: "fallback text"} - monkeypatch.setattr(enum_explanations_helpers, "_binding_version", lambda: version) - expl = enum_explanations_helpers.get_best_available_explanations( - _FakeEnumType({7: _FakeEnumMember("clean me")}), - fallback, - ) - if expects_docstrings: - assert isinstance(expl, DocstringBackedExplanations) - assert expl.get(7) == "clean me" - else: - assert expl is fallback - - -def test_get_best_available_explanations_calls_loader_before_docstrings(monkeypatch): - fallback = {7: "fallback text"} - calls = [] - - def load_fallback(): - calls.append("loaded") - return fallback - - monkeypatch.setattr(enum_explanations_helpers, "_binding_version", lambda: (13, 1, 1)) - expl = enum_explanations_helpers.get_best_available_explanations( - _FakeEnumType({7: _FakeEnumMember("clean me")}), - load_fallback, - ) - assert expl is fallback - assert calls == ["loaded"] - - -def test_driver_explanations_module_skips_fallback_import_when_docstrings_available(monkeypatch): - import cuda.core._utils.driver_cu_result_explanations as driver_explanations - - monkeypatch.setattr(enum_explanations_helpers, "_binding_version", lambda: (13, 2, 0)) - sys.modules.pop("cuda.core._utils.driver_cu_result_explanations_frozen", None) - - importlib.reload(driver_explanations) - - assert "cuda.core._utils.driver_cu_result_explanations_frozen" not in sys.modules - assert isinstance(driver_explanations.DRIVER_CU_RESULT_EXPLANATIONS, DocstringBackedExplanations) - - -def test_runtime_explanations_module_skips_fallback_import_when_docstrings_available(monkeypatch): - import cuda.core._utils.runtime_cuda_error_explanations as runtime_explanations - - monkeypatch.setattr(enum_explanations_helpers, "_binding_version", lambda: (13, 2, 0)) - sys.modules.pop("cuda.core._utils.runtime_cuda_error_explanations_frozen", None) - - importlib.reload(runtime_explanations) - - assert "cuda.core._utils.runtime_cuda_error_explanations_frozen" not in sys.modules - assert isinstance(runtime_explanations.RUNTIME_CUDA_ERROR_EXPLANATIONS, DocstringBackedExplanations) From 65c63b09ea9385347387b640a296324acc7c7dcb Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Fri, 18 Sep 2026 14:23:34 -0700 Subject: [PATCH 05/18] cuda.core tests: gate the checkpoint helper tests on the CUDA 13 build (#2783) test_checkpoint.py probed cuda.core.checkpoint._REQUIRED_BINDING_ATTRS, which the gate collapse removed, so the whole session failed at collection. The helpers build CUcheckpointGpuPair, a CUDA 13 type; skip them on the CUDA 12 build instead. --- cuda_core/tests/test_checkpoint.py | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/cuda_core/tests/test_checkpoint.py b/cuda_core/tests/test_checkpoint.py index a9aa1726f42..d2ce7491532 100644 --- a/cuda_core/tests/test_checkpoint.py +++ b/cuda_core/tests/test_checkpoint.py @@ -407,14 +407,12 @@ def test_pid_is_read_only(self): import ctypes from cuda.bindings import driver as _bindings_driver +from cuda.core._utils.version import BUILD_CUDA_MAJOR -# The checkpoint functions, structs, and enums are generated and shipped -# together from the same CUDA headers, so probe them as one atomic API surface. -_HAS_CHECKPOINT_BINDINGS = all(hasattr(_bindings_driver, name) for name in checkpoint._REQUIRED_BINDING_ATTRS) - +# The helpers build CUcheckpointGpuPair, a CUDA 13 type the CUDA 12 bindings lack. needs_checkpoint_bindings = pytest.mark.skipif( - not _HAS_CHECKPOINT_BINDINGS, - reason="cuda.bindings does not expose the CUDA checkpoint API", + BUILD_CUDA_MAJOR < 13, + reason="the checkpoint helpers use CUDA 13 binding types", ) From af883372e1df7ed77a298c29f5762ae1107db14e Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Fri, 18 Sep 2026 14:52:14 -0700 Subject: [PATCH 06/18] cuda.core: regenerate stubs and drop exec from the floor tool and tests (#2783) stubgen-pyx regenerated version.pyi (BUILD_CUDA_MAJOR), system/_system.pyi (the constant is now True) and system/_device.pyi (NVLink 6.0 mapping is unconditional). ci/tools/cuda_core_bindings_floor.py reads the CUDA_BINDINGS_FLOOR literal with ast instead of executing the module; its tests and test_build_hooks.py load modules with importlib instead of exec, and the stubbed configuration check takes any arguments. --- ci/tools/cuda_core_bindings_floor.py | 24 +++++++++++++++---- .../tests/test_cuda_core_bindings_floor.py | 11 ++++----- cuda_core/cuda/core/_utils/version.pyi | 3 +-- cuda_core/cuda/core/system/_device.pyi | 3 +-- cuda_core/cuda/core/system/_system.pyi | 4 ++-- cuda_core/tests/test_build_hooks.py | 17 ++++++------- 6 files changed, 37 insertions(+), 25 deletions(-) diff --git a/ci/tools/cuda_core_bindings_floor.py b/ci/tools/cuda_core_bindings_floor.py index d0983547baa..1af202317de 100644 --- a/ci/tools/cuda_core_bindings_floor.py +++ b/ci/tools/cuda_core_bindings_floor.py @@ -16,12 +16,14 @@ The wheel carries the import-free module cuda/core/_bindings_floor.py, at top level in a single-major build and under cuda/core/cu/ in the merged -wheel; this script evaluates it and prints CUDA_BINDINGS_FLOOR[major]. +wheel; this script reads the CUDA_BINDINGS_FLOOR literal out of it (without +running the module) and prints the entry for `major` as a dotted version. """ from __future__ import annotations import argparse +import ast import sys import zipfile from pathlib import Path @@ -29,13 +31,25 @@ MODULE = "_bindings_floor.py" +def floors_from_source(source: str) -> dict[int, tuple[int, int, int]]: + """The CUDA_BINDINGS_FLOOR literal of _bindings_floor.py, parsed without executing it.""" + for node in ast.parse(source, MODULE).body: + if isinstance(node, ast.AnnAssign): + targets = [node.target] + elif isinstance(node, ast.Assign): + targets = node.targets + else: + continue + if node.value is not None and any(isinstance(t, ast.Name) and t.id == "CUDA_BINDINGS_FLOOR" for t in targets): + return ast.literal_eval(node.value) + raise SystemExit(f"{MODULE} does not assign CUDA_BINDINGS_FLOOR") + + def floor_from_source(source: str, major: int) -> str: - namespace: dict = {} - exec(compile(source, MODULE, "exec"), namespace) - floors = namespace["CUDA_BINDINGS_FLOOR"] + floors = floors_from_source(source) if major not in floors: raise SystemExit(f"CUDA {major} is not a supported major (floors: {sorted(floors)})") - return namespace["format_version"](floors[major]) + return ".".join(str(part) for part in floors[major]) def floor_from_wheel(wheel: Path, major: int) -> str: diff --git a/ci/tools/tests/test_cuda_core_bindings_floor.py b/ci/tools/tests/test_cuda_core_bindings_floor.py index 69f2d81050d..a8e427020a9 100644 --- a/ci/tools/tests/test_cuda_core_bindings_floor.py +++ b/ci/tools/tests/test_cuda_core_bindings_floor.py @@ -13,20 +13,19 @@ FLOOR_MODULE = REPO / "cuda_core" / "cuda" / "core" / "_bindings_floor.py" -def _load_tool(): - spec = importlib.util.spec_from_file_location("cuda_core_bindings_floor", TOOLS / "cuda_core_bindings_floor.py") +def _load(name, path): + spec = importlib.util.spec_from_file_location(name, path) module = importlib.util.module_from_spec(spec) spec.loader.exec_module(module) return module -tool = _load_tool() +tool = _load("cuda_core_bindings_floor", TOOLS / "cuda_core_bindings_floor.py") +floor_module = _load("cuda_core_bindings_floor_module", FLOOR_MODULE) def _expected(major): - namespace = {} - exec(FLOOR_MODULE.read_text(encoding="utf-8"), namespace) - return namespace["format_version"](namespace["CUDA_BINDINGS_FLOOR"][major]) + return floor_module.format_version(floor_module.CUDA_BINDINGS_FLOOR[major]) def _wheel(tmp_path, entries): diff --git a/cuda_core/cuda/core/_utils/version.pyi b/cuda_core/cuda/core/_utils/version.pyi index bd8defe7eab..128b2ff4f20 100644 --- a/cuda_core/cuda/core/_utils/version.pyi +++ b/cuda_core/cuda/core/_utils/version.pyi @@ -2,6 +2,7 @@ import functools +BUILD_CUDA_MAJOR: int = ... def _parse_version_triple(version_str: str) -> tuple[int, int, int]: """Parse a PEP 440 version string into a (major, minor, patch) triple. @@ -10,8 +11,6 @@ def _parse_version_triple(version_str: str) -> tuple[int, int, int]: ``0b1`` or ``0rc1`` by extracting only the leading integer from each release segment. """ -BUILD_CUDA_MAJOR: int - @functools.cache def binding_version() -> tuple[int, int, int]: """Return the cuda-bindings version as a (major, minor, patch) triple.""" diff --git a/cuda_core/cuda/core/system/_device.pyi b/cuda_core/cuda/core/system/_device.pyi index 3e2a6bd018c..6fea049b21b 100644 --- a/cuda_core/cuda/core/system/_device.pyi +++ b/cuda_core/cuda/core/system/_device.pyi @@ -22,8 +22,7 @@ _EVENT_TYPE_MAPPING = {nvml.EventType.NONE: EventType.NONE, nvml.EventType.SINGL _EVENT_TYPE_INV_MAPPING = {v: k for k, v in _EVENT_TYPE_MAPPING.items()} _FAN_CONTROL_POLICY_MAPPING = {nvml.FanControlPolicy.TEMPERATURE_CONTINUOUS_SW: FanControlPolicy.TEMPERATURE_CONTROLLED, nvml.FanControlPolicy.MANUAL: FanControlPolicy.MANUAL} _INFOROM_OBJECT_MAPPING = {InforomObject.OEM: nvml.InforomObject.INFOROM_OEM, InforomObject.ECC: nvml.InforomObject.INFOROM_ECC, InforomObject.POWER: nvml.InforomObject.INFOROM_POWER, InforomObject.DEN: nvml.InforomObject.INFOROM_DEN} -_NVLINK_VERSION_MAPPING = {nvml.NvlinkVersion.VERSION_1_0: (1, 0), nvml.NvlinkVersion.VERSION_2_0: (2, 0), nvml.NvlinkVersion.VERSION_2_2: (2, 2), nvml.NvlinkVersion.VERSION_3_0: (3, 0), nvml.NvlinkVersion.VERSION_3_1: (3, 1), nvml.NvlinkVersion.VERSION_4_0: (4, 0), nvml.NvlinkVersion.VERSION_5_0: (5, 0)} -_NVLINK_VERSION_6_0 = getattr(nvml.NvlinkVersion, 'VERSION_6_0', None) +_NVLINK_VERSION_MAPPING = {nvml.NvlinkVersion.VERSION_1_0: (1, 0), nvml.NvlinkVersion.VERSION_2_0: (2, 0), nvml.NvlinkVersion.VERSION_2_2: (2, 2), nvml.NvlinkVersion.VERSION_3_0: (3, 0), nvml.NvlinkVersion.VERSION_3_1: (3, 1), nvml.NvlinkVersion.VERSION_4_0: (4, 0), nvml.NvlinkVersion.VERSION_5_0: (5, 0), nvml.NvlinkVersion.VERSION_6_0: (6, 0)} _TEMPERATURE_THRESHOLD_MAPPING = {TemperatureThresholds.SHUTDOWN: nvml.TemperatureThresholds.TEMPERATURE_THRESHOLD_SHUTDOWN, TemperatureThresholds.SLOWDOWN: nvml.TemperatureThresholds.TEMPERATURE_THRESHOLD_SLOWDOWN, TemperatureThresholds.MEM_MAX: nvml.TemperatureThresholds.TEMPERATURE_THRESHOLD_MEM_MAX, TemperatureThresholds.GPU_MAX: nvml.TemperatureThresholds.TEMPERATURE_THRESHOLD_GPU_MAX, TemperatureThresholds.ACOUSTIC_MIN: nvml.TemperatureThresholds.TEMPERATURE_THRESHOLD_ACOUSTIC_MIN, TemperatureThresholds.ACOUSTIC_CURR: nvml.TemperatureThresholds.TEMPERATURE_THRESHOLD_ACOUSTIC_CURR, TemperatureThresholds.ACOUSTIC_MAX: nvml.TemperatureThresholds.TEMPERATURE_THRESHOLD_ACOUSTIC_MAX, TemperatureThresholds.GPS_CURR: nvml.TemperatureThresholds.TEMPERATURE_THRESHOLD_GPS_CURR} _THERMAL_CONTROLLER_MAPPING = {nvml.ThermalController.GPU_INTERNAL: ThermalController.GPU_INTERNAL, nvml.ThermalController.ADM1032: ThermalController.ADM1032, nvml.ThermalController.ADT7461: ThermalController.ADT7461, nvml.ThermalController.MAX6649: ThermalController.MAX6649, nvml.ThermalController.MAX1617: ThermalController.MAX1617, nvml.ThermalController.LM99: ThermalController.LM99, nvml.ThermalController.LM89: ThermalController.LM89, nvml.ThermalController.LM64: ThermalController.LM64, nvml.ThermalController.G781: ThermalController.G781, nvml.ThermalController.ADT7473: ThermalController.ADT7473, nvml.ThermalController.SBMAX6649: ThermalController.SBMAX6649, nvml.ThermalController.VBIOSEVT: ThermalController.VBIOSEVT, nvml.ThermalController.OS: ThermalController.OS, nvml.ThermalController.NVSYSCON_CANOAS: ThermalController.NVSYSCON_CANOAS, nvml.ThermalController.NVSYSCON_E551: ThermalController.NVSYSCON_E551, nvml.ThermalController.MAX6649R: ThermalController.MAX6649R, nvml.ThermalController.ADT7473S: ThermalController.ADT7473S, nvml.ThermalController.UNKNOWN: ThermalController.UNKNOWN} _THERMAL_TARGET_MAPPING = {nvml.ThermalTarget.NONE: ThermalTarget.NONE, nvml.ThermalTarget.GPU: ThermalTarget.GPU, nvml.ThermalTarget.MEMORY: ThermalTarget.MEMORY, nvml.ThermalTarget.POWER_SUPPLY: ThermalTarget.POWER_SUPPLY, nvml.ThermalTarget.BOARD: ThermalTarget.BOARD, nvml.ThermalTarget.VCD_BOARD: ThermalTarget.VCD_BOARD, nvml.ThermalTarget.VCD_INLET: ThermalTarget.VCD_INLET, nvml.ThermalTarget.VCD_OUTLET: ThermalTarget.VCD_OUTLET, nvml.ThermalTarget.ALL: ThermalTarget.ALL} diff --git a/cuda_core/cuda/core/system/_system.pyi b/cuda_core/cuda/core/system/_system.pyi index b29b16f16aa..9306795f3c4 100644 --- a/cuda_core/cuda/core/system/_system.pyi +++ b/cuda_core/cuda/core/system/_system.pyi @@ -1,6 +1,6 @@ # This file was generated by stubgen-pyx v0.2.22 from cuda_core/cuda/core/system/_system.pyx -CUDA_BINDINGS_NVML_IS_COMPATIBLE: bool +CUDA_BINDINGS_NVML_IS_COMPATIBLE: bool = True __all__ = ['get_driver_branch', 'get_kernel_mode_driver_version', 'get_user_mode_driver_version', 'get_nvml_version', 'get_num_devices', 'get_process_name', 'CUDA_BINDINGS_NVML_IS_COMPATIBLE'] def get_user_mode_driver_version() -> tuple[int, ...]: @@ -8,7 +8,7 @@ def get_user_mode_driver_version() -> tuple[int, ...]: Get the user-mode (UMD / CUDA) driver version. This is the most commonly needed version when checking CUDA driver - compatibility. It works with all ``cuda-bindings`` versions. + compatibility. Returns ------- diff --git a/cuda_core/tests/test_build_hooks.py b/cuda_core/tests/test_build_hooks.py index 34fc10d534d..b8f4a06de84 100644 --- a/cuda_core/tests/test_build_hooks.py +++ b/cuda_core/tests/test_build_hooks.py @@ -234,7 +234,7 @@ def fake_cythonize(ext_modules, **kwargs): monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: "/nonexistent-cuda") # The configuration check reads that header and the installed cuda-bindings; # it has its own tests (TestBuildConfigurationCheck). - monkeypatch.setattr(build_hooks, "_check_build_configuration", lambda cuda_path, cuda_major: None) + monkeypatch.setattr(build_hooks, "_check_build_configuration", lambda *_: None) monkeypatch.setattr(build_hooks, "cythonize", fake_cythonize) monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", cuda_major) build_hooks._determine_cuda_major_version.cache_clear() @@ -494,12 +494,13 @@ def test_floor_bindings_and_matching_header_pass_and_are_recorded(self, tmp_path build_hooks._check_build_configuration(cuda_path, str(major)) - info = {} - exec(build_hooks._BUILD_INFO_PATH.read_text(), info) - assert info["CUDA_MAJOR"] == major - assert info["CUDA_VERSION"] == floor[0] * 1000 + floor[1] * 10 - assert info["CUDA_BINDINGS_FLOOR"] == floor - assert info["CUDA_BINDINGS_BUILD_VERSION"] == version + spec = importlib.util.spec_from_file_location("_build_info_under_test", build_hooks._BUILD_INFO_PATH) + info = importlib.util.module_from_spec(spec) + spec.loader.exec_module(info) + assert major == info.CUDA_MAJOR + assert floor[0] * 1000 + floor[1] * 10 == info.CUDA_VERSION + assert floor == info.CUDA_BINDINGS_FLOOR + assert version == info.CUDA_BINDINGS_BUILD_VERSION @pytest.mark.agent_authored(model="claude-fable-5-1") def test_bindings_below_the_floor_fail(self, tmp_path, monkeypatch): @@ -597,7 +598,7 @@ def fake_cythonize(ext_modules, **kwargs): return [] monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: "/nonexistent-cuda") - monkeypatch.setattr(build_hooks, "_check_build_configuration", lambda cuda_path, cuda_major: None) + monkeypatch.setattr(build_hooks, "_check_build_configuration", lambda *_: None) monkeypatch.setattr(build_hooks, "cythonize", fake_cythonize) monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", "13") build_hooks._determine_cuda_major_version.cache_clear() From b4fedecce74706b37b0f16a6e5a0d2fd7b283114 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Fri, 18 Sep 2026 16:09:01 -0700 Subject: [PATCH 07/18] cuda.core: import cuda.bindings for the build check through the #1824 namespace repair (#2783) The build-configuration check imported cuda.bindings directly. In an isolated PEP 517 build with an in-tree backend (the wheel-from-sdist CI job), the project's own cuda/ directory is the whole `cuda` namespace, so the cuda-bindings pip installed into the build environment is not importable (#1824) and the check failed with "requires cuda-bindings to build". Import it the way _import_get_cuda_path_or_home() imports cuda.pathfinder: add the site-packages cuda/ directory to the namespace path when the plain import fails. The pxd-path block uses the same helper. Package metadata is not a substitute: pip's in-process hook runner forwards find_distributions without the requested name, so it returns this project's own metadata for `cuda-bindings`. test_managed_ops.py asserted the old wording of the NUMA-host message ("cuda-bindings 13.0+"), which now names the CUDA 13 build; the four regexes follow. --- cuda_core/build_hooks.py | 42 ++++++++++++++++++---- cuda_core/tests/memory/test_managed_ops.py | 8 ++--- cuda_core/tests/test_build_hooks.py | 15 +++++--- 3 files changed, 50 insertions(+), 15 deletions(-) diff --git a/cuda_core/build_hooks.py b/cuda_core/build_hooks.py index 70938f4091e..5e74448b045 100644 --- a/cuda_core/build_hooks.py +++ b/cuda_core/build_hooks.py @@ -66,6 +66,37 @@ def _import_get_cuda_path_or_home(): return cuda.pathfinder.get_cuda_path_or_home +def _import_cuda_bindings(): + """Import cuda.bindings, working around PEP 517 namespace shadowing. + + Same problem and same repair as _import_get_cuda_path_or_home() (see + https://github.com/NVIDIA/cuda-python/issues/1824): in an isolated build the + project's own ``cuda/`` directory is the whole ``cuda`` namespace, so the + cuda-bindings pip installed into the build environment is not importable + until its ``cuda/`` directory is added to the namespace path. Raises + ModuleNotFoundError when no cuda-bindings is installed at all. + (importlib.metadata is no alternative: pip's in-process hook runner forwards + ``find_distributions`` without the requested name, so it reports this + project's own metadata for any name.) + """ + try: + import cuda.bindings + except ModuleNotFoundError as exc: + if exc.name not in ("cuda", "cuda.bindings"): + raise + import cuda + + for p in sys.path: + sp_cuda = Path(p) / "cuda" + if (sp_cuda / "bindings").is_dir(): + cuda.__path__ = list(cuda.__path__) + [str(sp_cuda)] + break + else: + raise + import cuda.bindings + return cuda.bindings + + @functools.cache def _get_cuda_path() -> str: get_cuda_path_or_home = _import_get_cuda_path_or_home() @@ -178,13 +209,12 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: requirement = floor.pip_requirement(major) try: - bindings_module = importlib.import_module("cuda.bindings") - except ImportError as exc: + bindings_version = _import_cuda_bindings().__version__ + except ModuleNotFoundError as exc: raise RuntimeError( f"cuda.core requires cuda-bindings to build (install '{requirement}'). " "Isolated builds install it automatically; other builds must provide it." ) from exc - bindings_version = bindings_module.__version__ bindings = floor.release_triple(bindings_version) if bindings is None: raise RuntimeError( @@ -360,10 +390,10 @@ def _build_cuda_core(debug=False): # We need to add the directory containing the 'cuda' package so Cython can resolve # "from cuda.bindings cimport cydriver" try: - import cuda.bindings + cuda_bindings = _import_cuda_bindings() - bindings_path = Path(cuda.bindings.__file__).parent # .../cuda/bindings/ - print(f"Using cuda-bindings {cuda.bindings.__version__} from {bindings_path}", file=sys.stderr) + bindings_path = Path(cuda_bindings.__file__).parent # .../cuda/bindings/ + print(f"Using cuda-bindings {cuda_bindings.__version__} from {bindings_path}", file=sys.stderr) cuda_package_dir = bindings_path.parent.parent # .../cuda_bindings/ (contains cuda/) if str(cuda_package_dir) not in sys.path: sys.path.insert(0, str(cuda_package_dir)) diff --git a/cuda_core/tests/memory/test_managed_ops.py b/cuda_core/tests/memory/test_managed_ops.py index 5fa44f3f43a..6915d261e1d 100644 --- a/cuda_core/tests/memory/test_managed_ops.py +++ b/cuda_core/tests/memory/test_managed_ops.py @@ -465,7 +465,7 @@ def test_accessed_by_set_assignment_validates_kind_before_mutation( with pytest.raises( (ValueError, TypeError), - match=r"does not support location_type='host_numa'|cuda-bindings 13\.0\+", + match=r"does not support location_type='host_numa'|CUDA 13 build of cuda\.core", ): buf.accessed_by = {device, Host(numa_id=0)} @@ -517,7 +517,7 @@ def test_operation_validation(self, managed_buffer): # rejected at the boundary first (TypeError). with pytest.raises( (ValueError, TypeError), - match=r"does not support location_type='host_numa'|cuda-bindings 13\.0\+", + match=r"does not support location_type='host_numa'|CUDA 13 build of cuda\.core", ): buf.accessed_by.add(Host(numa_id=_INVALID_HOST_DEVICE_ORDINAL)) @@ -536,13 +536,13 @@ def test_advise_location_validation(self, location_ops_device, external_managed_ # accessed_by rejects host_numa (CUDA 13: kind check; CUDA 12: boundary) with pytest.raises( (ValueError, TypeError), - match=r"does not support location_type='host_numa'|cuda-bindings 13\.0\+", + match=r"does not support location_type='host_numa'|CUDA 13 build of cuda\.core", ): buf.accessed_by.add(Host(numa_id=0)) # accessed_by rejects host_numa_current (same reasoning) with pytest.raises( (ValueError, TypeError), - match=r"does not support location_type='host_numa_current'|cuda-bindings 13\.0\+", + match=r"does not support location_type='host_numa_current'|CUDA 13 build of cuda\.core", ): buf.accessed_by.add(Host.numa_current()) diff --git a/cuda_core/tests/test_build_hooks.py b/cuda_core/tests/test_build_hooks.py index b8f4a06de84..3d9d4fc8541 100644 --- a/cuda_core/tests/test_build_hooks.py +++ b/cuda_core/tests/test_build_hooks.py @@ -457,12 +457,17 @@ def test_serial_builds_and_compilers_without_the_hook_keep_the_stock_path(self, def _fake_bindings(monkeypatch, version): - """Install a stand-in cuda.bindings whose __version__ is `version`.""" + """Make the build see an installed cuda-bindings of `version` (None: not installed).""" import types - module = types.ModuleType("cuda.bindings") - module.__version__ = version - monkeypatch.setitem(sys.modules, "cuda.bindings", module) + def import_cuda_bindings(): + if version is None: + raise ModuleNotFoundError("No module named 'cuda.bindings'", name="cuda.bindings") + module = types.ModuleType("cuda.bindings") + module.__version__ = version + return module + + monkeypatch.setattr(build_hooks, "_import_cuda_bindings", import_cuda_bindings) def _write_cuda_h(tmp_path, cuda_version): @@ -539,7 +544,7 @@ def test_header_is_read_even_when_the_major_override_is_set(self, tmp_path, monk @pytest.mark.agent_authored(model="claude-fable-5-1") def test_missing_bindings_is_a_build_error(self, tmp_path, monkeypatch): - monkeypatch.setitem(sys.modules, "cuda.bindings", None) # makes `import cuda.bindings` fail + _fake_bindings(monkeypatch, None) # no cuda-bindings in the build environment cuda_path = _write_cuda_h(tmp_path, 13040) with pytest.raises(RuntimeError, match="requires cuda-bindings to build"): build_hooks._check_build_configuration(cuda_path, "13") From b06cc7c6df149732e42fe90781ea00f1109be0c4 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Tue, 22 Sep 2026 10:32:30 -0700 Subject: [PATCH 08/18] cuda.core: read linked LTOIR through cynvjitlink instead of probing the Python bindings (#2783) #2867 fetched linked LTOIR through the Python-level cuda.bindings.nvjitlink module, probing it with hasattr for get_linked_ltoir_size/get_linked_ltoir because the cuda-bindings in use might predate them, and left a TODO(#2783) to switch once a floor existed. Both floors (12.9.8, 13.4.1) have the cynvjitlink getters, so Linker.link("ltoir") now calls nvJitLinkGetLinkedLTOIRSize/nvJitLinkGetLinkedLTOIR directly, like the cubin and ptx branches. The nvJitLink library-version gate (13.3) stays: that is the library, not the bindings. The cached module object is gone; only the cached library version remains. Tests that faked the module probe or asserted the cached module follow. --- cuda_core/cuda/core/_linker.pyi | 3 -- cuda_core/cuda/core/_linker.pyx | 37 +++++-------------- cuda_core/tests/test_linker.py | 34 ++--------------- .../tests/test_optional_dependency_imports.py | 5 --- 4 files changed, 13 insertions(+), 66 deletions(-) diff --git a/cuda_core/cuda/core/_linker.pyi b/cuda_core/cuda/core/_linker.pyi index 95700199d54..9ff5dfd53c7 100644 --- a/cuda_core/cuda/core/_linker.pyi +++ b/cuda_core/cuda/core/_linker.pyi @@ -21,7 +21,6 @@ const_char_ptr: TypeAlias = bytes __all__ = ['Linker', 'LinkerOptions'] LinkerHandleT = Union['cuda.bindings.nvjitlink.nvJitLinkHandle', 'cuda.bindings.driver.CUlinkState'] _driver = None -_nvjitlink = None _nvjitlink_version = None _inited = False _use_nvjitlink_backend = None @@ -263,8 +262,6 @@ class LinkerOptions: def _require_nvjitlink_version(minimum_version: tuple[int, int], feature: str) -> None: """Check that the cached nvJitLink runtime meets a feature's requirement.""" -def _linked_ltoir_output_module(): - """Return bindings that can retrieve linked LTOIR without a Cython dependency.""" def _nvjitlink_has_version_symbol(nvjitlink) -> bool: ... def _decide_nvjitlink_or_driver() -> bool: """Return True if falling back to the cuLink* driver APIs.""" diff --git a/cuda_core/cuda/core/_linker.pyx b/cuda_core/cuda/core/_linker.pyx index aea00c9d82d..f30a3fc30a7 100644 --- a/cuda_core/cuda/core/_linker.pyx +++ b/cuda_core/cuda/core/_linker.pyx @@ -18,7 +18,6 @@ from cuda.bindings cimport cynvjitlink from ._rt cimport ( as_cu, - as_intptr, as_py, create_culink_handle, create_nvjitlink_handle, @@ -687,14 +686,13 @@ cdef inline object Linker_link(Linker self, str target_type): 'LTOIR output is not supported with "ptx" or "cubin" inputs; ' "they carry no LTOIR and would be omitted from the output" ) - nvjitlink_module = _linked_ltoir_output_module() + _require_nvjitlink_version((13, 3), "LTOIR output") cdef cynvjitlink.nvJitLinkHandle c_nvjitlink_h cdef cydriver.CUlinkState c_culink_state cdef size_t c_output_size = 0 cdef char* c_code_ptr cdef void* c_cubin_out = NULL - cdef intptr_t c_handle if self._use_nvjitlink: c_nvjitlink_h = as_cu(self._nvjitlink_handle) @@ -717,10 +715,13 @@ cdef inline object Linker_link(Linker self, str target_type): HANDLE_RETURN_NVJITLINK(c_nvjitlink_h, cynvjitlink.nvJitLinkGetLinkedPtx(c_nvjitlink_h, c_code_ptr)) else: - c_handle = as_intptr(self._nvjitlink_handle) - output_size = nvjitlink_module.get_linked_ltoir_size(c_handle) - code = bytearray(output_size) - nvjitlink_module.get_linked_ltoir(c_handle, code) + HANDLE_RETURN_NVJITLINK(c_nvjitlink_h, + cynvjitlink.nvJitLinkGetLinkedLTOIRSize(c_nvjitlink_h, &c_output_size)) + code = bytearray(c_output_size) + c_code_ptr = (code) + with nogil: + HANDLE_RETURN_NVJITLINK(c_nvjitlink_h, + cynvjitlink.nvJitLinkGetLinkedLTOIR(c_nvjitlink_h, c_code_ptr)) else: c_culink_state = as_cu(self._culink_handle) try: @@ -752,7 +753,6 @@ cdef inline void Linker_annotate_error_log(Linker self, object e): # TODO: revisit this treatment for py313t builds _driver = None # populated if nvJitLink cannot be used -_nvjitlink = None # populated if nvJitLink can be used _nvjitlink_version = None _inited = False _use_nvjitlink_backend = None # set by _decide_nvjitlink_or_driver() @@ -770,23 +770,6 @@ def _require_nvjitlink_version(minimum_version: tuple[int, int], feature: str) - raise RuntimeError(f"{feature} requires nvJitLink {required} or newer; found {detected}") -# TODO(#2783): Replace this Python-level dispatch with direct cimports once -# the cuda-bindings runtime floor includes the linked-LTOIR getters. -def _linked_ltoir_output_module(): - """Return bindings that can retrieve linked LTOIR without a Cython dependency.""" - _require_nvjitlink_version((13, 3), "LTOIR output") - missing = [ - name - for name in ("get_linked_ltoir_size", "get_linked_ltoir") - if not hasattr(_nvjitlink, name) - ] - if missing: - raise RuntimeError( - "LTOIR output requires cuda-bindings with " + " and ".join(missing) - ) - return _nvjitlink - - def _nvjitlink_has_version_symbol(nvjitlink) -> bool: # This condition is equivalent to testing for version >= 12.3 return bool(nvjitlink._inspect_function_pointer("__nvJitLinkVersion")) @@ -795,11 +778,10 @@ def _nvjitlink_has_version_symbol(nvjitlink) -> bool: # Note: this function is reused in the tests def _decide_nvjitlink_or_driver() -> bool: """Return True if falling back to the cuLink* driver APIs.""" - global _driver, _nvjitlink, _nvjitlink_version, _use_nvjitlink_backend + global _driver, _nvjitlink_version, _use_nvjitlink_backend if _use_nvjitlink_backend is not None: return not _use_nvjitlink_backend - _nvjitlink = None _nvjitlink_version = None warn_txt_common = ( @@ -820,7 +802,6 @@ def _decide_nvjitlink_or_driver() -> bool: ) else: if has_version_symbol: - _nvjitlink = _nvjitlink_bindings _nvjitlink_version = _nvjitlink_bindings.version() _use_nvjitlink_backend = True return False # Use nvjitlink diff --git a/cuda_core/tests/test_linker.py b/cuda_core/tests/test_linker.py index 1366df419fb..e25578c3dd1 100644 --- a/cuda_core/tests/test_linker.py +++ b/cuda_core/tests/test_linker.py @@ -39,10 +39,8 @@ nvJitLinkError = nvjitlink.nvJitLinkError nvjitlink_version = nvjitlink.version() - has_linked_ltoir_bindings = all(hasattr(nvjitlink, name) for name in ("get_linked_ltoir_size", "get_linked_ltoir")) else: nvjitlink_version = (0, 0) - has_linked_ltoir_bindings = False class nvJitLinkError(Exception): pass @@ -343,7 +341,7 @@ def fake_decide(): assert called, "_decide_nvjitlink_or_driver was not called" @pytest.mark.agent_authored(model="gpt-5.6") - def test_which_backend_caches_nvjitlink_module_and_version(self, monkeypatch): + def test_which_backend_caches_nvjitlink_version(self, monkeypatch): class NvJitLink: version_calls = 0 @@ -353,13 +351,11 @@ def version(cls): return (13, 4) monkeypatch.setattr(_linker, "_use_nvjitlink_backend", None) - monkeypatch.setattr(_linker, "_nvjitlink", None) monkeypatch.setattr(_linker, "_nvjitlink_version", None) monkeypatch.setattr(_linker, "_nvjitlink_bindings", NvJitLink) monkeypatch.setattr(_linker, "_nvjitlink_has_version_symbol", lambda _nvjitlink: True) assert Linker.which_backend() == "nvJitLink" - assert _linker._nvjitlink is NvJitLink assert _linker._nvjitlink_version == (13, 4) assert NvJitLink.version_calls == 1 @@ -371,7 +367,6 @@ def test_which_backend_falls_back_when_nvjitlink_too_old(self, monkeypatch): """Regression test for #2408: old nvJitLink must not crash which_backend().""" monkeypatch.setattr(_linker, "_use_nvjitlink_backend", None) monkeypatch.setattr(_linker, "_driver", None) - monkeypatch.setattr(_linker, "_nvjitlink", None) monkeypatch.setattr(_linker, "_nvjitlink_version", None) monkeypatch.setattr(_linker, "_nvjitlink_has_version_symbol", lambda _nvjitlink: False) @@ -388,7 +383,6 @@ def test_which_backend_falls_back_when_dylib_missing(self, monkeypatch): monkeypatch.setattr(_linker, "_use_nvjitlink_backend", None) monkeypatch.setattr(_linker, "_driver", None) - monkeypatch.setattr(_linker, "_nvjitlink", None) monkeypatch.setattr(_linker, "_nvjitlink_version", None) def raise_missing(_nvjitlink): @@ -562,26 +556,6 @@ def test_require_nvjitlink_version_accepts_boundary_version(monkeypatch): _linker._require_nvjitlink_version((13, 2), "incremental linking") -@pytest.mark.agent_authored(model="gpt-5.6") -def test_linked_ltoir_output_requires_new_enough_runtime(monkeypatch): - monkeypatch.setattr(_linker, "_nvjitlink_version", (13, 2)) - - with pytest.raises(RuntimeError, match=r"LTOIR output requires nvJitLink 13\.3 or newer; found 13\.2"): - _linker._linked_ltoir_output_module() - - -@pytest.mark.agent_authored(model="gpt-5.6") -def test_linked_ltoir_output_requires_new_enough_bindings(monkeypatch): - class NvJitLinkWithoutLinkedLtoir: - pass - - monkeypatch.setattr(_linker, "_nvjitlink", NvJitLinkWithoutLinkedLtoir) - monkeypatch.setattr(_linker, "_nvjitlink_version", (13, 3)) - - with pytest.raises(RuntimeError, match="cuda-bindings with get_linked_ltoir_size and get_linked_ltoir"): - _linker._linked_ltoir_output_module() - - incremental_caller = r""" extern "C" __device__ int incremental_helper(); extern "C" __global__ void incremental_kernel(int* result) { @@ -657,7 +631,7 @@ def test_incremental_cubin_round_trip(init_cuda): @pytest.mark.agent_authored(model="gpt-5.6") @pytest.mark.skipif( - is_culink_backend or nvjitlink_version < (13, 3) or not has_linked_ltoir_bindings, + is_culink_backend or nvjitlink_version < (13, 3), reason="linked LTOIR output requires nvJitLink 13.3 or newer and matching cuda-bindings", ) def test_incremental_ltoir_round_trip(init_cuda): @@ -683,7 +657,7 @@ def test_incremental_ltoir_round_trip(init_cuda): @pytest.mark.agent_authored(model="gpt-5.6") @pytest.mark.skipif( - is_culink_backend or nvjitlink_version < (13, 3) or not has_linked_ltoir_bindings, + is_culink_backend or nvjitlink_version < (13, 3), reason="linked LTOIR output requires nvJitLink 13.3 or newer and matching cuda-bindings", ) def test_complete_ltoir_round_trip_matches_direct_cubin(init_cuda): @@ -729,7 +703,7 @@ def test_incremental_lto_cubin_round_trip(init_cuda): @pytest.mark.agent_authored(model="gpt-5.6") @pytest.mark.skipif( - is_culink_backend or nvjitlink_version < (13, 3) or not has_linked_ltoir_bindings, + is_culink_backend or nvjitlink_version < (13, 3), reason="linked LTOIR output requires nvJitLink 13.3 or newer and matching cuda-bindings", ) @pytest.mark.parametrize("non_ltoir_type", ("ptx", "cubin")) diff --git a/cuda_core/tests/test_optional_dependency_imports.py b/cuda_core/tests/test_optional_dependency_imports.py index f77db01cfe6..085ec476ea6 100644 --- a/cuda_core/tests/test_optional_dependency_imports.py +++ b/cuda_core/tests/test_optional_dependency_imports.py @@ -18,7 +18,6 @@ def restore_optional_import_state(): saved_nvvm_attempted = _program._nvvm_import_attempted saved_driver = _linker._driver saved_inited = _linker._inited - saved_nvjitlink = _linker._nvjitlink saved_nvjitlink_version = _linker._nvjitlink_version saved_use_nvjitlink = _linker._use_nvjitlink_backend @@ -26,7 +25,6 @@ def restore_optional_import_state(): _program._nvvm_import_attempted = False _linker._driver = None _linker._inited = False - _linker._nvjitlink = None _linker._nvjitlink_version = None _linker._use_nvjitlink_backend = None @@ -36,7 +34,6 @@ def restore_optional_import_state(): _program._nvvm_import_attempted = saved_nvvm_attempted _linker._driver = saved_driver _linker._inited = saved_inited - _linker._nvjitlink = saved_nvjitlink _linker._nvjitlink_version = saved_nvjitlink_version _linker._use_nvjitlink_backend = saved_use_nvjitlink @@ -125,7 +122,6 @@ def version(self): assert use_driver_backend is False assert _linker._use_nvjitlink_backend is True - assert _linker._nvjitlink is nvjitlink_module assert _linker._nvjitlink_version == (13, 4) assert version_calls == 1 @@ -151,5 +147,4 @@ def fake_has_version(_nvjitlink): assert _linker._decide_nvjitlink_or_driver() is True assert called["inspect"] is True assert called["version"] is False - assert _linker._nvjitlink is None assert _linker._nvjitlink_version is None From ab980a9a829761a1f32ce68f9c2ccad2795f9ce3 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Tue, 22 Sep 2026 11:45:20 -0700 Subject: [PATCH 09/18] cuda.core: declare the cuda-bindings floor once, in the pyproject extras (#2783) The floor of each CUDA major was typed in three places: a table in cuda/core/_bindings_floor.py, the cu12/cu13 extras in pyproject.toml, and the support-policy table in the docs, kept in step by tests that only ran in the GPU test jobs. The extras are now the single declaration, in the form `cuda-bindings[all]>=,==.*`, and everything derives from them: - _bindings_floor.py keeps the logic only: floors_from_extras() parses the [project.optional-dependencies] table strictly (a malformed extra fails instead of shifting the floor), plus the version helpers and the import-time check, which now takes the build's recorded floor. - build_hooks.py reads pyproject.toml (tomllib; tomli on 3.10, added to build-system.requires) for the build-time check, the dynamic build requirement and -DCUDA_CORE_MIN_CUDA_VERSION, and records the floor in the generated _build_info.py, which cuda/core/__init__.py already reads. __init__.py no longer hard-codes the supported majors. - ci/tools/cuda_core_bindings_floor.py reads the floor out of the wheel's _build_info.py (per major in the merged wheel) instead of the table. - docs/source/conf.py defines |cuda-bindings-floor-cu12| and |cuda-bindings-floor-cu13| from pyproject.toml; support.rst uses them and labels the row with |release|. The stale cuda-bindings minimums in api_nvml.rst are gone: cuda.core.system has no requirement beyond the floor. - toolshed/check_cuda_core_bindings_floor.py, a new pre-commit hook, checks the two constraints that cannot be derived: ci/versions.yml must build each major against a toolkit of the floor's major.minor, and no documentation page (release notes excepted) may spell a floor out by hand. tests/test_bindings_floor.py runs the same checks. - AGENTS.md documents the coupling and a "Bumping the cuda-bindings floor" checklist. --- .pre-commit-config.yaml | 9 + ci/tools/cuda_core_bindings_floor.py | 26 +-- .../tests/test_cuda_core_bindings_floor.py | 52 +++-- cuda_core/AGENTS.md | 30 ++- cuda_core/build_hooks.py | 67 ++++-- cuda_core/cuda/core/__init__.py | 8 +- cuda_core/cuda/core/_bindings_floor.py | 134 +++++++---- cuda_core/docs/source/api_nvml.rst | 4 +- cuda_core/docs/source/conf.py | 26 +++ cuda_core/docs/source/support.rst | 11 +- cuda_core/pyproject.toml | 9 +- cuda_core/tests/test_bindings_floor.py | 216 +++++++++++++----- cuda_core/tests/test_build_hooks.py | 49 +++- toolshed/check_cuda_core_bindings_floor.py | 119 ++++++++++ 14 files changed, 592 insertions(+), 168 deletions(-) create mode 100644 toolshed/check_cuda_core_bindings_floor.py diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 451bb72fe8e..640a4d4ff66 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -42,6 +42,15 @@ repos: pass_filenames: false verbose: true + - id: check-cuda-core-bindings-floor + name: Check the cuda-bindings floor against ci/versions.yml and the docs + entry: python ./toolshed/check_cuda_core_bindings_floor.py + language: python + files: ^(cuda_core/pyproject\.toml|ci/versions\.yml|cuda_core/cuda/core/_bindings_floor\.py|cuda_core/docs/source/.*\.rst|toolshed/check_cuda_core_bindings_floor\.py)$ + pass_filenames: false + additional_dependencies: + - "tomli>=1.1.0; python_version < '3.11'" + - id: check-spdx name: Check SPDX Headers entry: python ./toolshed/check_spdx.py diff --git a/ci/tools/cuda_core_bindings_floor.py b/ci/tools/cuda_core_bindings_floor.py index 1af202317de..0ac887b6ea0 100644 --- a/ci/tools/cuda_core_bindings_floor.py +++ b/ci/tools/cuda_core_bindings_floor.py @@ -14,10 +14,9 @@ from the checkout, so a nightly job that tests a wheel built from another commit reads that wheel's floor. -The wheel carries the import-free module cuda/core/_bindings_floor.py, at top -level in a single-major build and under cuda/core/cu/ in the merged -wheel; this script reads the CUDA_BINDINGS_FLOOR literal out of it (without -running the module) and prints the entry for `major` as a dotted version. +Each build records its floor in the generated cuda/core/_build_info.py (at top +level in a single-major build, under cuda/core/cu/ in the merged wheel); +this script reads the CUDA_BINDINGS_FLOOR literal out of it without running it. """ from __future__ import annotations @@ -28,11 +27,11 @@ import zipfile from pathlib import Path -MODULE = "_bindings_floor.py" +MODULE = "_build_info.py" -def floors_from_source(source: str) -> dict[int, tuple[int, int, int]]: - """The CUDA_BINDINGS_FLOOR literal of _bindings_floor.py, parsed without executing it.""" +def _literal(source: str, name: str): + """The literal assigned to `name` at module level of `source`, parsed without executing it.""" for node in ast.parse(source, MODULE).body: if isinstance(node, ast.AnnAssign): targets = [node.target] @@ -40,16 +39,15 @@ def floors_from_source(source: str) -> dict[int, tuple[int, int, int]]: targets = node.targets else: continue - if node.value is not None and any(isinstance(t, ast.Name) and t.id == "CUDA_BINDINGS_FLOOR" for t in targets): + if node.value is not None and any(isinstance(t, ast.Name) and t.id == name for t in targets): return ast.literal_eval(node.value) - raise SystemExit(f"{MODULE} does not assign CUDA_BINDINGS_FLOOR") + raise SystemExit(f"{MODULE} does not assign {name}") def floor_from_source(source: str, major: int) -> str: - floors = floors_from_source(source) - if major not in floors: - raise SystemExit(f"CUDA {major} is not a supported major (floors: {sorted(floors)})") - return ".".join(str(part) for part in floors[major]) + if _literal(source, "CUDA_MAJOR") != major: + raise SystemExit(f"{MODULE} records a CUDA {_literal(source, 'CUDA_MAJOR')} build, not CUDA {major}") + return ".".join(str(part) for part in _literal(source, "CUDA_BINDINGS_FLOOR")) def floor_from_wheel(wheel: Path, major: int) -> str: @@ -58,7 +56,7 @@ def floor_from_wheel(wheel: Path, major: int) -> str: for candidate in (f"cuda/core/cu{major}/{MODULE}", f"cuda/core/{MODULE}"): if candidate in names: return floor_from_source(zf.read(candidate).decode("utf-8"), major) - raise SystemExit(f"{wheel.name} contains no {MODULE}; is it a cuda-core wheel?") + raise SystemExit(f"{wheel.name} contains no build for CUDA {major} (no {MODULE}); is it a cuda-core wheel?") def main(argv: list[str] | None = None) -> int: diff --git a/ci/tools/tests/test_cuda_core_bindings_floor.py b/ci/tools/tests/test_cuda_core_bindings_floor.py index a8e427020a9..487e31418a1 100644 --- a/ci/tools/tests/test_cuda_core_bindings_floor.py +++ b/ci/tools/tests/test_cuda_core_bindings_floor.py @@ -9,8 +9,6 @@ import pytest TOOLS = Path(__file__).resolve().parent.parent -REPO = TOOLS.parent.parent -FLOOR_MODULE = REPO / "cuda_core" / "cuda" / "core" / "_bindings_floor.py" def _load(name, path): @@ -21,50 +19,64 @@ def _load(name, path): tool = _load("cuda_core_bindings_floor", TOOLS / "cuda_core_bindings_floor.py") -floor_module = _load("cuda_core_bindings_floor_module", FLOOR_MODULE) +FLOORS = {12: (12, 9, 8), 13: (13, 4, 1)} -def _expected(major): - return floor_module.format_version(floor_module.CUDA_BINDINGS_FLOOR[major]) + +def _build_info(major): + floor = FLOORS[major] + return ( + "# Generated by build_hooks.py at build time. Do not edit or commit.\n" + f"CUDA_MAJOR = {major}\n" + f"CUDA_VERSION = {major * 1000 + floor[1] * 10} # the cuda.h this build compiled against\n" + f"CUDA_BINDINGS_FLOOR = {floor!r}\n" + f"CUDA_BINDINGS_BUILD_VERSION = '{major}.{floor[1]}.{floor[2]}'\n" + ) def _wheel(tmp_path, entries): path = tmp_path / "cuda_core-1.3.0-cp312-cp312-linux_x86_64.whl" with zipfile.ZipFile(path, "w") as zf: - for name in entries: - zf.writestr(name, FLOOR_MODULE.read_text(encoding="utf-8")) + for name, major in entries.items(): + zf.writestr(name, _build_info(major)) return path @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize("major", [12, 13]) def test_reads_the_merged_wheel_layout(tmp_path, major): - wheel = _wheel(tmp_path, ["cuda/core/cu12/_bindings_floor.py", "cuda/core/cu13/_bindings_floor.py"]) - assert tool.floor_from_wheel(wheel, major) == _expected(major) + wheel = _wheel(tmp_path, {"cuda/core/cu12/_build_info.py": 12, "cuda/core/cu13/_build_info.py": 13}) + assert tool.floor_from_wheel(wheel, major) == ".".join(map(str, FLOORS[major])) @pytest.mark.agent_authored(model="claude-fable-5-1") def test_reads_a_single_major_wheel(tmp_path): - wheel = _wheel(tmp_path, ["cuda/core/_bindings_floor.py"]) - assert tool.floor_from_wheel(wheel, 13) == _expected(13) + wheel = _wheel(tmp_path, {"cuda/core/_build_info.py": 13}) + assert tool.floor_from_wheel(wheel, 13) == "13.4.1" + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_rejects_a_single_major_wheel_of_another_major(tmp_path): + wheel = _wheel(tmp_path, {"cuda/core/_build_info.py": 13}) + with pytest.raises(SystemExit, match="records a CUDA 13 build, not CUDA 12"): + tool.floor_from_wheel(wheel, 12) @pytest.mark.agent_authored(model="claude-fable-5-1") -def test_rejects_a_wheel_without_the_module(tmp_path): - wheel = _wheel(tmp_path, []) - with pytest.raises(SystemExit, match="contains no _bindings_floor.py"): +def test_rejects_a_wheel_without_the_build_record(tmp_path): + wheel = _wheel(tmp_path, {}) + with pytest.raises(SystemExit, match="contains no build for CUDA 13"): tool.floor_from_wheel(wheel, 13) @pytest.mark.agent_authored(model="claude-fable-5-1") -def test_rejects_an_unsupported_major(tmp_path): - wheel = _wheel(tmp_path, ["cuda/core/_bindings_floor.py"]) - with pytest.raises(SystemExit, match="CUDA 11 is not a supported major"): - tool.floor_from_wheel(wheel, 11) +def test_rejects_a_build_record_without_the_floor(): + with pytest.raises(SystemExit, match="does not assign CUDA_BINDINGS_FLOOR"): + tool.floor_from_source("CUDA_MAJOR = 13\n", 13) @pytest.mark.agent_authored(model="claude-fable-5-1") def test_cli_prints_the_floor(tmp_path, capsys): - wheel = _wheel(tmp_path, ["cuda/core/_bindings_floor.py"]) + wheel = _wheel(tmp_path, {"cuda/core/_build_info.py": 13}) assert tool.main(["--wheel", str(wheel), "--major", "13"]) == 0 - assert capsys.readouterr().out.strip() == _expected(13) + assert capsys.readouterr().out.strip() == "13.4.1" diff --git a/cuda_core/AGENTS.md b/cuda_core/AGENTS.md index 115e6312619..8c99b23b0f2 100644 --- a/cuda_core/AGENTS.md +++ b/cuda_core/AGENTS.md @@ -31,7 +31,35 @@ This file describes `cuda_core`, the high-level Pythonic CUDA subpackage in the or CUDA headers (`CUDA_HOME`/`CUDA_PATH`) and uses it for build decisions. - Source builds require CUDA headers available through `CUDA_HOME` or `CUDA_PATH`. -- `cuda_core` expects `cuda.bindings` to be present and version-compatible. +- `cuda_core` requires `cuda.bindings` at or above a per-major *floor* at build + and run time, and a `cuda.h` of the same major.minor as that `cuda.bindings` + at build time (NVIDIA/cuda-python#2783). The floors are declared once, by the + `cu12`/`cu13` extras in `pyproject.toml` (`cuda-bindings[all]>=,==.*`); + `build_hooks.py`, the import-time check in `cuda/core/__init__.py`, the docs + and CI read them from there through `cuda/core/_bindings_floor.py`. The C++ + branches on `CUDA_CORE_BUILD_MAJOR` only; whether a feature is available at + run time depends on the driver alone, never on the `cuda.bindings` version. + +### Bumping the cuda-bindings floor + +By policy the floor of each major is the newest `cuda-bindings` release of that +major at the time of a `cuda.core` release, so the bump is a release step, not +something each `cuda-bindings` minor triggers. It is also required by the first +change that uses a `cuda-bindings` API newer than the old floor. To bump: + +1. Edit the `cuda-bindings` pin in the `cu12` or `cu13` extra in + `pyproject.toml`. That is the only version to type. +2. Align `ci/versions.yml`: the `build` (current major) and `prev_build` (prior + major) toolkit pins must share major.minor with the floors, or CI builds a + configuration the build rejects. The pre-commit hook + `check-cuda-core-bindings-floor` (`toolshed/check_cuda_core_bindings_floor.py`) + checks this, and that no documentation page spells a floor out by hand. +3. Add a "Breaking Changes" entry to the release notes naming the new floors. + The support-policy table in `docs/source/support.rst` reads the floors and + the release version at docs-build time; there is nothing to edit there. +4. Refresh the pixi lock files if the pins moved past what they resolve. +5. Pin `cuda-bindings` accordingly in the conda-forge `cuda-core` feedstock + (outside this repository). ## Testing expectations diff --git a/cuda_core/build_hooks.py b/cuda_core/build_hooks.py index 5e74448b045..9e0d43da5f8 100644 --- a/cuda_core/build_hooks.py +++ b/cuda_core/build_hooks.py @@ -113,9 +113,12 @@ def _get_cuda_path() -> str: _BUILD_INFO_PATH = _PACKAGE_DIR / "_build_info.py" +_PYPROJECT_PATH = Path(__file__).parent / "pyproject.toml" + + @functools.cache def _load_bindings_floor(): - """Load cuda/core/_bindings_floor.py, the floor's single source of truth. + """Load cuda/core/_bindings_floor.py, the floor logic (reading, formatting, checking). Loaded by file path: the package this backend builds is not importable during its own build, and the module is deliberately import-free. @@ -127,6 +130,38 @@ def _load_bindings_floor(): return module +@functools.cache +def _bindings_floors() -> dict: + """The cuda-bindings floor per CUDA major, from the cu extras of pyproject.toml. + + The extras are the single place the floors are declared; the build, the + import-time check, the docs and CI all derive from them (see + cuda/core/_bindings_floor.py). A malformed extra fails the build here. + """ + try: + import tomllib + except ModuleNotFoundError: # Python 3.10: tomli is in build-system.requires + import tomli as tomllib + with open(_PYPROJECT_PATH, "rb") as f: + extras = tomllib.load(f)["project"]["optional-dependencies"] + try: + return _load_bindings_floor().floors_from_extras(extras) + except ValueError as exc: + raise RuntimeError(f"{_PYPROJECT_PATH}: {exc}") from exc + + +def _floor_for(cuda_major) -> tuple: + """The floor of `cuda_major`, or a build error naming the supported majors.""" + floors = _bindings_floors() + major = int(cuda_major) + if major not in floors: + raise RuntimeError( + f"cuda.core does not support CUDA {major}; supported CUDA major versions: " + f"{', '.join(str(m) for m in floors)}" + ) + return floors[major] + + def _read_cuda_h_version(cuda_path: str) -> int: """The CUDA_VERSION macro (e.g. 13040 for 13.4) of the cuda.h under cuda_path.""" cuda_h = os.path.join(cuda_path, "include", "cuda.h") @@ -188,8 +223,8 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: """Reject build configurations cuda.core does not support, then record the build. cuda.core supports one configuration per CUDA major series: the installed - cuda-bindings is at least the series' floor (cuda/core/_bindings_floor.py) - and the cuda.h it compiles against has the same major.minor as that + cuda-bindings is at least the series' floor (the cu extra in + pyproject.toml) and the cuda.h it compiles against has the same major.minor as that cuda-bindings, which is the header cuda-bindings itself was generated from. The pip build requirement (get_requires_for_build_*) states the floor, but conda-forge, pixi and --no-build-isolation installs bypass it, so the check @@ -201,12 +236,8 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: """ floor = _load_bindings_floor() major = int(cuda_major) - if major not in floor.CUDA_BINDINGS_FLOOR: - raise RuntimeError( - f"cuda.core does not support CUDA {major}; supported CUDA major versions: " - f"{', '.join(str(m) for m in floor.SUPPORTED_CUDA_MAJORS)}" - ) - requirement = floor.pip_requirement(major) + floor_triple = _floor_for(major) + requirement = floor.bindings_requirement(floor_triple) try: bindings_version = _import_cuda_bindings().__version__ @@ -226,9 +257,9 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: f"Building cuda.core for CUDA {major}, but the installed cuda-bindings is " f"{bindings_version}. Install '{requirement}'." ) - if bindings < floor.CUDA_BINDINGS_FLOOR[major]: + if bindings < floor_triple: raise RuntimeError( - f"cuda.core requires cuda-bindings >= {floor.format_version(floor.CUDA_BINDINGS_FLOOR[major])} " + f"cuda.core requires cuda-bindings >= {floor.format_version(floor_triple)} " f"for CUDA {major}, but {bindings_version} is installed. Install '{requirement}'." ) @@ -242,16 +273,15 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: "Point CUDA_PATH or CUDA_HOME at a matching CUDA Toolkit, or install matching cuda-bindings." ) print(f"Build configuration: CUDA {header[0]}.{header[1]} headers, cuda-bindings {bindings_version}") - _write_build_info(major, cuda_version, floor.CUDA_BINDINGS_FLOOR[major], bindings_version) + _write_build_info(major, cuda_version, floor_triple, bindings_version) def _build_define_macros(cuda_major: str) -> list: """Preprocessor macros that carry the build decision into the C++ (see _cpp/rt/versions.hpp).""" - floor = _load_bindings_floor() major = int(cuda_major) return [ ("CUDA_CORE_BUILD_MAJOR", str(major)), - ("CUDA_CORE_MIN_CUDA_VERSION", str(floor.cuda_version_of(floor.CUDA_BINDINGS_FLOOR[major]))), + ("CUDA_CORE_MIN_CUDA_VERSION", str(_load_bindings_floor().cuda_version_of(_floor_for(major)))), ] @@ -602,14 +632,7 @@ def _get_cuda_bindings_require(): Honored by isolated builds only; _check_build_configuration() enforces the same rule for every other build path. """ - floor = _load_bindings_floor() - cuda_major = int(_determine_cuda_major_version()) - if cuda_major not in floor.CUDA_BINDINGS_FLOOR: - raise RuntimeError( - f"cuda.core does not support CUDA {cuda_major}; supported CUDA major versions: " - f"{', '.join(str(m) for m in floor.SUPPORTED_CUDA_MAJORS)}" - ) - return [floor.pip_requirement(cuda_major)] + return [_load_bindings_floor().bindings_requirement(_floor_for(_determine_cuda_major_version()))] def get_requires_for_build_editable(config_settings=None): diff --git a/cuda_core/cuda/core/__init__.py b/cuda_core/cuda/core/__init__.py index 5c03c958611..065d4893b0e 100644 --- a/cuda_core/cuda/core/__init__.py +++ b/cuda_core/cuda/core/__init__.py @@ -41,9 +41,7 @@ def load_build_module(name: str, cuda_major: int): try: cuda_major = int(version_str.split(".")[0]) except ValueError: - cuda_major = -1 - if cuda_major not in (12, 13): - raise ImportError(f"cuda-bindings 12.x or 13.x must be installed (found {version_str})") + raise ImportError(f"a cuda-bindings release must be installed (found version {version_str!r})") from None try: floor = load_build_module("_bindings_floor", cuda_major) info = load_build_module("_build_info", cuda_major) @@ -51,7 +49,9 @@ def load_build_module(name: str, cuda_major: int): raise ImportError( f"this cuda.core installation has no build for CUDA {cuda_major} (installed cuda-bindings: {version_str})" ) from exc - floor.check_installed_bindings(version_str, info.CUDA_MAJOR, info.CUDA_VERSION, __version__) + floor.check_installed_bindings( + version_str, info.CUDA_MAJOR, info.CUDA_VERSION, info.CUDA_BINDINGS_FLOOR, __version__ + ) subdir = f"cu{cuda_major}" try: diff --git a/cuda_core/cuda/core/_bindings_floor.py b/cuda_core/cuda/core/_bindings_floor.py index 6e79f8392a1..7b5e3d2cf0b 100644 --- a/cuda_core/cuda/core/_bindings_floor.py +++ b/cuda_core/cuda/core/_bindings_floor.py @@ -2,50 +2,58 @@ # # SPDX-License-Identifier: Apache-2.0 -"""The cuda-bindings version floor: the single source of truth. +"""The cuda-bindings version floor: how it is read and how it is enforced. cuda.core supports two CUDA major series at a time and requires, for each, a minimum cuda-bindings version (the *floor*) at build time and at run time. The -floor is the newest cuda-bindings release of that series that the CI source -root can build, normally the release cuda.core's own wheels are built against. -See https://github.com/NVIDIA/cuda-python/issues/2783 and the support policy. - -This module is imported by the build backend (``build_hooks.py``, by file -path, because the package is not importable during its own build), by -``cuda/core/__init__.py`` at import time, and by tests that keep the static -pins in ``pyproject.toml`` and ``ci/versions.yml`` in step. It therefore uses -the standard library only and must not import anything from ``cuda``. - -Bumping a floor is a release-note item under "Breaking Changes". Bump it in the -same PR that first uses a cuda-bindings API newer than the old floor; the CI -rows that install the floor bindings fail otherwise. +floor is the newest cuda-bindings release of that series at the time of the +cuda.core release, normally the release cuda.core's own wheels are built +against. See https://github.com/NVIDIA/cuda-python/issues/2783 and the support +policy in the documentation. + +The floors are declared in exactly one place: the ``cu12`` and ``cu13`` extras +in ``pyproject.toml``, each of which pins ``cuda-bindings>=,==.*``. +Everything else derives from them through :func:`floors_from_extras`: + +- the build backend (``build_hooks.py``) checks the installed cuda-bindings + and header against the floor and records the floor of the build in the + generated ``_build_info.py``; +- ``cuda/core/__init__.py`` checks the installed cuda-bindings against that + record with :func:`check_installed_bindings`; +- the documentation reads the floors into substitutions (``docs/source/conf.py``); +- ``toolshed/check_cuda_core_bindings_floor.py`` (a pre-commit hook) checks + that the CI toolkit pins in ``ci/versions.yml`` sit in the floors' minors. + +This module is loaded by file path during the build and by the tools above, +so it uses the standard library only and must not import anything from +``cuda``. """ from __future__ import annotations import re +from collections.abc import Mapping, Sequence __all__ = [ - "CUDA_BINDINGS_FLOOR", - "SUPPORTED_CUDA_MAJORS", + "bindings_requirement", "check_installed_bindings", "cuda_version_of", + "floors_from_extras", "format_version", - "pip_requirement", "release_triple", "required_minimum", ] -# Minimum cuda-bindings release per CUDA major series, as a (major, minor, patch) -# triple. Keep in step with the `cu12`/`cu13` extras in pyproject.toml (tested). -CUDA_BINDINGS_FLOOR: dict[int, tuple[int, int, int]] = { - 12: (12, 9, 8), - 13: (13, 4, 1), -} - -SUPPORTED_CUDA_MAJORS = tuple(sorted(CUDA_BINDINGS_FLOOR)) - _RELEASE_RE = re.compile(r"^(\d+)\.(\d+)\.(\d+)") +_EXTRA_RE = re.compile(r"^cu(\d+)$") +# A cuda-bindings requirement, with or without extras such as `[all]`, up to +# an optional environment marker. The name must be followed by `[`, an +# operator, a marker, whitespace or the end so that other names do not match. +_REQUIREMENT_RE = re.compile( + r"^\s*cuda-bindings(?=[\[<>=!~;\s]|$)\s*(?:\[[^\]]*\])?\s*(?P[^;]*?)\s*(?:;.*)?$" +) +_FLOOR_SPEC_RE = re.compile(r"^>=(\d+)\.(\d+)\.(\d+)$") +_MAJOR_SPEC_RE = re.compile(r"^==(\d+)\.\*$") def release_triple(version: str) -> tuple[int, int, int] | None: @@ -71,12 +79,54 @@ def cuda_version_of(triple: tuple[int, int, int]) -> int: return triple[0] * 1000 + triple[1] * 10 -def pip_requirement(major: int) -> str: - """The pip requirement that pins cuda-bindings to the floor and the major.""" - return f"cuda-bindings>={format_version(CUDA_BINDINGS_FLOOR[major])},=={major}.*" +def floors_from_extras(extras: Mapping[str, Sequence[str]]) -> dict[int, tuple[int, int, int]]: + """The floor per CUDA major, from the ``cu`` extras of ``pyproject.toml``. - -def required_minimum(cuda_major: int, header_cuda_version: int) -> tuple[int, int, int]: + ``extras`` is the parsed ``[project.optional-dependencies]`` table. Each + ``cu`` extra must list exactly one cuda-bindings requirement of the + form ``cuda-bindings[...]>=..,==.*`` (the order + of the two specifiers does not matter). Anything else raises ValueError + naming the extra, so a typo fails the build and the pre-commit hook + instead of shifting the floor silently. + """ + floors: dict[int, tuple[int, int, int]] = {} + for extra, requirements in extras.items(): + m = _EXTRA_RE.match(extra) + if m is None: + continue + major = int(m.group(1)) + matches = [(r, rm) for r in requirements if (rm := _REQUIREMENT_RE.match(r)) is not None] + if len(matches) != 1: + raise ValueError( + f"the {extra!r} extra must list exactly one cuda-bindings requirement, found {len(matches)}" + ) + requirement, requirement_match = matches[0] + specifiers = [s.strip() for s in requirement_match.group("specifiers").split(",") if s.strip()] + floor_matches = [fm for s in specifiers if (fm := _FLOOR_SPEC_RE.match(s)) is not None] + major_matches = [mm for s in specifiers if (mm := _MAJOR_SPEC_RE.match(s)) is not None] + if len(specifiers) != 2 or len(floor_matches) != 1 or len(major_matches) != 1: + raise ValueError( + f"the {extra!r} extra must pin cuda-bindings as '>=..,==.*', " + f"found {requirement!r}" + ) + floor = (int(floor_matches[0].group(1)), int(floor_matches[0].group(2)), int(floor_matches[0].group(3))) + pinned_major = int(major_matches[0].group(1)) + if floor[0] != major or pinned_major != major: + raise ValueError( + f"the {extra!r} extra pins cuda-bindings {requirement!r}, whose majors do not match CUDA {major}" + ) + floors[major] = floor + if not floors: + raise ValueError("pyproject.toml declares no cu extra") + return dict(sorted(floors.items())) + + +def bindings_requirement(floor: tuple[int, int, int]) -> str: + """The pip requirement that pins cuda-bindings to ``floor`` and its major, without extras.""" + return f"cuda-bindings>={format_version(floor)},=={floor[0]}.*" + + +def required_minimum(floor: tuple[int, int, int], header_cuda_version: int) -> tuple[int, int, int]: """The minimum cuda-bindings a build accepts at run time. A build accepts the floor of its major series, and never a cuda-bindings @@ -84,7 +134,6 @@ def required_minimum(cuda_major: int, header_cuda_version: int) -> tuple[int, in driver function-pointer keys the C++ layer looks up in cuda-bindings are derived from that header's macros, so an older minor may lack them. """ - floor = CUDA_BINDINGS_FLOOR[cuda_major] header_minor = (header_cuda_version // 1000, header_cuda_version // 10 % 100, 0) return max(floor, header_minor) @@ -93,26 +142,27 @@ def check_installed_bindings( installed_version: str, build_cuda_major: int, build_cuda_version: int, + build_floor: tuple[int, int, int], core_version: str, ) -> tuple[int, int, int]: - """Validate the installed cuda-bindings against this build; return its triple. + """Validate the installed cuda-bindings against a build; return its triple. - Raises ImportError with an actionable message when the installed - cuda-bindings is not a release of a supported major, is not the major - this build was compiled for, or is older than the build's minimum. + ``build_cuda_major``, ``build_cuda_version`` and ``build_floor`` are the + build's record in ``_build_info.py``. Raises ImportError with an actionable + message when the installed cuda-bindings is not a release, is not of the + major this build was compiled for, or is older than the build's minimum. """ installed = release_triple(installed_version) - if installed is None or installed[0] not in CUDA_BINDINGS_FLOOR: - majors = " or ".join(f"{m}.x" for m in SUPPORTED_CUDA_MAJORS) - raise ImportError(f"cuda-bindings {majors} must be installed (found {installed_version})") + if installed is None: + raise ImportError(f"a cuda-bindings {build_cuda_major}.x release is required (found {installed_version})") major = installed[0] if major != build_cuda_major: raise ImportError( f"this cuda.core {core_version} build is for CUDA {build_cuda_major}, but the installed " - f"cuda-bindings is {installed_version}. Install cuda-bindings {build_cuda_major}.x, " - f"or a cuda.core build for CUDA {major}." + f"cuda-bindings is {installed_version}. Install cuda-bindings {build_cuda_major}.x " + f"(pip install 'cuda-bindings=={build_cuda_major}.*'), or a cuda.core build for CUDA {major} if one exists." ) - minimum = required_minimum(major, build_cuda_version) + minimum = required_minimum(build_floor, build_cuda_version) if installed < minimum: floor = format_version(minimum) raise ImportError( diff --git a/cuda_core/docs/source/api_nvml.rst b/cuda_core/docs/source/api_nvml.rst index c96b68ab701..06ceecfa515 100644 --- a/cuda_core/docs/source/api_nvml.rst +++ b/cuda_core/docs/source/api_nvml.rst @@ -10,7 +10,9 @@ This is the API reference for Pythonic access to CUDA system information, through the NVIDIA Management Library (NVML). .. note:: - ``cuda.core.system`` support requires ``cuda-bindings`` 12.9.6 or later for CUDA 12.x, or ``cuda-bindings`` 13.2.0 or later for CUDA 13.x. + ``cuda.core.system`` uses NVML through ``cuda-bindings``. It has no requirement beyond the + ``cuda-bindings`` floor of the release (see :ref:`cuda-core-bindings-floor`); the NVML library + itself is loaded on first use, so importing the module needs neither CUDA nor NVML installed. Basic functions --------------- diff --git a/cuda_core/docs/source/conf.py b/cuda_core/docs/source/conf.py index 451eb8b4453..fd62b3877ef 100644 --- a/cuda_core/docs/source/conf.py +++ b/cuda_core/docs/source/conf.py @@ -17,6 +17,32 @@ sys.path.insert(0, str((Path(__file__).parents[3] / "cuda_python" / "docs" / "exts").absolute())) +# -- cuda-bindings floors ---------------------------------------------------- + + +def _bindings_floor_substitutions() -> str: + """|cuda-bindings-floor-cu12| and |cuda-bindings-floor-cu13|, read from the + cu12/cu13 extras of pyproject.toml, the single place the floors are declared + (see cuda/core/_bindings_floor.py). Used by support.rst.""" + import importlib.util + + import tomllib + + cuda_core = Path(__file__).parents[2] + spec = importlib.util.spec_from_file_location("_bindings_floor", cuda_core / "cuda" / "core" / "_bindings_floor.py") + floor = importlib.util.module_from_spec(spec) + spec.loader.exec_module(floor) + with open(cuda_core / "pyproject.toml", "rb") as f: + extras = tomllib.load(f)["project"]["optional-dependencies"] + return "".join( + f".. |cuda-bindings-floor-cu{major}| replace:: {floor.format_version(triple)}\n" + for major, triple in floor.floors_from_extras(extras).items() + ) + + +rst_prolog = _bindings_floor_substitutions() + + # -- Project information ----------------------------------------------------- project = "cuda.core" diff --git a/cuda_core/docs/source/support.rst b/cuda_core/docs/source/support.rst index 6117ea83328..3ef59f931ba 100644 --- a/cuda_core/docs/source/support.rst +++ b/cuda_core/docs/source/support.rst @@ -72,8 +72,9 @@ CUDA library or CUDA driver versions. Refer to the individual module documentati Each ``cuda-core`` release declares, for each supported CUDA major version, a minimum ``cuda-bindings`` version, its *floor*: the newest ``cuda-bindings`` release of that major at the time of the ``cuda-core`` release, which is the version the published wheels are built against. -The floors of the current release are recorded in ``cuda/core/_bindings_floor.py`` and in the -``cu12``/``cu13`` extras of ``cuda-core``. +The floors of the current release are declared by the ``cu12``/``cu13`` extras of ``cuda-core`` +(in ``pyproject.toml``); the build, the import-time check, this page and CI all read them from +there. .. list-table:: ``cuda-bindings`` floors :header-rows: 1 @@ -81,9 +82,9 @@ The floors of the current release are recorded in ``cuda/core/_bindings_floor.py * - ``cuda-core`` version - CUDA 12 - CUDA 13 - * - 1.3.x - - ``cuda-bindings`` >= 12.9.8 - - ``cuda-bindings`` >= 13.4.1 + * - |release| + - ``cuda-bindings`` >= |cuda-bindings-floor-cu12| + - ``cuda-bindings`` >= |cuda-bindings-floor-cu13| - **At run time**, ``import cuda.core`` requires an installed ``cuda-bindings`` of the same major as the ``cuda-core`` build in use and at least as new as that build's floor. An older diff --git a/cuda_core/pyproject.toml b/cuda_core/pyproject.toml index c73cf3003b5..91ce246c4d1 100644 --- a/cuda_core/pyproject.toml +++ b/cuda_core/pyproject.toml @@ -7,7 +7,9 @@ requires = [ "setuptools>=80", "setuptools-scm[simple]>=8,!=10.1", "Cython>=3.2.5,<3.3", - "cuda-pathfinder>=1.5" + "cuda-pathfinder>=1.5", + # build_hooks.py reads the cuda-bindings floors from the cu12/cu13 extras below. + "tomli>=1.1.0; python_version < '3.11'", ] build-backend = "build_hooks" backend-path = ["."] @@ -53,6 +55,11 @@ dependencies = [ "backports.strenum; python_version < '3.11'", ] +# The cuda-bindings pins below are the cuda-bindings *floors* of this release, one +# per CUDA major, and the only place they are declared: build_hooks.py, the +# import-time check, the docs and CI read them from here (see +# cuda/core/_bindings_floor.py and the "Bumping the cuda-bindings floor" +# checklist in AGENTS.md). Keep the form `>=,==.*`. [project.optional-dependencies] cu12 = ["cuda-bindings[all]>=12.9.8,==12.*", "cuda-toolkit==12.*"] cu13 = ["cuda-bindings[all]>=13.4.1,==13.*", "cuda-toolkit==13.*"] diff --git a/cuda_core/tests/test_bindings_floor.py b/cuda_core/tests/test_bindings_floor.py index d01700d63ec..7810e0de727 100644 --- a/cuda_core/tests/test_bindings_floor.py +++ b/cuda_core/tests/test_bindings_floor.py @@ -2,8 +2,9 @@ # # SPDX-License-Identifier: Apache-2.0 -"""The cuda-bindings version floor (cuda/core/_bindings_floor.py) and the -import-time check built on it. +"""The cuda-bindings version floor (cuda/core/_bindings_floor.py): reading it +from the pyproject extras, the import-time check built on it, and the +consistency hook that guards ci/versions.yml and the docs. Source-tree properties and pure functions only: no GPU, so this file also runs with --noconftest (conftest.py initializes CUDA). The consistency tests read @@ -13,41 +14,96 @@ pytest tests/test_bindings_floor.py -v --noconftest """ +import importlib.util import re +import shutil from pathlib import Path import pytest from cuda.core import _bindings_floor as floor_mod from cuda.core._bindings_floor import ( - CUDA_BINDINGS_FLOOR, - SUPPORTED_CUDA_MAJORS, + bindings_requirement, check_installed_bindings, cuda_version_of, + floors_from_extras, format_version, - pip_requirement, release_triple, required_minimum, ) CUDA_CORE = Path(__file__).resolve().parent.parent REPO = CUDA_CORE.parent +HOOK = REPO / "toolshed" / "check_cuda_core_bindings_floor.py" + + +def _load(name, path): + spec = importlib.util.spec_from_file_location(name, path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +@pytest.fixture(scope="module") +def hook(): + """toolshed/check_cuda_core_bindings_floor.py lives outside cuda_core/, so an sdist tree lacks it.""" + if not HOOK.is_file(): + pytest.skip(f"{HOOK} is not in this tree; the hook tests need the monorepo checkout") + return _load("check_cuda_core_bindings_floor", HOOK) @pytest.mark.agent_authored(model="claude-fable-5-1") def test_floor_module_is_import_free(): - """build_hooks.py loads it by file path during the build; it must stay standard-library only.""" + """build_hooks.py, conf.py and the hook load it by file path; it must stay standard-library only.""" source = Path(floor_mod.__file__).read_text(encoding="utf-8") imports = re.findall(r"^\s*(?:from|import)\s+(\w+)", source, re.M) - assert set(imports) <= {"__future__", "re"} + assert set(imports) <= {"__future__", "collections", "re"} @pytest.mark.agent_authored(model="claude-fable-5-1") -def test_floors_are_release_triples_of_their_major(): - assert SUPPORTED_CUDA_MAJORS == (12, 13) - for major, floor in CUDA_BINDINGS_FLOOR.items(): - assert len(floor) == 3 +def test_floors_come_from_the_pyproject_extras(hook): + """The extras are the single source; reading them back gives one release triple per major.""" + floors = hook.read_floors(REPO) + assert sorted(floors) == [12, 13] + for major, floor in floors.items(): assert floor[0] == major + assert len(floor) == 3 + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_floors_from_extras_accepts_the_declared_form(): + extras = { + "cu12": ["cuda-bindings[all]>=12.9.8,==12.*", "cuda-toolkit==12.*"], + "cu13": ["cuda-bindings[all]==13.*,>=13.4.1", "cuda-toolkit==13.*"], + "test": ["pytest"], + } + assert floors_from_extras(extras) == {12: (12, 9, 8), 13: (13, 4, 1)} + assert floors_from_extras({"cu13": ['cuda-bindings>=13.4.1,==13.* ; python_version >= "3.10"']}) == {13: (13, 4, 1)} + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +@pytest.mark.parametrize( + ("extras", "message"), + [ + ({"cu13": ["cuda-toolkit==13.*"]}, "exactly one cuda-bindings requirement, found 0"), + ({"cu13": ["cuda-bindings>=13.4.1", "cuda-bindings==13.*"]}, "exactly one cuda-bindings requirement, found 2"), + ({"cu13": ["cuda-bindings>=13.4.1"]}, "must pin cuda-bindings as"), + ({"cu13": ["cuda-bindings>=13.4,==13.*"]}, "must pin cuda-bindings as"), + ({"cu13": ["cuda-bindings>=13.4.1,==13.*,<14"]}, "must pin cuda-bindings as"), + ({"cu13": ["cuda-bindings>=12.9.8,==13.*"]}, "majors do not match CUDA 13"), + ({"cu13": ["cuda-bindings>=13.4.1,==12.*"]}, "majors do not match CUDA 13"), + ({"test": ["pytest"]}, "declares no cu extra"), + ], +) +def test_floors_from_extras_rejects_other_forms(extras, message): + with pytest.raises(ValueError, match=message): + floors_from_extras(extras) + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_floors_from_extras_ignores_lookalike_names(): + extras = {"cu13": ["cuda-bindings-extra>=1.0", "cuda-bindings>=13.4.1,==13.*"]} + assert floors_from_extras(extras) == {13: (13, 4, 1)} @pytest.mark.agent_authored(model="claude-fable-5-1") @@ -73,89 +129,139 @@ def test_formatting_helpers(): assert format_version((13, 4, 1)) == "13.4.1" assert cuda_version_of((13, 4, 1)) == 13040 assert cuda_version_of((12, 9, 8)) == 12090 - assert pip_requirement(13) == f"cuda-bindings>={format_version(CUDA_BINDINGS_FLOOR[13])},==13.*" + assert bindings_requirement((13, 4, 1)) == "cuda-bindings>=13.4.1,==13.*" @pytest.mark.agent_authored(model="claude-fable-5-1") def test_required_minimum_is_the_floor_or_the_header_minor(): - floor = CUDA_BINDINGS_FLOOR[13] + floor = (13, 4, 1) header_at_floor = cuda_version_of(floor) - assert required_minimum(13, header_at_floor) == floor + assert required_minimum(floor, header_at_floor) == floor # A build against a newer header than the floor's minor demands that minor: # the driver-pointer keys are derived from the header's macros. - assert required_minimum(13, header_at_floor + 10) == (13, floor[1] + 1, 0) + assert required_minimum(floor, header_at_floor + 10) == (13, 5, 0) # An older header cannot win over the floor. - assert required_minimum(13, 13000) == floor + assert required_minimum(floor, 13000) == floor class TestCheckInstalledBindings: - FLOOR = CUDA_BINDINGS_FLOOR[13] + FLOOR = (13, 4, 1) HEADER = cuda_version_of(FLOOR) + def check(self, installed, major=13, header=None, floor=None): + return check_installed_bindings( + installed, major, self.HEADER if header is None else header, floor or self.FLOOR, "1.3.0" + ) + @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize( "installed", [ - format_version(FLOOR), - f"{FLOOR[0]}.{FLOOR[1]}.{FLOOR[2] + 1}", - f"{FLOOR[0]}.{FLOOR[1]}.{FLOOR[2] + 1}.dev249+gabcdef0", # main-built bindings in CI - f"{FLOOR[0]}.{FLOOR[1] + 1}.0b1", # newer bindings than the build: supported + "13.4.1", + "13.4.2", + "13.4.2.dev249+gabcdef0", # main-built bindings in CI + "13.5.0b1", # newer bindings than the build: supported ], ) def test_accepts_the_floor_and_newer(self, installed): - assert check_installed_bindings(installed, 13, self.HEADER, "1.3.0") == release_triple(installed) + assert self.check(installed) == release_triple(installed) @pytest.mark.agent_authored(model="claude-fable-5-1") def test_rejects_older_than_the_floor_with_the_fix(self): - older = f"{self.FLOOR[0]}.{self.FLOOR[1] - 1}.1" with pytest.raises(ImportError) as excinfo: - check_installed_bindings(older, 13, self.HEADER, "1.3.0") + self.check("13.3.1") message = str(excinfo.value) - assert f"requires cuda-bindings >= {format_version(self.FLOOR)} for CUDA 13" in message - assert f"(found {older})" in message - assert f"pip install -U 'cuda-bindings>={format_version(self.FLOOR)},==13.*'" in message + assert "requires cuda-bindings >= 13.4.1 for CUDA 13" in message + assert "(found 13.3.1)" in message + assert "pip install -U 'cuda-bindings>=13.4.1,==13.*'" in message @pytest.mark.agent_authored(model="claude-fable-5-1") def test_rejects_a_minor_older_than_the_header(self): # Built against a header one minor above the floor; the floor itself no longer suffices. - header = self.HEADER + 10 - with pytest.raises(ImportError, match=rf"requires cuda-bindings >= 13\.{self.FLOOR[1] + 1}\.0"): - check_installed_bindings(format_version(self.FLOOR), 13, header, "1.3.0") + with pytest.raises(ImportError, match=r"requires cuda-bindings >= 13\.5\.0"): + self.check("13.4.1", header=self.HEADER + 10) + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_the_floor_comes_from_the_build_record(self): + # A build recorded with a lower floor (and header) accepts what the default floor rejects. + assert self.check("13.3.0", header=cuda_version_of((13, 2, 0)), floor=(13, 2, 0)) == (13, 3, 0) + # A higher recorded floor rejects what the default floor accepts, even with the header at 13.4. + with pytest.raises(ImportError) as excinfo: + self.check("13.4.1", floor=(13, 5, 0)) + message = str(excinfo.value) + assert "requires cuda-bindings >= 13.5.0 for CUDA 13" in message + assert "pip install -U 'cuda-bindings>=13.5.0,==13.*'" in message @pytest.mark.agent_authored(model="claude-fable-5-1") def test_rejects_another_major_than_the_build(self): with pytest.raises(ImportError, match="build is for CUDA 12, but the installed cuda-bindings is 13.4.1"): - check_installed_bindings("13.4.1", 12, 12090, "1.3.0") + self.check("13.4.1", major=12, header=12090, floor=(12, 9, 8)) + + @pytest.mark.agent_authored(model="claude-fable-5-1") + @pytest.mark.parametrize("installed", ["11.8.0", "14.0.0"]) + def test_another_major_names_only_the_fix_it_knows(self, installed): + """A plain (single-build) install cannot know which other builds exist, so the message + must not promise one.""" + with pytest.raises(ImportError) as excinfo: + self.check(installed, major=12, header=12090, floor=(12, 9, 8)) + message = str(excinfo.value) + assert "Install cuda-bindings 12.x (pip install 'cuda-bindings==12.*')" in message + assert f"build for CUDA {installed.split('.')[0]} if one exists" in message @pytest.mark.agent_authored(model="claude-fable-5-1") - @pytest.mark.parametrize("installed", ["0.1.dev1+g0d22cb444", "11.8.0", "14.0.0", "garbage"]) - def test_rejects_unsupported_or_unparseable_versions(self, installed): + @pytest.mark.parametrize("installed", ["0.1.dev1+g0d22cb444", "garbage"]) + def test_rejects_unparseable_versions(self, installed): with pytest.raises( - ImportError, match=rf"cuda-bindings 12\.x or 13\.x must be installed \(found {re.escape(installed)}\)" + ImportError, match=rf"a cuda-bindings 13\.x release is required \(found {re.escape(installed)}\)" ): - check_installed_bindings(installed, 13, self.HEADER, "1.3.0") + self.check(installed) -@pytest.mark.agent_authored(model="claude-fable-5-1") -def test_pyproject_extras_pin_the_floor(): - """The static `cu12`/`cu13` extras cannot read the module; keep them in step by test.""" - pyproject = (CUDA_CORE / "pyproject.toml").read_text(encoding="utf-8") - for major in SUPPORTED_CUDA_MAJORS: - m = re.search(rf'^cu{major} = \["cuda-bindings\[all\]([^"]+)"', pyproject, re.M) - assert m, f"no cu{major} extra in pyproject.toml" - assert m.group(1) == pip_requirement(major).removeprefix("cuda-bindings") +class TestConsistencyHook: + """toolshed/check_cuda_core_bindings_floor.py, also the pre-commit hook.""" + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_the_checkout_is_consistent(self, hook): + assert hook.check(REPO) == [] + assert hook.main(["--repo-root", str(REPO)]) == 0 -@pytest.mark.agent_authored(model="claude-fable-5-1") -def test_ci_toolkit_pins_match_the_floors_minor(): - """CI builds each major against the toolkit pinned in ci/versions.yml; the - build requires that header's major.minor to equal the bindings', so the - floor of each major must sit in the same minor as its toolkit pin.""" - versions = (REPO / "ci" / "versions.yml").read_text(encoding="utf-8") - pins = dict(re.findall(r"^\s+(build|prev_build):\s*\n\s+version:\s*\"(\d+\.\d+)", versions, re.M)) - assert set(pins) == {"build", "prev_build"}, pins - by_major = {int(v.split(".")[0]): v for v in pins.values()} - for major, floor in CUDA_BINDINGS_FLOOR.items(): - assert by_major[major] == f"{floor[0]}.{floor[1]}", ( - f"ci/versions.yml builds CUDA {major} against {by_major[major]} but the floor is {format_version(floor)}" + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_a_malformed_extra_fails_the_hook(self, hook, tmp_path, capsys): + # check() returns before reading ci/versions.yml or the docs, so the tree needs only these two files. + core = tmp_path / "cuda_core" / "cuda" / "core" + core.mkdir(parents=True) + shutil.copy(CUDA_CORE / "cuda" / "core" / "_bindings_floor.py", core) + (tmp_path / "cuda_core" / "pyproject.toml").write_text( + '[project.optional-dependencies]\ncu13 = ["cuda-bindings[all]>=13.4,==13.*"]\n', encoding="utf-8" ) + assert hook.main(["--repo-root", str(tmp_path)]) == 1 + assert "pyproject.toml: the 'cu13' extra must pin cuda-bindings" in capsys.readouterr().err + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_ci_toolkit_pins_must_sit_in_the_floors_minor(self, hook): + floors = {12: (12, 9, 8), 13: (13, 4, 1)} + good = 'cuda:\n build:\n version: "13.4.2"\n prev_build:\n version: "12.9.1"\n' + assert hook.ci_pin_problems(floors, good) == [] + stale = good.replace("13.4.2", "13.3.0") + (problem,) = hook.ci_pin_problems(floors, stale) + assert "cuda.build.version is 13.3 but the CUDA 13 floor is cuda-bindings 13.4.1" in problem + (problem,) = hook.ci_pin_problems({13: (13, 4, 1)}, good) + assert "pins CUDA 12, which has no cu12 extra" in problem + (problem,) = hook.ci_pin_problems(floors, 'cuda:\n build:\n version: "13.4.2"\n') + assert "no build or prev_build toolkit pin for CUDA 12" in problem + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_docs_must_use_the_substitutions(self, hook, tmp_path): + docs = tmp_path / "cuda_core" / "docs" / "source" + (docs / "release").mkdir(parents=True) + (docs / "support.rst").write_text("``cuda-bindings`` >= |cuda-bindings-floor-cu13|\n", encoding="utf-8") + (docs / "release" / "1.3.0-notes.rst").write_text("requires ``cuda-bindings`` >= 13.4.1\n", encoding="utf-8") + assert hook.docs_problems(tmp_path) == [] + (docs / "install.rst").write_text("Install ``cuda-bindings`` >= 13.4.1.\n", encoding="utf-8") + (docs / "api_nvml.rst").write_text("requires ``cuda-bindings`` 13.4.1 or later\n", encoding="utf-8") + problems = hook.docs_problems(tmp_path) + assert [p.split(":")[0] for p in problems] == [ + "cuda_core/docs/source/api_nvml.rst", + "cuda_core/docs/source/install.rst", + ] + assert problems[1].startswith("cuda_core/docs/source/install.rst:1:") diff --git a/cuda_core/tests/test_build_hooks.py b/cuda_core/tests/test_build_hooks.py index 3d9d4fc8541..fe15c7c38e1 100644 --- a/cuda_core/tests/test_build_hooks.py +++ b/cuda_core/tests/test_build_hooks.py @@ -483,7 +483,7 @@ class TestBuildConfigurationCheck: major.minor as that cuda-bindings. Anything else is a build error that names what was found and what is required.""" - FLOOR = build_hooks._load_bindings_floor().CUDA_BINDINGS_FLOOR + FLOOR = build_hooks._bindings_floors() @pytest.fixture(autouse=True) def _isolate_build_info(self, tmp_path, monkeypatch): @@ -506,6 +506,17 @@ def test_floor_bindings_and_matching_header_pass_and_are_recorded(self, tmp_path assert floor[0] * 1000 + floor[1] * 10 == info.CUDA_VERSION assert floor == info.CUDA_BINDINGS_FLOOR assert version == info.CUDA_BINDINGS_BUILD_VERSION + # ci/tools/cuda_core_bindings_floor.py reads this record out of the wheel (BINDINGS_SOURCE=floor). + tool_path = Path(__file__).resolve().parents[2] / "ci" / "tools" / "cuda_core_bindings_floor.py" + if tool_path.is_file(): # absent from an sdist tree + spec = importlib.util.spec_from_file_location("cuda_core_bindings_floor_tool", tool_path) + tool = importlib.util.module_from_spec(spec) + spec.loader.exec_module(tool) + text = build_hooks._BUILD_INFO_PATH.read_text(encoding="utf-8") + assert tool.floor_from_source(text, major) == f"{floor[0]}.{floor[1]}.{floor[2]}" + other = 25 - major + with pytest.raises(SystemExit, match=f"records a CUDA {major} build, not CUDA {other}"): + tool.floor_from_source(text, other) @pytest.mark.agent_authored(model="claude-fable-5-1") def test_bindings_below_the_floor_fail(self, tmp_path, monkeypatch): @@ -564,13 +575,45 @@ def test_unsupported_major_is_a_build_error(self, tmp_path, monkeypatch): build_hooks._check_build_configuration(cuda_path, "14") +class TestBindingsFloorsFromPyproject: + """_bindings_floors() reads the cu extras of pyproject.toml; a malformed extra fails the build.""" + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_malformed_extra_is_a_build_error(self, tmp_path, monkeypatch): + pyproject = tmp_path / "pyproject.toml" + pyproject.write_text( + '[project]\nname = "cuda-core"\n[project.optional-dependencies]\ncu13 = ["cuda-bindings>=13.4"]\n' + ) + monkeypatch.setattr(build_hooks, "_PYPROJECT_PATH", pyproject) + monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", "13") + build_hooks._bindings_floors.cache_clear() + build_hooks._determine_cuda_major_version.cache_clear() + try: + for call in ( + lambda: build_hooks._check_build_configuration(str(tmp_path), "13"), + build_hooks._get_cuda_bindings_require, + lambda: build_hooks._build_define_macros("13"), + ): + with pytest.raises(RuntimeError, match=r"pyproject\.toml: the 'cu13' extra must pin cuda-bindings"): + call() + finally: + build_hooks._bindings_floors.cache_clear() + build_hooks._determine_cuda_major_version.cache_clear() + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_the_checkout_declares_both_majors(self): + floors = build_hooks._bindings_floors() + assert list(floors) == [12, 13] + assert all(floor[0] == major for major, floor in floors.items()) + + class TestBuildRequirement: @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize("major", ["12", "13"]) def test_pins_the_floor_and_the_major(self, monkeypatch, major): monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", major) build_hooks._determine_cuda_major_version.cache_clear() - floor = build_hooks._load_bindings_floor().CUDA_BINDINGS_FLOOR[int(major)] + floor = build_hooks._bindings_floors()[int(major)] (requirement,) = build_hooks._get_cuda_bindings_require() assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},=={major}.*" @@ -588,7 +631,7 @@ class TestDefineMacros: @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize("major", ["12", "13"]) def test_major_and_floor_header_version(self, major): - floor = build_hooks._load_bindings_floor().CUDA_BINDINGS_FLOOR[int(major)] + floor = build_hooks._bindings_floors()[int(major)] assert build_hooks._build_define_macros(major) == [ ("CUDA_CORE_BUILD_MAJOR", major), ("CUDA_CORE_MIN_CUDA_VERSION", str(floor[0] * 1000 + floor[1] * 10)), diff --git a/toolshed/check_cuda_core_bindings_floor.py b/toolshed/check_cuda_core_bindings_floor.py new file mode 100644 index 00000000000..22d770eafb6 --- /dev/null +++ b/toolshed/check_cuda_core_bindings_floor.py @@ -0,0 +1,119 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Check that everything that depends on the cuda-bindings floor agrees with it. + +The floor of each CUDA major is declared once, by the `cu12`/`cu13` extras in +cuda_core/pyproject.toml (see cuda_core/cuda/core/_bindings_floor.py). Most +consumers read it from there, but two constraints cannot be derived and are +checked here, as a pre-commit hook and from cuda_core/tests/test_bindings_floor.py: + +1. The extras parse: each `cu` extra pins exactly + `cuda-bindings[...]>=..,==.*`. +2. ci/versions.yml builds each major against a CUDA Toolkit of the floor's + major.minor. The build requires the header's major.minor to equal the + cuda-bindings', so a toolkit pin in another minor cannot build the floor, + and a floor bump must move the pin (or the reverse). +3. No documentation page spells a floor out by hand; docs/source/conf.py + provides |cuda-bindings-floor-cu12| and |cuda-bindings-floor-cu13|. + Release notes are history and are exempt. + +Usage: python toolshed/check_cuda_core_bindings_floor.py [--repo-root DIR] +""" + +from __future__ import annotations + +import argparse +import importlib.util +import re +import sys +from pathlib import Path + +try: + import tomllib +except ModuleNotFoundError: # Python 3.10 + import tomli as tomllib # type: ignore[no-redef] + +REPO_ROOT = Path(__file__).resolve().parents[1] +FLOOR_MODULE = Path("cuda_core", "cuda", "core", "_bindings_floor.py") +PYPROJECT = Path("cuda_core", "pyproject.toml") +CI_VERSIONS = Path("ci", "versions.yml") +DOCS_SOURCE = Path("cuda_core", "docs", "source") + +_CI_PIN_RE = re.compile(r"^\s+(build|prev_build):\s*\n\s+version:\s*\"(\d+)\.(\d+)(?:\.\d+)?\"", re.M) +# `cuda-bindings >= 13.4.1`, ``cuda-bindings`` 13.4.1 or later, ... (release notes are exempt). +_HAND_WRITTEN_FLOOR_RE = re.compile(r"cuda-bindings[`'\" ]{0,4}(?:>=\s*)?\d+\.\d+\.\d+") + + +def load_floor_module(repo_root: Path): + path = repo_root / FLOOR_MODULE + spec = importlib.util.spec_from_file_location("_cuda_core_bindings_floor", path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def read_floors(repo_root: Path) -> dict[int, tuple[int, int, int]]: + """The floors declared by the pyproject extras; ValueError if they do not parse.""" + with open(repo_root / PYPROJECT, "rb") as f: + extras = tomllib.load(f)["project"]["optional-dependencies"] + return load_floor_module(repo_root).floors_from_extras(extras) + + +def ci_pin_problems(floors: dict[int, tuple[int, int, int]], versions_yml: str) -> list[str]: + """Toolkit pins in ci/versions.yml whose major.minor is not the floor's.""" + pins = {int(major): (int(major), int(minor), key) for key, major, minor in _CI_PIN_RE.findall(versions_yml)} + problems = [] + for major, floor in floors.items(): + if major not in pins: + problems.append(f"{CI_VERSIONS}: no build or prev_build toolkit pin for CUDA {major} (floor {floor})") + continue + pinned_major, pinned_minor, key = pins[major] + if (pinned_major, pinned_minor) != floor[:2]: + problems.append( + f"{CI_VERSIONS}: cuda.{key}.version is {pinned_major}.{pinned_minor} but the CUDA {major} floor " + f"is cuda-bindings {'.'.join(map(str, floor))}; the toolkit and the floor must share major.minor" + ) + for major in pins: + if major not in floors: + problems.append(f"{CI_VERSIONS}: pins CUDA {major}, which has no cu{major} extra in {PYPROJECT}") + return problems + + +def docs_problems(repo_root: Path) -> list[str]: + """Documentation pages (release notes excepted) that spell out a floor by hand.""" + problems = [] + for path in sorted((repo_root / DOCS_SOURCE).rglob("*.rst")): + if "release" in path.relative_to(repo_root / DOCS_SOURCE).parts: + continue + for number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1): + if _HAND_WRITTEN_FLOOR_RE.search(line): + problems.append( + f"{path.relative_to(repo_root).as_posix()}:{number}: spells out a cuda-bindings floor; " + "use the |cuda-bindings-floor-cu| substitution from conf.py" + ) + return problems + + +def check(repo_root: Path) -> list[str]: + try: + floors = read_floors(repo_root) + except ValueError as exc: + return [f"{PYPROJECT}: {exc}"] + problems = ci_pin_problems(floors, (repo_root / CI_VERSIONS).read_text(encoding="utf-8")) + problems += docs_problems(repo_root) + return problems + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--repo-root", type=Path, default=REPO_ROOT) + args = parser.parse_args(argv) + problems = check(args.repo_root) + for problem in problems: + print(problem, file=sys.stderr) + return 1 if problems else 0 + + +if __name__ == "__main__": + sys.exit(main()) From 18c5a348c6deb0b70550b439cc03431a929eb2b2 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Tue, 22 Sep 2026 14:20:35 -0700 Subject: [PATCH 10/18] cuda.core: compare headers, not version strings, in the cuda-bindings header rule (#2783) The build required cuda.h to match the installed cuda-bindings' version string by major.minor, and the import-time check demanded a cuda-bindings version at least as new as the build's header. Both fail on a cuda-bindings built from main right after a toolkit minor bump: its version string is the previous release's (13.4.3.devN) while it was generated from the new header. Compare headers instead: cuda.bindings.driver.CUDA_VERSION, present at both floors, is the CUDA_VERSION macro of the header cuda-bindings was generated from. The build check and check_installed_bindings() use it; the floor stays a version-string comparison; required_minimum() is gone. Isolated builds now request `cuda-bindings>=,==.*,<.
` when cuda.h is readable, so pip cannot pick a newer minor than the toolkit, while the development build above (old version string, new header) still resolves and reaches the header rule. The pre-commit hook lets the CI toolkit run ahead of the floor's minor (the toolkit-bump window) and rejects only a toolkit below it. AGENTS.md documents the bump order and the accepted red window for the rows that install the literal floor. __init__.py reads cuda.bindings.driver by module object: the `cuda` namespace package carries no `bindings` attribute when the submodule comes from sys.modules. Found by the new child-interpreter test that imports cuda.core against fake cuda-bindings and asserts each ImportError, which issue #2783 asked for. The wrong-major message no longer promises a build for the installed major, and the pip commands use double quotes so they work in cmd.exe. --- cuda_core/AGENTS.md | 28 ++++- cuda_core/build_hooks.py | 67 ++++++++--- cuda_core/cuda/core/__init__.py | 14 ++- cuda_core/cuda/core/_bindings_floor.py | 53 +++++---- cuda_core/docs/source/install.rst | 12 +- cuda_core/docs/source/support.rst | 17 +-- cuda_core/tests/test_bindings_floor.py | 131 +++++++++++++++++---- cuda_core/tests/test_build_hooks.py | 93 ++++++++++++--- toolshed/check_cuda_core_bindings_floor.py | 14 ++- 9 files changed, 329 insertions(+), 100 deletions(-) diff --git a/cuda_core/AGENTS.md b/cuda_core/AGENTS.md index 8c99b23b0f2..a749c6a9024 100644 --- a/cuda_core/AGENTS.md +++ b/cuda_core/AGENTS.md @@ -49,11 +49,12 @@ change that uses a `cuda-bindings` API newer than the old floor. To bump: 1. Edit the `cuda-bindings` pin in the `cu12` or `cu13` extra in `pyproject.toml`. That is the only version to type. -2. Align `ci/versions.yml`: the `build` (current major) and `prev_build` (prior - major) toolkit pins must share major.minor with the floors, or CI builds a - configuration the build rejects. The pre-commit hook - `check-cuda-core-bindings-floor` (`toolshed/check_cuda_core_bindings_floor.py`) - checks this, and that no documentation page spells a floor out by hand. +2. Check `ci/versions.yml`: the `build` (current major) and `prev_build` (prior + major) toolkit pins must not sit below the floors' major.minor, or CI builds a + configuration the build rejects; a toolkit ahead of the floor is the bump + window described below. The pre-commit hook `check-cuda-core-bindings-floor` + (`toolshed/check_cuda_core_bindings_floor.py`) checks this, and that no + documentation page spells a floor out by hand. 3. Add a "Breaking Changes" entry to the release notes naming the new floors. The support-policy table in `docs/source/support.rst` reads the floors and the release version at docs-build time; there is nothing to edit there. @@ -61,6 +62,21 @@ change that uses a `cuda-bindings` API newer than the old floor. To bump: 5. Pin `cuda-bindings` accordingly in the conda-forge `cuda-core` feedstock (outside this repository). +### CUDA Toolkit minor bumps + +The build compares the toolkit's `cuda.h` with the header the installed +`cuda-bindings` was generated from (`cuda.bindings.driver.CUDA_VERSION`), not +with its version string, so the commit that moves `ci/versions.yml` to a new +minor builds `cuda.core` against the `cuda-bindings` built from the same +commit (isolated builds request `cuda-bindings` no newer than the toolkit's +minor, a cap that admits that development build). The import-time check +compares headers the same way. The floor stays +where it is until a `cuda-bindings` release of the new minor exists on PyPI; +until then the CI rows that install the literal floor (`BINDINGS_SOURCE=floor`) +fail the header check and stay red. That window is accepted; do not add +fallback logic for it. Order: bump the toolkit, release `cuda-bindings` for the +new minor, then bump the floor here. + ## Testing expectations - **Primary tests**: `pytest tests/` @@ -178,7 +194,7 @@ below are for contributors. Reviewers and agents should flag violations. uses `warnings.warn(..., CUDAWarning)`; a CUDA callback thread does nothing that needs the GIL and hands its work to the deferred-cleanup queue (`Py_AddPendingCall` is GIL-free and allowed there). The table in - `_cpp/DESIGN.md` ("Which channel to use") spells this out. + `_cpp/rt/DESIGN.md` ("Which channel to use") spells this out. - **`pw_*` runs user Python**: a `p_` pointer only calls the driver; its `pw_` twin also acquires the GIL on failure and runs the warning filters, `showwarning`, or `sys.unraisablehook`, any of which may call back into diff --git a/cuda_core/build_hooks.py b/cuda_core/build_hooks.py index 9e0d43da5f8..62676a0cacc 100644 --- a/cuda_core/build_hooks.py +++ b/cuda_core/build_hooks.py @@ -97,6 +97,20 @@ def _import_cuda_bindings(): return cuda.bindings +def _installed_cuda_bindings() -> tuple: + """(version string, CUDA_VERSION) of the cuda-bindings in the build environment. + + ``cuda.bindings.driver.CUDA_VERSION`` is the ``CUDA_VERSION`` macro of the + ``cuda.h`` the installed cuda-bindings was generated from (e.g. 13040). + Unlike the version string, which a development build inherits from the + previous release's tag, it is exact, so the header rule compares against + it. Raises ModuleNotFoundError when no cuda-bindings is installed. + """ + bindings = _import_cuda_bindings() + driver = importlib.import_module("cuda.bindings.driver") + return bindings.__version__, int(driver.CUDA_VERSION) + + @functools.cache def _get_cuda_path() -> str: get_cuda_path_or_home = _import_get_cuda_path_or_home() @@ -224,9 +238,9 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: cuda.core supports one configuration per CUDA major series: the installed cuda-bindings is at least the series' floor (the cu extra in - pyproject.toml) and the cuda.h it compiles against has the same major.minor as that - cuda-bindings, which is the header cuda-bindings itself was generated from. - The pip build requirement (get_requires_for_build_*) states the floor, but + pyproject.toml) and the cuda.h it compiles against is, by major.minor, the + header that cuda-bindings was generated from (its driver.CUDA_VERSION). + The pip build requirement (get_requires_for_build_*) states both, but conda-forge, pixi and --no-build-isolation installs bypass it, so the check lives here, where every build path passes. @@ -240,7 +254,7 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: requirement = floor.bindings_requirement(floor_triple) try: - bindings_version = _import_cuda_bindings().__version__ + bindings_version, bindings_cuda_version = _installed_cuda_bindings() except ModuleNotFoundError as exc: raise RuntimeError( f"cuda.core requires cuda-bindings to build (install '{requirement}'). " @@ -264,13 +278,15 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: ) cuda_version = _read_cuda_h_version(cuda_path) - header = (cuda_version // 1000, cuda_version // 10 % 100) - if header != bindings[:2]: + header = floor.header_minor(cuda_version) + generated_from = floor.header_minor(bindings_cuda_version) + if header != generated_from: raise RuntimeError( f"cuda.h under {cuda_path} is CUDA {header[0]}.{header[1]}, but the installed cuda-bindings " - f"is {bindings_version}. cuda.core must be built against a cuda.h of the same " - "major.minor as its cuda-bindings (the header cuda-bindings was generated from). " - "Point CUDA_PATH or CUDA_HOME at a matching CUDA Toolkit, or install matching cuda-bindings." + f"{bindings_version} was generated from CUDA {generated_from[0]}.{generated_from[1]} headers. " + "cuda.core must be built against the cuda.h its cuda-bindings was generated from. Point " + "CUDA_PATH or CUDA_HOME at that CUDA Toolkit, or install a matching cuda-bindings (for an " + "isolated build, constrain it with PIP_CONSTRAINT or build with --no-build-isolation)." ) print(f"Build configuration: CUDA {header[0]}.{header[1]} headers, cuda-bindings {bindings_version}") _write_build_info(major, cuda_version, floor_triple, bindings_version) @@ -290,15 +306,16 @@ def _write_build_info(cuda_major: int, cuda_version: int, floor: tuple, bindings cuda/core/__init__.py reads this module before it selects the versioned subpackage and refuses an installed cuda-bindings older than the floor or - older, by minor, than the header (see _bindings_floor.required_minimum). - Like _version.py, the file is generated, gitignored, and shipped. + generated from an older header minor than the build's (see + _bindings_floor.check_installed_bindings). Like _version.py, the file is + generated, gitignored, and shipped. """ _BUILD_INFO_PATH.write_text( "# Generated by build_hooks.py at build time. Do not edit or commit.\n" f"CUDA_MAJOR = {cuda_major}\n" f"CUDA_VERSION = {cuda_version} # the cuda.h this build compiled against\n" f"CUDA_BINDINGS_FLOOR = {tuple(floor)!r}\n" - f"CUDA_BINDINGS_BUILD_VERSION = {bindings_version!r}\n", + f"CUDA_BINDINGS_BUILD_VERSION = {bindings_version!r} # informational; not read at import\n", encoding="utf-8", ) @@ -627,12 +644,28 @@ def build_wheel(wheel_directory, config_settings=None, metadata_directory=None): def _get_cuda_bindings_require(): - """The cuda-bindings build requirement: the floor of the CUDA major being built. - - Honored by isolated builds only; _check_build_configuration() enforces the - same rule for every other build path. + """The cuda-bindings build requirement for isolated builds. + + The floor of the CUDA major being built, capped at the header's minor when + cuda.h is readable: pip would otherwise install the newest cuda-bindings of + the major, which a newer minor on PyPI turns into a header mismatch that an + isolated build cannot fix from the outside. A cap rather than a pin, because + a cuda-bindings built from main right after a toolkit bump carries the + previous release's version string (13.4.3.devN) with the new header; the + header rule in _check_build_configuration() judges it, not pip. When the + header's minor is below the floor's, the floor alone is requested so that + the configuration check reports the mismatch in its own words. Honored by + isolated builds only; the configuration check covers every build path. """ - return [_load_bindings_floor().bindings_requirement(_floor_for(_determine_cuda_major_version()))] + floor = _load_bindings_floor() + floor_triple = _floor_for(_determine_cuda_major_version()) + try: + header = floor.header_minor(_read_cuda_h_version(_get_cuda_path())) + except RuntimeError: + return [floor.bindings_requirement(floor_triple)] + if header[0] != floor_triple[0] or header[1] < floor_triple[1]: + return [floor.bindings_requirement(floor_triple)] + return [f"{floor.bindings_requirement(floor_triple)},<{header[0]}.{header[1] + 1}"] def get_requires_for_build_editable(config_settings=None): diff --git a/cuda_core/cuda/core/__init__.py b/cuda_core/cuda/core/__init__.py index 065d4893b0e..83044801c6c 100644 --- a/cuda_core/cuda/core/__init__.py +++ b/cuda_core/cuda/core/__init__.py @@ -13,8 +13,9 @@ def _import_versioned_module() -> None: build carries one build at the top level. Each build records the CUDA header it compiled against and its cuda-bindings floor in ``_build_info`` (generated by build_hooks.py). The installed cuda-bindings must be of the - build's major and at least as new as the build's minimum - (see ``_bindings_floor.required_minimum``), or import fails here with an + build's major, at least as new as the floor, and generated from a + ``cuda.h`` at least as new as the build's (see + ``_bindings_floor.check_installed_bindings``), or import fails here with an actionable message instead of later with a missing C function or a silently disabled feature. """ @@ -49,8 +50,15 @@ def load_build_module(name: str, cuda_major: int): raise ImportError( f"this cuda.core installation has no build for CUDA {cuda_major} (installed cuda-bindings: {version_str})" ) from exc + # By module object: the `cuda` namespace package need not carry a `bindings` attribute. + bindings_driver = importlib.import_module("cuda.bindings.driver") floor.check_installed_bindings( - version_str, info.CUDA_MAJOR, info.CUDA_VERSION, info.CUDA_BINDINGS_FLOOR, __version__ + version_str, + int(bindings_driver.CUDA_VERSION), + info.CUDA_MAJOR, + info.CUDA_VERSION, + info.CUDA_BINDINGS_FLOOR, + __version__, ) subdir = f"cu{cuda_major}" diff --git a/cuda_core/cuda/core/_bindings_floor.py b/cuda_core/cuda/core/_bindings_floor.py index 7b5e3d2cf0b..28b274fabc3 100644 --- a/cuda_core/cuda/core/_bindings_floor.py +++ b/cuda_core/cuda/core/_bindings_floor.py @@ -16,8 +16,9 @@ Everything else derives from them through :func:`floors_from_extras`: - the build backend (``build_hooks.py``) checks the installed cuda-bindings - and header against the floor and records the floor of the build in the - generated ``_build_info.py``; + against the floor, checks that the ``cuda.h`` it compiles against is the one + that cuda-bindings was generated from, and records the floor and the header + in the generated ``_build_info.py``; - ``cuda/core/__init__.py`` checks the installed cuda-bindings against that record with :func:`check_installed_bindings`; - the documentation reads the floors into substitutions (``docs/source/conf.py``); @@ -41,7 +42,6 @@ "floors_from_extras", "format_version", "release_triple", - "required_minimum", ] _RELEASE_RE = re.compile(r"^(\d+)\.(\d+)\.(\d+)") @@ -126,20 +126,14 @@ def bindings_requirement(floor: tuple[int, int, int]) -> str: return f"cuda-bindings>={format_version(floor)},=={floor[0]}.*" -def required_minimum(floor: tuple[int, int, int], header_cuda_version: int) -> tuple[int, int, int]: - """The minimum cuda-bindings a build accepts at run time. - - A build accepts the floor of its major series, and never a cuda-bindings - whose minor is older than the ``cuda.h`` the build compiled against: the - driver function-pointer keys the C++ layer looks up in cuda-bindings are - derived from that header's macros, so an older minor may lack them. - """ - header_minor = (header_cuda_version // 1000, header_cuda_version // 10 % 100, 0) - return max(floor, header_minor) +def header_minor(cuda_version: int) -> tuple[int, int]: + """The (major, minor) of a ``CUDA_VERSION`` macro value: 13040 -> (13, 4).""" + return cuda_version // 1000, cuda_version // 10 % 100 def check_installed_bindings( installed_version: str, + installed_cuda_version: int, build_cuda_major: int, build_cuda_version: int, build_floor: tuple[int, int, int], @@ -147,10 +141,18 @@ def check_installed_bindings( ) -> tuple[int, int, int]: """Validate the installed cuda-bindings against a build; return its triple. - ``build_cuda_major``, ``build_cuda_version`` and ``build_floor`` are the - build's record in ``_build_info.py``. Raises ImportError with an actionable - message when the installed cuda-bindings is not a release, is not of the - major this build was compiled for, or is older than the build's minimum. + ``installed_version`` and ``installed_cuda_version`` are the installed + cuda-bindings' ``__version__`` and ``driver.CUDA_VERSION`` (the ``cuda.h`` + it was generated from, e.g. 13040); ``build_cuda_major``, + ``build_cuda_version`` and ``build_floor`` are the build's record in + ``_build_info.py``. Raises ImportError with an actionable message when the + installed cuda-bindings is not a release, is not of the major this build + was compiled for, is older than the floor, or was generated from an older + ``cuda.h`` minor than the build compiled against: the driver function + table the C++ layer looks up in cuda-bindings is keyed by that header's + macros, so an older minor may lack entries. Headers are compared as + ``CUDA_VERSION`` values, not version strings, because a development build + of cuda-bindings carries the previous release's version string. """ installed = release_triple(installed_version) if installed is None: @@ -160,13 +162,20 @@ def check_installed_bindings( raise ImportError( f"this cuda.core {core_version} build is for CUDA {build_cuda_major}, but the installed " f"cuda-bindings is {installed_version}. Install cuda-bindings {build_cuda_major}.x " - f"(pip install 'cuda-bindings=={build_cuda_major}.*'), or a cuda.core build for CUDA {major} if one exists." + f'(pip install "cuda-bindings=={build_cuda_major}.*"), or a cuda.core build for CUDA {major} if one exists.' ) - minimum = required_minimum(build_floor, build_cuda_version) - if installed < minimum: - floor = format_version(minimum) + if installed < tuple(build_floor): + floor = format_version(build_floor) raise ImportError( f"cuda.core {core_version} requires cuda-bindings >= {floor} for CUDA {major} " - f"(found {installed_version}). Upgrade with: pip install -U 'cuda-bindings>={floor},=={major}.*'" + f'(found {installed_version}). Upgrade with: pip install -U "cuda-bindings>={floor},=={major}.*"' + ) + built_against, generated_from = header_minor(build_cuda_version), header_minor(installed_cuda_version) + if generated_from < built_against: + needed = f"{built_against[0]}.{built_against[1]}" + raise ImportError( + f"cuda.core {core_version} was built against CUDA {needed} headers, but the installed cuda-bindings " + f"{installed_version} was generated from CUDA {generated_from[0]}.{generated_from[1]} headers. " + f'Install cuda-bindings {needed} or newer: pip install -U "cuda-bindings>={needed}.0,=={major}.*"' ) return installed diff --git a/cuda_core/docs/source/install.rst b/cuda_core/docs/source/install.rst index 3fdb8f71cc4..f241793b318 100644 --- a/cuda_core/docs/source/install.rst +++ b/cuda_core/docs/source/install.rst @@ -71,8 +71,9 @@ Same as above, ``cuda.core`` can be installed in a CUDA 12 or 13 environment. Fo and likewise use ``cuda-version=13`` for CUDA 13. -The conda-forge package pins ``cuda-bindings`` to the version it was built against, so a -compatible ``cuda-bindings`` is installed alongside it. +The conda-forge package depends on ``cuda-bindings`` of the same CUDA major; the +``cuda-bindings`` floor of the release (see :ref:`cuda-core-bindings-floor`) applies to it as +well. Development environment @@ -162,9 +163,12 @@ Installing from Source A source build requires two things to agree (see :ref:`cuda-core-bindings-floor`): - ``cuda-bindings`` 12.x or 13.x at or above the release's floor for that major. An isolated - build (the default ``pip install``) installs it; other builds must provide it. + build (the default ``pip install``) installs one no newer than the toolkit's minor; other + builds must provide it. - A CUDA Toolkit, located through ``CUDA_PATH`` or ``CUDA_HOME``, whose ``cuda.h`` has the same - major.minor as that ``cuda-bindings``. The build fails early otherwise. + major.minor as the header that ``cuda-bindings`` was generated from. The build fails early + otherwise. To build against a particular ``cuda-bindings`` in an isolated build, constrain it + with ``PIP_CONSTRAINT``; or use ``--no-build-isolation`` with it installed. .. note:: diff --git a/cuda_core/docs/source/support.rst b/cuda_core/docs/source/support.rst index 3ef59f931ba..a7c554f3cab 100644 --- a/cuda_core/docs/source/support.rst +++ b/cuda_core/docs/source/support.rst @@ -87,20 +87,23 @@ there. - ``cuda-bindings`` >= |cuda-bindings-floor-cu13| - **At run time**, ``import cuda.core`` requires an installed ``cuda-bindings`` of the same major - as the ``cuda-core`` build in use and at least as new as that build's floor. An older + as the ``cuda-core`` build in use, at least as new as that build's floor, and generated from a + ``cuda.h`` at least as new (by major.minor) as the one the build compiled against; the published + wheels are built against the floor's header, so the floor alone satisfies them. An older ``cuda-bindings`` fails at import with a message that names the version found, the version required, and the ``pip`` command that fixes it. A newer ``cuda-bindings`` of the same major is supported. - **At build time**, a source build requires ``cuda-bindings`` at or above the floor and a - ``cuda.h`` (``CUDA_PATH`` or ``CUDA_HOME``) of the same major.minor as that ``cuda-bindings``, - which is the header ``cuda-bindings`` itself was generated from. Any other configuration fails - the build with a message that names what was found and what is required. Building against an - older CUDA Toolkit than the floor's minor is not supported. + ``cuda.h`` (``CUDA_PATH`` or ``CUDA_HOME``) of the same major.minor as the header that + ``cuda-bindings`` was generated from. Any other configuration fails the build with a message + that names what was found and what is required. Building against an older CUDA Toolkit than + the floor's minor is not supported. - **The CUDA driver** is unaffected. Feature availability is decided by the driver alone: a feature the installed driver lacks raises when it is used, as before. -A floor is raised only in a release that needs a newer ``cuda-bindings`` API, and every such -change is listed under "Breaking Changes" in the :doc:`release notes `. +A floor moves with each ``cuda-core`` release, to the newest ``cuda-bindings`` of each major at +that time, and in any release whose changes need a newer ``cuda-bindings`` API. Every move is +listed under "Breaking Changes" in the :doc:`release notes `. Python Version Support ---------------------- diff --git a/cuda_core/tests/test_bindings_floor.py b/cuda_core/tests/test_bindings_floor.py index 7810e0de727..38dfd7471db 100644 --- a/cuda_core/tests/test_bindings_floor.py +++ b/cuda_core/tests/test_bindings_floor.py @@ -15,8 +15,12 @@ """ import importlib.util +import os import re import shutil +import subprocess +import sys +import textwrap from pathlib import Path import pytest @@ -28,8 +32,8 @@ cuda_version_of, floors_from_extras, format_version, + header_minor, release_triple, - required_minimum, ) CUDA_CORE = Path(__file__).resolve().parent.parent @@ -133,24 +137,24 @@ def test_formatting_helpers(): @pytest.mark.agent_authored(model="claude-fable-5-1") -def test_required_minimum_is_the_floor_or_the_header_minor(): - floor = (13, 4, 1) - header_at_floor = cuda_version_of(floor) - assert required_minimum(floor, header_at_floor) == floor - # A build against a newer header than the floor's minor demands that minor: - # the driver-pointer keys are derived from the header's macros. - assert required_minimum(floor, header_at_floor + 10) == (13, 5, 0) - # An older header cannot win over the floor. - assert required_minimum(floor, 13000) == floor +def test_header_minor(): + assert header_minor(13040) == (13, 4) + assert header_minor(12090) == (12, 9) + assert header_minor(13000) == (13, 0) class TestCheckInstalledBindings: FLOOR = (13, 4, 1) HEADER = cuda_version_of(FLOOR) - def check(self, installed, major=13, header=None, floor=None): + def check(self, installed, major=13, header=None, floor=None, installed_header=None): + """installed_header defaults to the header of the installed version's major.minor, + as a release of that version would have been generated from.""" + if installed_header is None: + triple = release_triple(installed) or (major, 0, 0) + installed_header = cuda_version_of(triple) return check_installed_bindings( - installed, major, self.HEADER if header is None else header, floor or self.FLOOR, "1.3.0" + installed, installed_header, major, self.HEADER if header is None else header, floor or self.FLOOR, "1.3.0" ) @pytest.mark.agent_authored(model="claude-fable-5-1") @@ -173,24 +177,38 @@ def test_rejects_older_than_the_floor_with_the_fix(self): message = str(excinfo.value) assert "requires cuda-bindings >= 13.4.1 for CUDA 13" in message assert "(found 13.3.1)" in message - assert "pip install -U 'cuda-bindings>=13.4.1,==13.*'" in message + assert 'pip install -U "cuda-bindings>=13.4.1,==13.*"' in message # double quotes: cmd.exe too @pytest.mark.agent_authored(model="claude-fable-5-1") - def test_rejects_a_minor_older_than_the_header(self): - # Built against a header one minor above the floor; the floor itself no longer suffices. - with pytest.raises(ImportError, match=r"requires cuda-bindings >= 13\.5\.0"): + def test_rejects_bindings_generated_from_an_older_header_than_the_build(self): + # Built against 13.5 headers; a 13.4-generated cuda-bindings lacks table entries. + with pytest.raises(ImportError) as excinfo: self.check("13.4.1", header=self.HEADER + 10) + message = str(excinfo.value) + assert "was built against CUDA 13.5 headers" in message + assert "13.4.1 was generated from CUDA 13.4 headers" in message + assert 'pip install -U "cuda-bindings>=13.5.0,==13.*"' in message + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_header_rule_compares_headers_not_version_strings(self): + # A development cuda-bindings carries the previous release's version string + # (13.4.2.dev5) but was generated from the new 13.5 header: accepted. + floor = (13, 4, 2) + assert self.check("13.4.2.dev5+gabc", header=13050, floor=floor, installed_header=13050) == (13, 4, 2) + # The converse, a 13.5 version string generated from 13.4 headers, is rejected. + with pytest.raises(ImportError, match="was generated from CUDA 13.4 headers"): + self.check("13.5.0", header=13050, floor=floor, installed_header=13040) @pytest.mark.agent_authored(model="claude-fable-5-1") def test_the_floor_comes_from_the_build_record(self): # A build recorded with a lower floor (and header) accepts what the default floor rejects. assert self.check("13.3.0", header=cuda_version_of((13, 2, 0)), floor=(13, 2, 0)) == (13, 3, 0) - # A higher recorded floor rejects what the default floor accepts, even with the header at 13.4. + # A higher recorded floor rejects what the default floor accepts. with pytest.raises(ImportError) as excinfo: self.check("13.4.1", floor=(13, 5, 0)) message = str(excinfo.value) assert "requires cuda-bindings >= 13.5.0 for CUDA 13" in message - assert "pip install -U 'cuda-bindings>=13.5.0,==13.*'" in message + assert 'pip install -U "cuda-bindings>=13.5.0,==13.*"' in message @pytest.mark.agent_authored(model="claude-fable-5-1") def test_rejects_another_major_than_the_build(self): @@ -205,7 +223,7 @@ def test_another_major_names_only_the_fix_it_knows(self, installed): with pytest.raises(ImportError) as excinfo: self.check(installed, major=12, header=12090, floor=(12, 9, 8)) message = str(excinfo.value) - assert "Install cuda-bindings 12.x (pip install 'cuda-bindings==12.*')" in message + assert 'Install cuda-bindings 12.x (pip install "cuda-bindings==12.*")' in message assert f"build for CUDA {installed.split('.')[0]} if one exists" in message @pytest.mark.agent_authored(model="claude-fable-5-1") @@ -217,6 +235,78 @@ def test_rejects_unparseable_versions(self, installed): self.check(installed) +class TestImportTimeCheck: + """`import cuda.core` runs check_installed_bindings against the installed build's record + before importing any extension module. A fake cuda.bindings in a child interpreter + exercises the reject paths end to end (issue #2783 asked for this test).""" + + _CHILD = textwrap.dedent(""" + import sys, types + fake = types.ModuleType("cuda.bindings") + fake.__version__ = {version!r} + driver = types.ModuleType("cuda.bindings.driver") + driver.CUDA_VERSION = {cuda_version} + fake.driver = driver + sys.modules["cuda.bindings"] = fake + sys.modules["cuda.bindings.driver"] = driver + try: + import cuda.core + except ImportError as exc: + print("IMPORTERROR:", exc) + raise SystemExit(0) + raise SystemExit("cuda.core imported with a fake cuda-bindings " + fake.__version__) + """) + + @staticmethod + def _build(): + from cuda.core import _build_info + + return _build_info.CUDA_MAJOR, _build_info.CUDA_VERSION, tuple(_build_info.CUDA_BINDINGS_FLOOR) + + def _import_error(self, version, cuda_version, tmp_path): + env = {k: v for k, v in os.environ.items() if k != "PYTHONPATH"} # the installed build, not a source tree + result = subprocess.run( # noqa: S603 + [sys.executable, "-c", self._CHILD.format(version=version, cuda_version=cuda_version)], + cwd=tmp_path, + env=env, + capture_output=True, + text=True, + stdin=subprocess.DEVNULL, + timeout=120, + check=False, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert result.stdout.startswith("IMPORTERROR:"), result.stdout + return result.stdout + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_below_the_floor_fails_at_import_with_the_fix(self, tmp_path): + major, cuda_version, floor = self._build() + below = f"{major}.0.1" + message = self._import_error(below, major * 1000, tmp_path) + assert f"requires cuda-bindings >= {format_version(floor)} for CUDA {major}" in message + assert f'pip install -U "cuda-bindings>={format_version(floor)},=={major}.*"' in message + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_bindings_from_an_older_header_fail_at_import(self, tmp_path): + major, cuda_version, floor = self._build() + # At the floor by version, but generated from a header one minor below the build's. + message = self._import_error(format_version(floor), cuda_version - 10, tmp_path) + assert f"was built against CUDA {major}.{header_minor(cuda_version)[1]} headers" in message + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_unparseable_version_fails_at_import(self, tmp_path): + major, cuda_version, floor = self._build() + # A major but no release triple. The build's major keeps the merged wheel on its cu + # build; a foreign major (a shallow clone's 0.1.dev1) stops earlier there with "no build for CUDA 0". + no_triple = f"{major}.4" + message = self._import_error(no_triple, cuda_version, tmp_path) + assert f"a cuda-bindings {major}.x release is required (found {no_triple})" in message + # No major at all. + message = self._import_error("garbage", cuda_version, tmp_path) + assert "a cuda-bindings release must be installed (found version 'garbage')" in message + + class TestConsistencyHook: """toolshed/check_cuda_core_bindings_floor.py, also the pre-commit hook.""" @@ -238,10 +328,11 @@ def test_a_malformed_extra_fails_the_hook(self, hook, tmp_path, capsys): assert "pyproject.toml: the 'cu13' extra must pin cuda-bindings" in capsys.readouterr().err @pytest.mark.agent_authored(model="claude-fable-5-1") - def test_ci_toolkit_pins_must_sit_in_the_floors_minor(self, hook): + def test_ci_toolkit_pins_must_not_sit_below_the_floors_minor(self, hook): floors = {12: (12, 9, 8), 13: (13, 4, 1)} good = 'cuda:\n build:\n version: "13.4.2"\n prev_build:\n version: "12.9.1"\n' assert hook.ci_pin_problems(floors, good) == [] + assert hook.ci_pin_problems(floors, good.replace("13.4.2", "13.5.0")) == [] # the toolkit-bump window stale = good.replace("13.4.2", "13.3.0") (problem,) = hook.ci_pin_problems(floors, stale) assert "cuda.build.version is 13.3 but the CUDA 13 floor is cuda-bindings 13.4.1" in problem diff --git a/cuda_core/tests/test_build_hooks.py b/cuda_core/tests/test_build_hooks.py index fe15c7c38e1..a87d222b326 100644 --- a/cuda_core/tests/test_build_hooks.py +++ b/cuda_core/tests/test_build_hooks.py @@ -456,18 +456,23 @@ def test_serial_builds_and_compilers_without_the_hook_keep_the_stock_path(self, assert cmd.compiler.compile(["a.cpp"]) == "stock" -def _fake_bindings(monkeypatch, version): - """Make the build see an installed cuda-bindings of `version` (None: not installed).""" - import types +def _fake_bindings(monkeypatch, version, cuda_version=None): + """Make the build see an installed cuda-bindings of `version` (None: not installed), + generated from the header `cuda_version` (default: the header of its major.minor).""" - def import_cuda_bindings(): + def installed_cuda_bindings(): if version is None: raise ModuleNotFoundError("No module named 'cuda.bindings'", name="cuda.bindings") - module = types.ModuleType("cuda.bindings") - module.__version__ = version - return module + if cuda_version is not None: + return version, cuda_version + major, minor = (int(part) for part in version.split(".")[:2]) + return version, major * 1000 + minor * 10 - monkeypatch.setattr(build_hooks, "_import_cuda_bindings", import_cuda_bindings) + monkeypatch.setattr(build_hooks, "_installed_cuda_bindings", installed_cuda_bindings) + + +def _floor_str(major): + return ".".join(str(part) for part in build_hooks._bindings_floors()[major]) def _write_cuda_h(tmp_path, cuda_version): @@ -529,30 +534,41 @@ def test_bindings_below_the_floor_fail(self, tmp_path, monkeypatch): @pytest.mark.agent_authored(model="claude-fable-5-1") def test_bindings_of_another_major_fail(self, tmp_path, monkeypatch): - _fake_bindings(monkeypatch, "13.4.1") + _fake_bindings(monkeypatch, _floor_str(13)) cuda_path = _write_cuda_h(tmp_path, 12090) with pytest.raises( - RuntimeError, match="Building cuda.core for CUDA 12, but the installed cuda-bindings is 13.4.1" + RuntimeError, match=f"Building cuda.core for CUDA 12, but the installed cuda-bindings is {_floor_str(13)}" ): build_hooks._check_build_configuration(cuda_path, "12") @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize("header", [13030, 13050, 12090]) def test_header_minor_must_match_bindings(self, tmp_path, monkeypatch, header): - _fake_bindings(monkeypatch, "13.4.1") + _fake_bindings(monkeypatch, _floor_str(13)) cuda_path = _write_cuda_h(tmp_path, header) - with pytest.raises(RuntimeError, match="same major.minor as its cuda-bindings"): + with pytest.raises(RuntimeError, match="must be built against the cuda.h its cuda-bindings was generated from"): build_hooks._check_build_configuration(cuda_path, "13") @pytest.mark.agent_authored(model="claude-fable-5-1") def test_header_is_read_even_when_the_major_override_is_set(self, tmp_path, monkeypatch): # CUDA_CORE_BUILD_MAJOR skips header detection of the major, not this check. monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", "13") - _fake_bindings(monkeypatch, "13.4.1") + _fake_bindings(monkeypatch, _floor_str(13)) cuda_path = _write_cuda_h(tmp_path, 13030) - with pytest.raises(RuntimeError, match="same major.minor"): + with pytest.raises(RuntimeError, match="was generated from CUDA 13.4 headers"): build_hooks._check_build_configuration(cuda_path, "13") + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_development_bindings_pass_on_their_generated_header(self, tmp_path, monkeypatch): + """The header rule compares headers, not version strings: a cuda-bindings built from + main after a toolkit bump still carries the previous release's version string.""" + floor = self.FLOOR[13] + new_header = floor[0] * 1000 + (floor[1] + 1) * 10 + _fake_bindings(monkeypatch, f"{floor[0]}.{floor[1]}.{floor[2]}.dev5+gabcdef0", cuda_version=new_header) + cuda_path = _write_cuda_h(tmp_path, new_header) + build_hooks._check_build_configuration(cuda_path, "13") + assert f"CUDA_VERSION = {new_header}" in build_hooks._BUILD_INFO_PATH.read_text() + @pytest.mark.agent_authored(model="claude-fable-5-1") def test_missing_bindings_is_a_build_error(self, tmp_path, monkeypatch): _fake_bindings(monkeypatch, None) # no cuda-bindings in the build environment @@ -608,15 +624,62 @@ def test_the_checkout_declares_both_majors(self): class TestBuildRequirement: + """get_requires_for_build_wheel pins cuda-bindings for isolated builds: the floor, and the + header's minor when cuda.h is readable, so pip cannot pick a newer minor than the toolkit.""" + + @staticmethod + def _no_cuda_path(): + raise RuntimeError("no CUDA") + @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize("major", ["12", "13"]) - def test_pins_the_floor_and_the_major(self, monkeypatch, major): + def test_pins_the_floor_and_the_major_without_a_header(self, monkeypatch, major): monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", major) + monkeypatch.setattr(build_hooks, "_get_cuda_path", self._no_cuda_path) build_hooks._determine_cuda_major_version.cache_clear() floor = build_hooks._bindings_floors()[int(major)] (requirement,) = build_hooks._get_cuda_bindings_require() assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},=={major}.*" + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_caps_at_the_headers_minor_when_cuda_h_is_readable(self, tmp_path, monkeypatch): + from packaging.specifiers import SpecifierSet + + monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", "13") + build_hooks._determine_cuda_major_version.cache_clear() + floor = build_hooks._bindings_floors()[13] + cuda_path = _write_cuda_h(tmp_path, floor[0] * 1000 + (floor[1] + 1) * 10) + monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: cuda_path) + (requirement,) = build_hooks._get_cuda_bindings_require() + assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},==13.*,<13.{floor[1] + 2}" + specifiers = SpecifierSet(requirement.removeprefix("cuda-bindings")) + # The dev cuda-bindings built alongside a toolkit bump still carries the old minor's version string. + assert specifiers.contains(f"13.{floor[1]}.{floor[2] + 1}.dev133", prereleases=True) + assert specifiers.contains(f"13.{floor[1] + 1}.0") + assert not specifiers.contains(f"13.{floor[1] + 2}.0") # a newer minor on PyPI stays out + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_requests_the_floor_alone_when_the_header_is_of_another_major(self, tmp_path, monkeypatch): + # CUDA_CORE_BUILD_MAJOR=13 with a CUDA 12 toolkit: the configuration check reports the mismatch. + monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", "13") + build_hooks._determine_cuda_major_version.cache_clear() + floor = build_hooks._bindings_floors()[13] + cuda_path = _write_cuda_h(tmp_path, 12090) + monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: cuda_path) + (requirement,) = build_hooks._get_cuda_bindings_require() + assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},==13.*" + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_requests_the_floor_alone_when_the_header_is_below_it(self, tmp_path, monkeypatch): + # The configuration check reports this mismatch in its own words. + monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", "13") + build_hooks._determine_cuda_major_version.cache_clear() + floor = build_hooks._bindings_floors()[13] + cuda_path = _write_cuda_h(tmp_path, floor[0] * 1000 + (floor[1] - 1) * 10) + monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: cuda_path) + (requirement,) = build_hooks._get_cuda_bindings_require() + assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},==13.*" + @pytest.mark.agent_authored(model="claude-fable-5-1") def test_unsupported_major_names_the_supported_ones(self, monkeypatch): monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", "11") diff --git a/toolshed/check_cuda_core_bindings_floor.py b/toolshed/check_cuda_core_bindings_floor.py index 22d770eafb6..0fd9db023a5 100644 --- a/toolshed/check_cuda_core_bindings_floor.py +++ b/toolshed/check_cuda_core_bindings_floor.py @@ -10,10 +10,11 @@ 1. The extras parse: each `cu` extra pins exactly `cuda-bindings[...]>=..,==.*`. -2. ci/versions.yml builds each major against a CUDA Toolkit of the floor's - major.minor. The build requires the header's major.minor to equal the - cuda-bindings', so a toolkit pin in another minor cannot build the floor, - and a floor bump must move the pin (or the reverse). +2. ci/versions.yml builds each major against a CUDA Toolkit of at least the + floor's major.minor. A toolkit below the floor's minor cannot build the + floor's cuda-bindings (the build requires the header cuda-bindings was + generated from); a toolkit ahead of the floor is the toolkit-bump window + described in cuda_core/AGENTS.md. 3. No documentation page spells a floor out by hand; docs/source/conf.py provides |cuda-bindings-floor-cu12| and |cuda-bindings-floor-cu13|. Release notes are history and are exempt. @@ -69,10 +70,11 @@ def ci_pin_problems(floors: dict[int, tuple[int, int, int]], versions_yml: str) problems.append(f"{CI_VERSIONS}: no build or prev_build toolkit pin for CUDA {major} (floor {floor})") continue pinned_major, pinned_minor, key = pins[major] - if (pinned_major, pinned_minor) != floor[:2]: + if pinned_major != floor[0] or pinned_minor < floor[1]: problems.append( f"{CI_VERSIONS}: cuda.{key}.version is {pinned_major}.{pinned_minor} but the CUDA {major} floor " - f"is cuda-bindings {'.'.join(map(str, floor))}; the toolkit and the floor must share major.minor" + f"is cuda-bindings {'.'.join(map(str, floor))}; the toolkit must not sit below the floor's " + "major.minor (ahead of it is the toolkit-bump window, see cuda_core/AGENTS.md)" ) for major in pins: if major not in floors: From 194f9eaec0daa2894a5dae40c202484046168ad9 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Tue, 22 Sep 2026 14:20:35 -0700 Subject: [PATCH 11/18] cuda.core: latch a failed driver-table fill, keep pending exceptions, attach the reason (#2783) A fill of the driver function table that failed was retried by every later DRIVER_CALL, re-importing cuda-bindings and re-warning each time, and its Python calls ran with whatever exception the caller had pending: a deleter making the first driver call during unwinding clobbered the user's exception. The failure is now latched per table (transient causes such as KeyboardInterrupt, MemoryError and RecursionError excepted), a PendingExceptionGuard saves and restores the pending exception around the fill, and fn_table_error() copies its text under the mutex. The CUDAError raised for the trampoline's CUDA_ERROR_NOT_INITIALIZED used to explain that cuInit() was not called. report_unavailable_fn() now records the fill's reason as the error detail (note_driver_table_failure), which the Cython error path attaches to the exception as a note, on every affected call; the fill itself warns once, and a null entry in a filled table (a gate bug) warns once per table. format_cuda_error() no longer null-checks the pointers, deviceptr_import_ipc() returns an error when the fill failed, and the graphics entry's requested version is annotated. The raw-pointer lint in test_rt_layout.py now catches null checks too. tests/test_driver_table.py runs a child interpreter whose _inspect_function_pointers() returns a table with a null baseline entry or a missing entry, and asserts the note, the single warning and the latch. --- cuda_core/cuda/core/_cpp/rt/driver_api.hpp | 18 ++- cuda_core/cuda/core/_cpp/rt/error.cpp | 9 +- cuda_core/cuda/core/_cpp/rt/error.hpp | 6 + cuda_core/cuda/core/_cpp/rt/memory.cpp | 6 +- cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp | 110 ++++++++++--- cuda_core/tests/test_driver_table.py | 150 ++++++++++++++++++ cuda_core/tests/test_rt_layout.py | 2 +- 7 files changed, 270 insertions(+), 31 deletions(-) create mode 100644 cuda_core/tests/test_driver_table.py diff --git a/cuda_core/cuda/core/_cpp/rt/driver_api.hpp b/cuda_core/cuda/core/_cpp/rt/driver_api.hpp index b2955816647..08f17380441 100644 --- a/cuda_core/cuda/core/_cpp/rt/driver_api.hpp +++ b/cuda_core/cuda/core/_cpp/rt/driver_api.hpp @@ -112,7 +112,7 @@ namespace cuda_core::rt { X(cuGraphChildGraphNodeGetGraph, 10000) \ /* Linker */ \ X(cuLinkDestroy, 5050) \ - /* Graphics interop */ \ + /* Graphics interop; cuda-bindings requests 7000 (PTDS) or 3000 (legacy) */ \ X(cuGraphicsUnmapResources, 7000) \ X(cuGraphicsUnregisterResource, 3000) \ /* Texture / surface / array (PR #467) */ \ @@ -167,15 +167,19 @@ bool fn_table_ready(FnTable table) noexcept; // report_message() when cuda-bindings cannot load the library, a key is // missing (the installed cuda-bindings does not match the header this build // compiled against), or a baseline driver function is null (the driver is -// older than the CUDA major series supports). Never leaves a Python error set. -// Implemented in py_driver_fns.cpp. +// older than the CUDA major series supports). A failed fill is latched: later +// calls return false at once without retrying. Preserves a pending Python +// exception and never leaves one set. Implemented in py_driver_fns.cpp. bool ensure_fn_table(FnTable table) noexcept; -// The reason the last fill of `table` failed, or nullptr. Implemented in py_driver_fns.cpp. -const char* fn_table_error(FnTable table) noexcept; +// Copy the reason the fill of `table` failed into `buffer`; false if it did +// not fail. Implemented in py_driver_fns.cpp. +bool fn_table_error(FnTable table, char* buffer, std::size_t size) noexcept; -// Report, once per table, that `name` was called while unavailable: a gate -// bug, or a failed fill (whose reason is included). Implemented in py_driver_fns.cpp. +// Record that `name` was called while unavailable. After a failed fill the +// fill's reason is attached to the error the caller raises (it was reported +// when the fill failed); a null entry in a filled table is a gate bug and is +// reported once per table. Implemented in py_driver_fns.cpp. void report_unavailable_fn(FnTable table, const char* name) noexcept; namespace detail { diff --git a/cuda_core/cuda/core/_cpp/rt/error.cpp b/cuda_core/cuda/core/_cpp/rt/error.cpp index 2f9f06f285c..8ebc30945f8 100644 --- a/cuda_core/cuda/core/_cpp/rt/error.cpp +++ b/cuda_core/cuda/core/_cpp/rt/error.cpp @@ -38,8 +38,8 @@ void format_cuda_error(char* buffer, size_t size, const char* operation, CUresul const char* detail) noexcept { const char* error_name = nullptr; const char* error_description = nullptr; - bool decoded = p_cuGetErrorName && p_cuGetErrorString - && DRIVER_CALL(cuGetErrorName, status, &error_name) == CUDA_SUCCESS + // With the table unavailable the trampolines fail and the numeric fallback is used. + bool decoded = DRIVER_CALL(cuGetErrorName, status, &error_name) == CUDA_SUCCESS && DRIVER_CALL(cuGetErrorString, status, &error_description) == CUDA_SUCCESS; const char* outcome = detail ? detail : "failed"; if (decoded) { @@ -92,6 +92,11 @@ void clear_last_error_detail() noexcept { last_error_detail_status = CUDA_SUCCESS; } +void note_driver_table_failure(const char* reason) noexcept { + std::snprintf(last_error_detail, sizeof(last_error_detail), "%s", reason); + last_error_detail_status = CUDA_ERROR_NOT_INITIALIZED; +} + namespace detail { // Record that the caller's context was not restored as the detail of the // CUresult about to be returned and raised: the operation status if the diff --git a/cuda_core/cuda/core/_cpp/rt/error.hpp b/cuda_core/cuda/core/_cpp/rt/error.hpp index 92b5a706544..b3edbab744a 100644 --- a/cuda_core/cuda/core/_cpp/rt/error.hpp +++ b/cuda_core/cuda/core/_cpp/rt/error.hpp @@ -66,6 +66,12 @@ void attach_rollback_failure(const char* operation, CUresult status, const char* const char* take_last_error_detail(CUresult status) noexcept; void clear_last_error_detail() noexcept; +// Record why the driver function table is unavailable as the detail of the +// CUDA_ERROR_NOT_INITIALIZED the trampoline is about to return (see +// driver_api.hpp), so the raised CUDAError explains the failed fill instead +// of suggesting that cuInit() was not called. Implemented in error.cpp. +void note_driver_table_failure(const char* reason) noexcept; + // Tests only: make the next context restoration on this thread fail with // `status`, leaving the target context current as a real failure would. // Implemented in context.cpp diff --git a/cuda_core/cuda/core/_cpp/rt/memory.cpp b/cuda_core/cuda/core/_cpp/rt/memory.cpp index 9ad1b64bccb..8831fc4191f 100644 --- a/cuda_core/cuda/core/_cpp/rt/memory.cpp +++ b/cuda_core/cuda/core/_cpp/rt/memory.cpp @@ -406,7 +406,11 @@ DevicePtrHandle deviceptr_import_ipc(const MemoryPoolHandle& h_pool, const void* // Resolve the table before any lock is taken: a fill acquires the GIL, and // nothing under ipc_import_mutex may (#2840). The raw p_ calls below rely on it. - ensure_fn_table(FnTable::driver); + if (!ensure_fn_table(FnTable::driver)) { + report_unavailable_fn(FnTable::driver, "cuMemPoolImportPointer"); + err = CUDA_ERROR_NOT_INITIALIZED; + return {}; + } if (use_ipc_ptr_cache()) { ExportDataKey key; diff --git a/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp b/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp index bf4a31e8e1e..d0c2b548d1e 100644 --- a/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp +++ b/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp @@ -18,8 +18,13 @@ // values, one after the other. // // Failures never propagate as exceptions and never leave a Python error set: -// they are recorded, reported through report_message(), and every affected -// DRIVER_CALL then returns an error status from a trampoline (driver_api.hpp). +// they are recorded, reported through report_message(), latched (a failed +// table is not retried, so no call re-imports or re-warns), and every +// affected DRIVER_CALL then returns an error status from a trampoline +// (driver_api.hpp) with the reason attached to the raised error as a note. +// A fill can run while a Python exception is propagating (a deleter making its +// first driver call during unwinding), so the pending exception is saved +// around the Python calls and restored afterwards. #include "py.hpp" #include "driver_api.hpp" @@ -38,9 +43,43 @@ constexpr std::size_t kTables = 4; constexpr std::size_t kMaxEntries = 128; std::atomic table_ready[kTables]; +std::atomic table_failed[kTables]; std::atomic unavailable_reported[kTables]; std::mutex fill_mutex; -char fill_error[kTables][512] = {}; +char fill_error[kTables][512] = {}; // guarded by fill_mutex + +// Saves the pending Python exception on construction and restores it on +// destruction, so the Python calls in between start from a clean error state +// and the caller's exception survives. Requires the GIL. +class PendingExceptionGuard { +public: + PendingExceptionGuard() noexcept { +#if PY_VERSION_HEX >= 0x030C0000 + exc_ = PyErr_GetRaisedException(); +#else + PyErr_Fetch(&type_, &value_, &traceback_); +#endif + } + ~PendingExceptionGuard() { + PyErr_Clear(); // drop anything the guarded calls left set +#if PY_VERSION_HEX >= 0x030C0000 + PyErr_SetRaisedException(exc_); +#else + PyErr_Restore(type_, value_, traceback_); +#endif + } + PendingExceptionGuard(const PendingExceptionGuard&) = delete; + PendingExceptionGuard& operator=(const PendingExceptionGuard&) = delete; + +private: +#if PY_VERSION_HEX >= 0x030C0000 + PyObject* exc_ = nullptr; +#else + PyObject* type_ = nullptr; + PyObject* value_ = nullptr; + PyObject* traceback_ = nullptr; +#endif +}; std::size_t index_of(FnTable table) noexcept { return static_cast(table); } @@ -64,8 +103,15 @@ const char* library_name(FnTable table) noexcept { return "library"; } -// Copy the pending Python exception's text into buf and clear it. -void take_python_error(char* buf, std::size_t size) noexcept { +// Copy the pending Python exception's text into buf and clear it. Returns +// true when the exception says nothing about cuda-bindings or the driver: an +// interruption (KeyboardInterrupt, SystemExit) or exhaustion (MemoryError, +// RecursionError). Such a failure is reported but not latched; the next call +// tries again. +bool take_python_error(char* buf, std::size_t size) noexcept { + const bool transient = PyErr_Occurred() + && (!PyErr_ExceptionMatches(PyExc_Exception) || PyErr_ExceptionMatches(PyExc_MemoryError) + || PyErr_ExceptionMatches(PyExc_RecursionError)); #if PY_VERSION_HEX >= 0x030C0000 PyObject* exc = PyErr_GetRaisedException(); #else @@ -82,13 +128,17 @@ void take_python_error(char* buf, std::size_t size) noexcept { Py_XDECREF(text); Py_XDECREF(exc); PyErr_Clear(); + return transient; } -void record_failure(FnTable table, const char* message) noexcept { +void record_failure(FnTable table, const char* message, bool latch = true) noexcept { { std::lock_guard lock(fill_mutex); std::snprintf(fill_error[index_of(table)], sizeof(fill_error[0]), "%s", message); } + if (latch) { + table_failed[index_of(table)].store(true, std::memory_order_release); + } report_message(message); } @@ -98,9 +148,14 @@ bool fn_table_ready(FnTable table) noexcept { return table_ready[index_of(table)].load(std::memory_order_acquire); } -const char* fn_table_error(FnTable table) noexcept { +bool fn_table_error(FnTable table, char* buffer, std::size_t size) noexcept { + std::lock_guard lock(fill_mutex); const char* text = fill_error[index_of(table)]; - return text[0] ? text : nullptr; + if (!text[0]) { + return false; + } + std::snprintf(buffer, size, "%s", text); + return true; } bool ensure_fn_table(FnTable table) noexcept { @@ -108,6 +163,9 @@ bool ensure_fn_table(FnTable table) noexcept { if (table_ready[idx].load(std::memory_order_acquire)) { return true; } + if (table_failed[idx].load(std::memory_order_acquire)) { + return false; // latched: the reason was reported when the fill failed + } std::size_t count = 0; const FnEntry* entries = fn_table_entries(table, &count); if (entries == nullptr || count > kMaxEntries) { @@ -128,21 +186,22 @@ bool ensure_fn_table(FnTable table) noexcept { record_failure(table, "cuda.core cannot resolve driver functions while the interpreter is shutting down"); return false; } + PendingExceptionGuard pending; PyObject* module = PyImport_ImportModule(module_name(table)); if (module == nullptr) { - take_python_error(cause, sizeof(cause)); + const bool transient = take_python_error(cause, sizeof(cause)); std::snprintf(message, sizeof(message), "cuda.core cannot import %s from the installed cuda-bindings: %s", module_name(table), cause); - record_failure(table, message); + record_failure(table, message, !transient); return false; } PyObject* pointers = PyObject_CallMethod(module, "_inspect_function_pointers", nullptr); Py_DECREF(module); if (pointers == nullptr) { - take_python_error(cause, sizeof(cause)); + const bool transient = take_python_error(cause, sizeof(cause)); std::snprintf(message, sizeof(message), "cuda-bindings could not load the %s: %s", library_name(table), cause); - record_failure(table, message); + record_failure(table, message, !transient); return false; } if (!PyDict_Check(pointers)) { @@ -187,9 +246,10 @@ bool ensure_fn_table(FnTable table) noexcept { for (std::size_t i = 0; i < count; ++i) { if (values[i] == nullptr && entries[i].introduced <= CUDA_CORE_BUILD_MAJOR * 1000) { std::snprintf(message, sizeof(message), - "the installed CUDA driver does not provide %s, which every driver of the CUDA %d " - "series provides; this cuda.core build requires a CUDA %d driver", - entries[i].name, CUDA_CORE_BUILD_MAJOR, CUDA_CORE_BUILD_MAJOR); + "the installed CUDA driver lacks %s, which every CUDA %d driver provides " + "(introduced in CUDA %d.%d); this cuda.core build needs a newer driver", + entries[i].name, CUDA_CORE_BUILD_MAJOR, entries[i].introduced / 1000, + entries[i].introduced / 10 % 100); record_failure(table, message); return false; } @@ -210,12 +270,10 @@ bool ensure_fn_table(FnTable table) noexcept { } void report_unavailable_fn(FnTable table, const char* name) noexcept { - // Once per table: the first unavailable call is the informative one. - if (unavailable_reported[index_of(table)].exchange(true)) { - return; - } + char reason[sizeof(fill_error[0])]; + const bool failed_fill = fn_table_error(table, reason, sizeof(reason)); char message[768]; - if (const char* reason = fn_table_error(table)) { + if (failed_fill) { std::snprintf(message, sizeof(message), "cuda.core could not call %s: %s", name, reason); } else { std::snprintf(message, sizeof(message), @@ -223,6 +281,18 @@ void report_unavailable_fn(FnTable table, const char* name) noexcept { "provide it; a feature gate is missing or wrong. The call returned an error instead.", name, library_name(table)); } + if (table == FnTable::driver) { + // The trampoline returns CUDA_ERROR_NOT_INITIALIZED; the Cython error + // path attaches this as a note to the CUDAError it raises for it. + note_driver_table_failure(message); + } + if (failed_fill) { + return; // the fill reported its reason when it failed; the note carries it to each raised error + } + // A gate bug: warn once per table, the first unavailable call is the informative one. + if (unavailable_reported[index_of(table)].exchange(true)) { + return; + } report_message(message); } diff --git a/cuda_core/tests/test_driver_table.py b/cuda_core/tests/test_driver_table.py new file mode 100644 index 00000000000..523898e20ac --- /dev/null +++ b/cuda_core/tests/test_driver_table.py @@ -0,0 +1,150 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# SPDX-License-Identifier: Apache-2.0 + +"""The C++ driver function table (cuda/core/_cpp/rt/py_driver_fns.cpp) when its fill fails. + +The table is filled from ``cuda.bindings._internal.driver._inspect_function_pointers()`` on +the first driver call. A child interpreter replaces that function so the fill fails in a +controlled way, then makes driver calls through ``cuda.core`` and reports what happened. +Expected: the failure is reported once as a :class:`CUDAWarning`, every affected call raises +:class:`CUDAError` with the reason attached as a note, and the failure is latched (no second +warning, no retry). + +The child needs a loadable CUDA driver and a visible device, so this module skips without them. +Runs with ``--noconftest``. +""" + +import os +import re +import subprocess +import sys +import textwrap +from pathlib import Path + +import pytest + +CORE = Path(__file__).resolve().parents[1] / "cuda" / "core" +LOADER = Path(__file__).resolve().parents[2] / "cuda_bindings" / "cuda" / "bindings" / "_internal" / "driver_linux.pyx" + + +def _gpu_available() -> bool: + """The child calls cuInit and cuDeviceGetCount through cuda-bindings before it reaches + the C++ table, so it needs a loadable driver and a visible device.""" + try: + from cuda.bindings import driver + + (status,) = driver.cuInit(0) + if int(status) != 0: + return False + status, count = driver.cuDeviceGetCount() + return int(status) == 0 and count > 0 + except Exception: + return False + + +pytestmark = [ + pytest.mark.skipif(not _gpu_available(), reason="the child needs a CUDA driver and a visible device"), + pytest.mark.thread_unsafe(reason="spawns child interpreters"), +] + + +def _table_keys() -> list[str]: + """The keys the fill looks up: "__" + the symbol cuda-bindings' loader requests for each + name in driver_api.hpp (cuStreamDestroy -> __cuStreamDestroy_v2).""" + if not LOADER.is_file(): + pytest.skip("needs the cuda_bindings source tree next to cuda_core") + names = re.findall( + r"^\s*X\((cu\w+), \d+\)", (CORE / "_cpp" / "rt" / "driver_api.hpp").read_text(encoding="utf-8"), re.M + ) + loader = LOADER.read_text(encoding="utf-8") + keys = [] + for name in names: + m = re.search(rf"cuGetProcAddress_v2\('{name}', &(__\w+),", loader) + assert m is not None, name + keys.append(m.group(1)) + return keys + + +_CHILD = textwrap.dedent(""" + import warnings + import cuda.bindings._internal.driver as loader + + KEYS = {keys!r} + + def fake_inspect_function_pointers(): + table = {{key: 1 for key in KEYS}} # placeholder addresses; the fill fails before any call + {mutation} + return table + + loader._inspect_function_pointers = fake_inspect_function_pointers + + from cuda.core import CUDAError, CUDAWarning, Device + + with warnings.catch_warnings(record=True) as caught: + warnings.simplefilter("always") + for attempt in range(2): + try: + # cuInit and the device query are Cython calls; the primary context + # retain is the first call through the C++ table. + Device(0).set_current() + except CUDAError as exc: + print(f"ATTEMPT {{attempt}} CUDAError: {{exc}} NOTES: {{getattr(exc, '__notes__', [])}}") + else: + print(f"ATTEMPT {{attempt}} no error") + cuda_warnings = [w for w in caught if issubclass(w.category, CUDAWarning)] + print(f"CUDAWARNINGS {{len(cuda_warnings)}}") + for w in cuda_warnings: + print("WARNING:", str(w.message)) +""") + + +def _run_child(mutation: str, tmp_path: Path) -> str: + code = _CHILD.format(keys=_table_keys(), mutation=mutation) + env = {k: v for k, v in os.environ.items() if k != "PYTHONPATH"} + result = subprocess.run( # noqa: S603 + [sys.executable, "-c", code], + cwd=tmp_path, + env=env, + capture_output=True, + text=True, + stdin=subprocess.DEVNULL, + timeout=180, + check=False, + ) + assert result.returncode == 0, result.stdout + result.stderr + return result.stdout + + +def _build_major() -> int: + from cuda.core import _build_info + + return _build_info.CUDA_MAJOR + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_null_baseline_entry_fails_the_fill_once_and_latches(tmp_path): + out = _run_child('table["__cuGetErrorName"] = 0', tmp_path) + lines = out.splitlines() + attempts = [line for line in lines if line.startswith("ATTEMPT")] + assert len(attempts) == 2 + for line in attempts: + assert "CUDAError: CUDA_ERROR_NOT_INITIALIZED" in line + assert "could not call cu" in line # the reason rides on every raised error + assert f"lacks cuGetErrorName, which every CUDA {_build_major()} driver provides" in line + assert "CUDAWARNINGS 1" in lines # reported once, then latched + warning = next(line for line in lines if line.startswith("WARNING:")) + assert "lacks cuGetErrorName" in warning + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_missing_table_entry_names_the_mismatch(tmp_path): + out = _run_child('table.pop("__cuDevicePrimaryCtxRetain", None)', tmp_path) + lines = out.splitlines() + attempts = [line for line in lines if line.startswith("ATTEMPT")] + assert len(attempts) == 2 + for line in attempts: + assert "CUDAError: CUDA_ERROR_NOT_INITIALIZED" in line + assert "has no entry for cuDevicePrimaryCtxRetain" in line + assert "install the cuda-bindings this cuda.core requires" in line + assert "CUDAWARNINGS 1" in lines diff --git a/cuda_core/tests/test_rt_layout.py b/cuda_core/tests/test_rt_layout.py index 228ddfd433a..d9c611a11fc 100644 --- a/cuda_core/tests/test_rt_layout.py +++ b/cuda_core/tests/test_rt_layout.py @@ -141,7 +141,7 @@ def test_driver_calls_go_through_the_table(): the table's own machinery and the sites under ipc_import_mutex, where the table is resolved before the lock and marked `// raw:`.""" machinery = {"driver_api.hpp", "driver_api.cpp", "py_driver_fns.cpp", "internal.hpp"} - raw_call = re.compile(r"\bp_(cu|nv)\w+\(") + raw_call = re.compile(r"\bp_(cu|nv)\w+\b") # calls and null checks alike offenders = [] for path in HEADERS + SOURCES: if path.name in machinery: From a37c10a42e69bece1b4ecf4e7c0fbd077dd2b407 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Tue, 22 Sep 2026 14:20:35 -0700 Subject: [PATCH 12/18] cuda.core: keep the NVML constant's stub annotation-only; finish the text sweep (#2783) The API check gates "Check job status", so the stub of CUDA_BINDINGS_NVML_IS_COMPATIBLE must keep the bare `bool` annotation it had: the constant is declared with the annotation and assigned in a runtime-only block, which the stub generator does not record. It is always True and no longer called deprecated. DeviceArch is a plain enum.IntEnum instead of subclassing cuda-bindings' private FastEnum. Remaining text that named cuda-bindings versions as feature requirements now names the driver and the CUDA 13 build: the copy-options messages and docstrings in _buffer.pyx and _copy_ops.pyx, the graph node update docstrings, api.rst, the checkpoint section and module docstring, install.rst on conda-forge, DESIGN.md's summary, AGENTS.md's path to it, and the test helpers. The hasattr/getattr probes for NVML members present at both floors are gone, as is the probe-derived allow-list in test_enum_coverage.py. --- cuda_core/cuda/core/_cpp/rt/DESIGN.md | 3 ++- cuda_core/cuda/core/_memory/_buffer.pyi | 8 ++++---- cuda_core/cuda/core/_memory/_buffer.pyx | 10 +++++----- cuda_core/cuda/core/_memory/_copy_ops.pyi | 5 ++--- cuda_core/cuda/core/_memory/_copy_ops.pyx | 9 ++++----- .../cuda/core/_memory/_managed_memory_ops.pyx | 6 +++--- cuda_core/cuda/core/checkpoint.py | 7 +++++++ cuda_core/cuda/core/graph/_subclasses.pyi | 14 ++++++++------ cuda_core/cuda/core/graph/_subclasses.pyx | 14 ++++++++------ cuda_core/cuda/core/system/_clock.pxi | 4 ++-- cuda_core/cuda/core/system/_device.pyi | 2 +- cuda_core/cuda/core/system/_device.pyx | 11 +++++------ cuda_core/cuda/core/system/_system.pyi | 2 +- cuda_core/cuda/core/system/_system.pyx | 12 +++++++++--- cuda_core/cuda/core/system/typing.py | 11 ++++++----- cuda_core/docs/source/api.rst | 19 ++++++++++--------- cuda_core/docs/source/release/1.3.0-notes.rst | 2 +- cuda_core/tests/test_enum_coverage.py | 10 ++-------- cuda_core/tests/test_program.py | 4 +--- .../cuda_python_test_helpers/arch_check.py | 15 ++------------- 20 files changed, 83 insertions(+), 85 deletions(-) diff --git a/cuda_core/cuda/core/_cpp/rt/DESIGN.md b/cuda_core/cuda/core/_cpp/rt/DESIGN.md index ed0073a7392..783dc53a4df 100644 --- a/cuda_core/cuda/core/_cpp/rt/DESIGN.md +++ b/cuda_core/cuda/core/_cpp/rt/DESIGN.md @@ -467,7 +467,8 @@ The resource handle design: 2. **Encodes lifetimes structurally** via embedded handle dependencies. 3. **Uses Cython's `cimport` mechanism** to share C++ code across modules without duplicate static/thread-local state. -4. **Resolves CUDA driver symbols** dynamically through cuda-bindings' `__pyx_capi__` capsules. +4. **Resolves CUDA driver symbols** through the driver entry points cuda-bindings resolves + (`_inspect_function_pointers()`), filled lazily by `ensure_fn_table()` on first use. 5. **Provides overloaded accessors** (`as_cu`, `as_intptr`, `as_py`) since handles cannot have attributes without unnecessary Python object wrappers. diff --git a/cuda_core/cuda/core/_memory/_buffer.pyi b/cuda_core/cuda/core/_memory/_buffer.pyi index 756b8661488..e2718125582 100644 --- a/cuda_core/cuda/core/_memory/_buffer.pyi +++ b/cuda_core/cuda/core/_memory/_buffer.pyi @@ -171,8 +171,8 @@ class Buffer: asynchronous copy options : :class:`~utils.CopyOptions`, optional Transfer hints (source access order, location hints, overlap mode). - Honored when cuda.bindings and the driver are both CUDA 13.2 or - newer. Not accepted with ``LEGACY_DEFAULT_STREAM``; use + Honored on the CUDA 13 build of ``cuda.core`` with a driver of CUDA + 13.2 or newer. Not accepted with ``LEGACY_DEFAULT_STREAM``; use ``PER_THREAD_DEFAULT_STREAM`` instead. Not accepted with a capturing stream either, since a graph cannot represent these attributes; use :meth:`graph.GraphNode.memcpy` for a plain, @@ -205,8 +205,8 @@ class Buffer: asynchronous copy options : :class:`~utils.CopyOptions`, optional Transfer hints (source access order, location hints, overlap mode). - Honored when cuda.bindings and the driver are both CUDA 13.2 or - newer. Not accepted with ``LEGACY_DEFAULT_STREAM``; use + Honored on the CUDA 13 build of ``cuda.core`` with a driver of CUDA + 13.2 or newer. Not accepted with ``LEGACY_DEFAULT_STREAM``; use ``PER_THREAD_DEFAULT_STREAM`` instead. Not accepted with a capturing stream either, since a graph cannot represent these attributes; use :meth:`graph.GraphNode.memcpy` for a plain, diff --git a/cuda_core/cuda/core/_memory/_buffer.pyx b/cuda_core/cuda/core/_memory/_buffer.pyx index e6c4c6710bb..903a7ab0264 100644 --- a/cuda_core/cuda/core/_memory/_buffer.pyx +++ b/cuda_core/cuda/core/_memory/_buffer.pyx @@ -225,7 +225,7 @@ cdef void _dispatch_buffer_copy( else: _reject_unsupported_during_api_call( options.src_access_order, - "cuda.bindings and the driver to both report CUDA 13.2 or newer " + "the CUDA 13 build of cuda.core and a driver reporting CUDA 13.2 or newer " "(cuMemcpyWithAttributesAsync is unavailable here)", ) # STREAM and ANY never require access sooner than stream order, so @@ -498,8 +498,8 @@ cdef class Buffer: asynchronous copy options : :class:`~utils.CopyOptions`, optional Transfer hints (source access order, location hints, overlap mode). - Honored when cuda.bindings and the driver are both CUDA 13.2 or - newer. Not accepted with ``LEGACY_DEFAULT_STREAM``; use + Honored on the CUDA 13 build of ``cuda.core`` with a driver of CUDA + 13.2 or newer. Not accepted with ``LEGACY_DEFAULT_STREAM``; use ``PER_THREAD_DEFAULT_STREAM`` instead. Not accepted with a capturing stream either, since a graph cannot represent these attributes; use :meth:`graph.GraphNode.memcpy` for a plain, @@ -554,8 +554,8 @@ cdef class Buffer: asynchronous copy options : :class:`~utils.CopyOptions`, optional Transfer hints (source access order, location hints, overlap mode). - Honored when cuda.bindings and the driver are both CUDA 13.2 or - newer. Not accepted with ``LEGACY_DEFAULT_STREAM``; use + Honored on the CUDA 13 build of ``cuda.core`` with a driver of CUDA + 13.2 or newer. Not accepted with ``LEGACY_DEFAULT_STREAM``; use ``PER_THREAD_DEFAULT_STREAM`` instead. Not accepted with a capturing stream either, since a graph cannot represent these attributes; use :meth:`graph.GraphNode.memcpy` for a plain, diff --git a/cuda_core/cuda/core/_memory/_copy_ops.pyi b/cuda_core/cuda/core/_memory/_copy_ops.pyi index ff76140d2ba..cb30bedb294 100644 --- a/cuda_core/cuda/core/_memory/_copy_ops.pyi +++ b/cuda_core/cuda/core/_memory/_copy_ops.pyi @@ -68,9 +68,8 @@ def copy_batch(stream: Stream, srcs: Sequence[Buffer], dsts: Sequence[Buffer], * Notes ----- - Batching through ``cuMemcpyBatchAsync`` requires all three of: - ``cuda.core`` built against CUDA 13 headers, ``cuda.bindings`` 13.0 or - newer, and a driver reporting CUDA 13.0 or newer + Batching through ``cuMemcpyBatchAsync`` requires both the CUDA 13 build of + ``cuda.core`` and a driver reporting CUDA 13.0 or newer (``cuDriverGetVersion() >= 13000``). ``cuda.bindings`` binds only the CUDA 13.0 revision of the entry point, so a driver that predates it is refused even where it implements the earlier CUDA 12.8 signature. diff --git a/cuda_core/cuda/core/_memory/_copy_ops.pyx b/cuda_core/cuda/core/_memory/_copy_ops.pyx index be0ecafaed7..3143b4cc158 100644 --- a/cuda_core/cuda/core/_memory/_copy_ops.pyx +++ b/cuda_core/cuda/core/_memory/_copy_ops.pyx @@ -143,9 +143,8 @@ def copy_batch( Notes ----- - Batching through ``cuMemcpyBatchAsync`` requires all three of: - ``cuda.core`` built against CUDA 13 headers, ``cuda.bindings`` 13.0 or - newer, and a driver reporting CUDA 13.0 or newer + Batching through ``cuMemcpyBatchAsync`` requires both the CUDA 13 build of + ``cuda.core`` and a driver reporting CUDA 13.0 or newer (``cuDriverGetVersion() >= 13000``). ``cuda.bindings`` binds only the CUDA 13.0 revision of the entry point, so a driver that predates it is refused even where it implements the earlier CUDA 12.8 signature. @@ -241,8 +240,8 @@ cdef void _reject_during_api_call_fallback(tuple attr_tuple): for i in range(len(attr_tuple)): _reject_unsupported_during_api_call( (attr_tuple[i]).src_access_order, - "cuda.core built against CUDA 13 headers and cuda.bindings/driver " - "13.0 or newer (cuMemcpyBatchAsync is unavailable here)", + "the CUDA 13 build of cuda.core and a driver reporting CUDA 13.0 or newer " + "(cuMemcpyBatchAsync is unavailable here)", index=i, ) diff --git a/cuda_core/cuda/core/_memory/_managed_memory_ops.pyx b/cuda_core/cuda/core/_memory/_managed_memory_ops.pyx index 77cf423fab5..3c91f1b689d 100644 --- a/cuda_core/cuda/core/_memory/_managed_memory_ops.pyx +++ b/cuda_core/cuda/core/_memory/_managed_memory_ops.pyx @@ -369,9 +369,9 @@ IF CUDA_CORE_BUILD_MAJOR >= 13: ELSE: def _read_preferred_location_v2(Buffer buf) -> Device | Host | None: # Symbols exist so _managed_buffer.py can import the v2 readers - # unconditionally. Their properties gate on both binding_version() - # and driver_version() >= (13, 0, 0), so these paths are unreachable - # on a CUDA 12 build. + # unconditionally. Their properties gate on the CUDA 13 build and + # driver_version() >= (13, 0, 0), so these paths are unreachable on a + # CUDA 12 build. raise NotImplementedError( "_read_preferred_location_v2 requires a CUDA 13 build of cuda.core" ) diff --git a/cuda_core/cuda/core/checkpoint.py b/cuda_core/cuda/core/checkpoint.py index cf9e09ed14c..08a0f0189c0 100644 --- a/cuda_core/cuda/core/checkpoint.py +++ b/cuda_core/cuda/core/checkpoint.py @@ -2,6 +2,13 @@ # # SPDX-License-Identifier: Apache-2.0 +"""CUDA process checkpointing (Linux). + +Requires the CUDA 13 build of cuda.core (the driver structures it uses are +CUDA 13 types) and a CUDA driver of version 12.8 or newer with checkpoint API +support. +""" + import ctypes as _ctypes from collections.abc import Mapping from typing import Any diff --git a/cuda_core/cuda/core/graph/_subclasses.pyi b/cuda_core/cuda/core/graph/_subclasses.pyi index e207e05eaca..dcc526cbc54 100644 --- a/cuda_core/cuda/core/graph/_subclasses.pyi +++ b/cuda_core/cuda/core/graph/_subclasses.pyi @@ -137,9 +137,10 @@ class MemsetNode(GraphNode): Omitted parameters preserve their current values. ``dst_owner`` may only accompany a raw-address ``dst``. - With CUDA 12.2 through 13.1, the node's intended CUDA context must be - current when this method is called. CUDA driver and ``cuda.bindings`` - versions 13.2 and newer preserve the recorded context automatically. + With drivers from CUDA 12.2 through 13.1, the node's intended CUDA + context must be current when this method is called. With the CUDA 13 + build of ``cuda.core`` and a driver of CUDA 13.2 or newer, the recorded + context is preserved automatically. .. warning:: @@ -188,9 +189,10 @@ class MemcpyNode(GraphNode): Multidimensional, pitched, offset, and array-backed memcpy nodes are not supported. - With CUDA 12.2 through 13.1, the node's intended CUDA context must be - current when this method is called. CUDA driver and ``cuda.bindings`` - versions 13.2 and newer preserve the recorded context automatically. + With drivers from CUDA 12.2 through 13.1, the node's intended CUDA + context must be current when this method is called. With the CUDA 13 + build of ``cuda.core`` and a driver of CUDA 13.2 or newer, the recorded + context is preserved automatically. .. warning:: diff --git a/cuda_core/cuda/core/graph/_subclasses.pyx b/cuda_core/cuda/core/graph/_subclasses.pyx index 1e1d2c1953c..cf2d704facb 100644 --- a/cuda_core/cuda/core/graph/_subclasses.pyx +++ b/cuda_core/cuda/core/graph/_subclasses.pyx @@ -650,9 +650,10 @@ cdef class MemsetNode(GraphNode): Omitted parameters preserve their current values. ``dst_owner`` may only accompany a raw-address ``dst``. - With CUDA 12.2 through 13.1, the node's intended CUDA context must be - current when this method is called. CUDA driver and ``cuda.bindings`` - versions 13.2 and newer preserve the recorded context automatically. + With drivers from CUDA 12.2 through 13.1, the node's intended CUDA + context must be current when this method is called. With the CUDA 13 + build of ``cuda.core`` and a driver of CUDA 13.2 or newer, the recorded + context is preserved automatically. .. warning:: @@ -839,9 +840,10 @@ cdef class MemcpyNode(GraphNode): Multidimensional, pitched, offset, and array-backed memcpy nodes are not supported. - With CUDA 12.2 through 13.1, the node's intended CUDA context must be - current when this method is called. CUDA driver and ``cuda.bindings`` - versions 13.2 and newer preserve the recorded context automatically. + With drivers from CUDA 12.2 through 13.1, the node's intended CUDA + context must be current when this method is called. With the CUDA 13 + build of ``cuda.core`` and a driver of CUDA 13.2 or newer, the recorded + context is preserved automatically. .. warning:: diff --git a/cuda_core/cuda/core/system/_clock.pxi b/cuda_core/cuda/core/system/_clock.pxi index 3b3db4e85b0..0634028ea7b 100644 --- a/cuda_core/cuda/core/system/_clock.pxi +++ b/cuda_core/cuda/core/system/_clock.pxi @@ -20,8 +20,8 @@ _CLOCKS_EVENT_REASONS_MAPPING = { nvml.ClocksEventReasons.THROTTLE_REASON_HW_THERMAL_SLOWDOWN: ClocksEventReasons.HW_THERMAL_SLOWDOWN, nvml.ClocksEventReasons.THROTTLE_REASON_HW_POWER_BRAKE_SLOWDOWN: ClocksEventReasons.HW_POWER_BRAKE_SLOWDOWN, nvml.ClocksEventReasons.EVENT_REASON_DISPLAY_CLOCK_SETTING: ClocksEventReasons.DISPLAY_CLOCK_SETTING, - getattr(nvml.ClocksEventReasons, "EVENT_REASON_BOARD_LIMIT", 0x200): ClocksEventReasons.BOARD_LIMIT, - getattr(nvml.ClocksEventReasons, "EVENT_REASON_RELIABILITY", 0x400): ClocksEventReasons.RELIABILITY, + nvml.ClocksEventReasons.EVENT_REASON_BOARD_LIMIT: ClocksEventReasons.BOARD_LIMIT, + nvml.ClocksEventReasons.EVENT_REASON_RELIABILITY: ClocksEventReasons.RELIABILITY, } diff --git a/cuda_core/cuda/core/system/_device.pyi b/cuda_core/cuda/core/system/_device.pyi index 6fea049b21b..1414215300c 100644 --- a/cuda_core/cuda/core/system/_device.pyi +++ b/cuda_core/cuda/core/system/_device.pyi @@ -14,7 +14,7 @@ from cuda.core.system.typing import (AddressingMode, AffinityScope, ClockId, ThermalTarget) _CLOCK_ID_MAPPING = {ClockId.CURRENT: nvml.ClockId.CURRENT, ClockId.CUSTOMER_BOOST_MAX: nvml.ClockId.CUSTOMER_BOOST_MAX} -_CLOCKS_EVENT_REASONS_MAPPING = {nvml.ClocksEventReasons.EVENT_REASON_NONE: ClocksEventReasons.NONE, nvml.ClocksEventReasons.EVENT_REASON_GPU_IDLE: ClocksEventReasons.GPU_IDLE, nvml.ClocksEventReasons.EVENT_REASON_APPLICATIONS_CLOCKS_SETTING: ClocksEventReasons.APPLICATIONS_CLOCKS_SETTING, nvml.ClocksEventReasons.EVENT_REASON_SW_POWER_CAP: ClocksEventReasons.SW_POWER_CAP, nvml.ClocksEventReasons.THROTTLE_REASON_HW_SLOWDOWN: ClocksEventReasons.HW_SLOWDOWN, nvml.ClocksEventReasons.EVENT_REASON_SYNC_BOOST: ClocksEventReasons.SYNC_BOOST, nvml.ClocksEventReasons.EVENT_REASON_SW_THERMAL_SLOWDOWN: ClocksEventReasons.SW_THERMAL_SLOWDOWN, nvml.ClocksEventReasons.THROTTLE_REASON_HW_THERMAL_SLOWDOWN: ClocksEventReasons.HW_THERMAL_SLOWDOWN, nvml.ClocksEventReasons.THROTTLE_REASON_HW_POWER_BRAKE_SLOWDOWN: ClocksEventReasons.HW_POWER_BRAKE_SLOWDOWN, nvml.ClocksEventReasons.EVENT_REASON_DISPLAY_CLOCK_SETTING: ClocksEventReasons.DISPLAY_CLOCK_SETTING, getattr(nvml.ClocksEventReasons, 'EVENT_REASON_BOARD_LIMIT', 512): ClocksEventReasons.BOARD_LIMIT, getattr(nvml.ClocksEventReasons, 'EVENT_REASON_RELIABILITY', 1024): ClocksEventReasons.RELIABILITY} +_CLOCKS_EVENT_REASONS_MAPPING = {nvml.ClocksEventReasons.EVENT_REASON_NONE: ClocksEventReasons.NONE, nvml.ClocksEventReasons.EVENT_REASON_GPU_IDLE: ClocksEventReasons.GPU_IDLE, nvml.ClocksEventReasons.EVENT_REASON_APPLICATIONS_CLOCKS_SETTING: ClocksEventReasons.APPLICATIONS_CLOCKS_SETTING, nvml.ClocksEventReasons.EVENT_REASON_SW_POWER_CAP: ClocksEventReasons.SW_POWER_CAP, nvml.ClocksEventReasons.THROTTLE_REASON_HW_SLOWDOWN: ClocksEventReasons.HW_SLOWDOWN, nvml.ClocksEventReasons.EVENT_REASON_SYNC_BOOST: ClocksEventReasons.SYNC_BOOST, nvml.ClocksEventReasons.EVENT_REASON_SW_THERMAL_SLOWDOWN: ClocksEventReasons.SW_THERMAL_SLOWDOWN, nvml.ClocksEventReasons.THROTTLE_REASON_HW_THERMAL_SLOWDOWN: ClocksEventReasons.HW_THERMAL_SLOWDOWN, nvml.ClocksEventReasons.THROTTLE_REASON_HW_POWER_BRAKE_SLOWDOWN: ClocksEventReasons.HW_POWER_BRAKE_SLOWDOWN, nvml.ClocksEventReasons.EVENT_REASON_DISPLAY_CLOCK_SETTING: ClocksEventReasons.DISPLAY_CLOCK_SETTING, nvml.ClocksEventReasons.EVENT_REASON_BOARD_LIMIT: ClocksEventReasons.BOARD_LIMIT, nvml.ClocksEventReasons.EVENT_REASON_RELIABILITY: ClocksEventReasons.RELIABILITY} _CLOCK_TYPE_MAPPING = {ClockType.GRAPHICS: nvml.ClockType.CLOCK_GRAPHICS, ClockType.SM: nvml.ClockType.CLOCK_SM, ClockType.MEMORY: nvml.ClockType.CLOCK_MEM, ClockType.VIDEO: nvml.ClockType.CLOCK_VIDEO} _COOLER_CONTROL_MAPPING = {nvml.CoolerControl.THERMAL_COOLER_SIGNAL_TOGGLE: CoolerControl.TOGGLE, nvml.CoolerControl.THERMAL_COOLER_SIGNAL_VARIABLE: CoolerControl.VARIABLE} _COOLER_TARGET_MAPPING = {nvml.CoolerTarget.THERMAL_NONE: CoolerTarget.NONE, nvml.CoolerTarget.THERMAL_GPU: CoolerTarget.GPU, nvml.CoolerTarget.THERMAL_MEMORY: CoolerTarget.MEMORY, nvml.CoolerTarget.THERMAL_POWER_SUPPLY: CoolerTarget.POWER_SUPPLY} diff --git a/cuda_core/cuda/core/system/_device.pyx b/cuda_core/cuda/core/system/_device.pyx index c3bf23fe025..2192a2a3f03 100644 --- a/cuda_core/cuda/core/system/_device.pyx +++ b/cuda_core/cuda/core/system/_device.pyx @@ -108,12 +108,11 @@ _BRAND_TYPE_MAPPING = { } -if hasattr(nvml.BrandType, "BRAND_NVIDIA_DLA"): - _BRAND_TYPE_MAPPING.update({ - nvml.BrandType.BRAND_NVIDIA_DLA: "NVIDIA DLA", - nvml.BrandType.BRAND_NVIDIA_VGAMEDEV: "NVIDIA vGameDev", - nvml.BrandType.BRAND_NVIDIA_NPU: "NVIDIA NPU", - }) +_BRAND_TYPE_MAPPING.update({ + nvml.BrandType.BRAND_NVIDIA_DLA: "NVIDIA DLA", + nvml.BrandType.BRAND_NVIDIA_VGAMEDEV: "NVIDIA vGameDev", + nvml.BrandType.BRAND_NVIDIA_NPU: "NVIDIA NPU", +}) _GPU_P2P_CAPS_INDEX_MAPPING = { diff --git a/cuda_core/cuda/core/system/_system.pyi b/cuda_core/cuda/core/system/_system.pyi index 9306795f3c4..51aa0c92ef1 100644 --- a/cuda_core/cuda/core/system/_system.pyi +++ b/cuda_core/cuda/core/system/_system.pyi @@ -1,6 +1,6 @@ # This file was generated by stubgen-pyx v0.2.22 from cuda_core/cuda/core/system/_system.pyx -CUDA_BINDINGS_NVML_IS_COMPATIBLE: bool = True +CUDA_BINDINGS_NVML_IS_COMPATIBLE: bool __all__ = ['get_driver_branch', 'get_kernel_mode_driver_version', 'get_user_mode_driver_version', 'get_nvml_version', 'get_num_devices', 'get_process_name', 'CUDA_BINDINGS_NVML_IS_COMPATIBLE'] def get_user_mode_driver_version() -> tuple[int, ...]: diff --git a/cuda_core/cuda/core/system/_system.pyx b/cuda_core/cuda/core/system/_system.pyx index c7414f46e1a..f962613c82d 100644 --- a/cuda_core/cuda/core/system/_system.pyx +++ b/cuda_core/cuda/core/system/_system.pyx @@ -9,9 +9,15 @@ # this module stays importable without CUDA or NVML installed. -# Always True: kept for callers that read it before the cuda-bindings floor -# made NVML support unconditional. Deprecated. -CUDA_BINDINGS_NVML_IS_COMPATIBLE: bool = True +from typing import TYPE_CHECKING + +# Always True since the cuda-bindings floor made NVML support unconditional; +# kept for callers that read it. Assigned in a runtime-only block so that the +# generated stub keeps the bare annotation the public API had (the API check +# reports a changed attribute value otherwise). +CUDA_BINDINGS_NVML_IS_COMPATIBLE: bool +if not TYPE_CHECKING: + CUDA_BINDINGS_NVML_IS_COMPATIBLE = True # Please keep in sync with the equivalent implementation in diff --git a/cuda_core/cuda/core/system/typing.py b/cuda_core/cuda/core/system/typing.py index 50e3356f8ff..a083163379b 100644 --- a/cuda_core/cuda/core/system/typing.py +++ b/cuda_core/cuda/core/system/typing.py @@ -2,8 +2,9 @@ # # SPDX-License-Identifier: Apache-2.0 +import enum + from cuda.bindings import nvml as _nvml -from cuda.bindings._internal._fast_enum import FastEnum as _FastEnum from cuda.core._utils.pycompat import StrEnum __all__ = [ @@ -325,9 +326,9 @@ class ThermalTarget(StrEnum): # DeviceArch values are derived from cuda.bindings.nvml at definition time. -# This uses FastEnum instead of StrEnum because the ordering of the values is -# meaningful, e.g. Kepler "or later" -class DeviceArch(_FastEnum): +# An IntEnum rather than a StrEnum because the ordering of the values is +# meaningful, e.g. Kepler "or later". +class DeviceArch(enum.IntEnum): """ Device architecture. """ @@ -346,7 +347,7 @@ class DeviceArch(_FastEnum): FieldId = _nvml.FieldId -del _nvml, _FastEnum +del _nvml del StrEnum diff --git a/cuda_core/docs/source/api.rst b/cuda_core/docs/source/api.rst index 76a228625e8..ec898c039ec 100644 --- a/cuda_core/docs/source/api.rst +++ b/cuda_core/docs/source/api.rst @@ -166,14 +166,15 @@ Parameter-bearing definition nodes expose subclass-specific ``update()`` methods: :class:`~graph.KernelNode`, :class:`~graph.MemcpyNode`, :class:`~graph.MemsetNode`, :class:`~graph.ChildGraphNode`, :class:`~graph.EventRecordNode`, :class:`~graph.EventWaitNode`, and -:class:`~graph.HostCallbackNode`. These methods require CUDA driver and -``cuda.bindings`` versions 12.2 or newer. Updates affect future graph +:class:`~graph.HostCallbackNode`. These methods require a CUDA driver of +version 12.2 or newer. Updates affect future graph instantiations; executable graphs that were already instantiated continue using their previous parameters and retained resources. Omitted optional arguments preserve their current values where supported. -On CUDA 12.2 through 13.1, the intended CUDA context must be current when -updating memcpy or memset nodes. CUDA driver and ``cuda.bindings`` versions -13.2 and newer preserve the recorded context automatically. +With drivers from CUDA 12.2 through 13.1, the intended CUDA context must be +current when updating memcpy or memset nodes. With the CUDA 13 build of +``cuda.core`` and a driver of CUDA 13.2 or newer, the recorded context is +preserved automatically. Multidimensional or array-backed memcpy nodes and clustered or cooperative kernel nodes cannot currently be updated. Clustered and cooperative kernel nodes also cannot currently be constructed explicitly. @@ -217,8 +218,8 @@ Memcpy and memset updates use the current CUDA context, which must match the original node context. Kernel, memcpy, and memset views also provide ``is_enabled``, ``enable()``, and -``disable()``. Executable-node updates require CUDA driver and -``cuda.bindings`` versions 12.2 or newer. +``disable()``. Executable-node updates require a CUDA driver of version 12.2 +or newer. .. autosummary:: :toctree: generated/ @@ -334,8 +335,8 @@ CUDA process checkpointing The :mod:`cuda.core.checkpoint` module wraps the CUDA driver process checkpoint APIs. These APIs are intended for Linux process checkpoint and -restore workflows, and require a CUDA driver with checkpoint API support and -a ``cuda-bindings`` version that exposes those driver entry points. +restore workflows, and require the CUDA 13 build of ``cuda.core`` and a CUDA +driver of version 12.8 or newer with checkpoint API support. Checkpointing is typically driven by a coordinator process acting on a target CUDA process, similar to attaching a debugger or sending a signal. The target diff --git a/cuda_core/docs/source/release/1.3.0-notes.rst b/cuda_core/docs/source/release/1.3.0-notes.rst index fc9f067c948..f403ee7e94e 100644 --- a/cuda_core/docs/source/release/1.3.0-notes.rst +++ b/cuda_core/docs/source/release/1.3.0-notes.rst @@ -25,7 +25,7 @@ Breaking Changes the ``cuda-bindings`` version are gone, and the error messages they produced with them; error messages that name a minimum now name a driver version. ``cuda.core.system`` always uses NVML through ``cuda-bindings``, so ``cuda.core.system.CUDA_BINDINGS_NVML_IS_COMPATIBLE`` is always - ``True`` and is deprecated. :mod:`cuda.core.checkpoint` requires the CUDA 13 build of + ``True``. :mod:`cuda.core.checkpoint` requires the CUDA 13 build of ``cuda.core`` (it did in effect before: the CUDA 12 ``cuda-bindings`` lack a type it uses) and now says so. The C++ layer calls the driver through the entry points ``cuda-bindings`` resolves rather than through its Cython wrappers, so a driver function that the installed driver lacks diff --git a/cuda_core/tests/test_enum_coverage.py b/cuda_core/tests/test_enum_coverage.py index d38a5b15d00..6e941b1dda7 100644 --- a/cuda_core/tests/test_enum_coverage.py +++ b/cuda_core/tests/test_enum_coverage.py @@ -127,14 +127,8 @@ _MODULES.append(system_typing) -_CLOCKS_EVENT_REASONS_STR_UNMAPPED = { - core_member - for binding_member, core_member in ( - ("EVENT_REASON_BOARD_LIMIT", "BOARD_LIMIT"), - ("EVENT_REASON_RELIABILITY", "RELIABILITY"), - ) - if binding_member not in nvml.ClocksEventReasons.__members__ -} +# Every ClocksEventReasons member is mapped: the floor cuda-bindings has them all. +_CLOCKS_EVENT_REASONS_STR_UNMAPPED = set() _CASES.extend( [ diff --git a/cuda_core/tests/test_program.py b/cuda_core/tests/test_program.py index b14c313cf20..d3d15eefa83 100644 --- a/cuda_core/tests/test_program.py +++ b/cuda_core/tests/test_program.py @@ -38,9 +38,7 @@ def _is_nvvm_available(): return False -nvvm_available = pytest.mark.skipif( - not _is_nvvm_available(), reason="NVVM not available (libNVVM not found or cuda-bindings < 12.9.0)" -) +nvvm_available = pytest.mark.skipif(not _is_nvvm_available(), reason="NVVM not available (libNVVM not found)") def _get_nvrtc_version_for_tests(): diff --git a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py index 2eb0a61ffca..0193c6ca003 100644 --- a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py +++ b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py @@ -30,24 +30,13 @@ def hardware_supports_nvml(): def _should_skip_nvml_tests() -> bool: - """Return True if NVML tests should be skipped on this system. - - Checks cuda.core's compatibility gate first (if cuda.core is installed), - then falls back to a hardware-level NVML probe. - """ - try: - from cuda.core import system - - if not system.CUDA_BINDINGS_NVML_IS_COMPATIBLE: - return True - except ImportError: - pass # cuda.core not installed; skip the compat gate + """Return True if NVML tests should be skipped on this system (a hardware-level NVML probe).""" return not hardware_supports_nvml() skip_if_nvml_unsupported = pytest.mark.skipif( _should_skip_nvml_tests(), - reason="NVML support requires cuda.bindings version 12.9.6+ for CUDA 12.x or 13.2.0+ for CUDA 13.x, and hardware that supports NVML", + reason="NVML support requires hardware that supports NVML", ) From 6ba90045fc96b1209a8e4e151a01ffe5a654076c Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Wed, 23 Sep 2026 18:10:47 -0700 Subject: [PATCH 13/18] cuda-bindings header check at build; final floor messages; bounds spelled (#2783) - cuda_bindings/build_hooks.py: before cythonize, compare the toolkit's cuda.h CUDA_VERSION with the one the sources were generated from (cydriver.pxd). A mismatch fails with a message that names the resolved cuda.h, both versions, the run-time compatibility, and the two remedies. Tests in cuda_bindings/tests/test_build_hooks.py; install page and 13.5.0 release note. - cuda.core build-time and import-time messages: name the resolved cuda.h, state what is needed and what was found, say that the rule does not constrain the CUDA driver or toolkit at run time, and link the support policy. - Review follow-ups: the cu12/cu13 extras read `>=,<`; the floor reader accepts ` --- cuda_bindings/AGENTS.md | 3 + cuda_bindings/build_hooks.py | 75 ++++++++++++++ cuda_bindings/docs/source/install.rst | 2 +- .../docs/source/release/13.5.0-notes.rst | 1 + cuda_bindings/tests/test_build_hooks.py | 98 +++++++++++++++++++ cuda_core/AGENTS.md | 2 +- cuda_core/build_hooks.py | 20 ++-- cuda_core/cuda/core/_bindings_floor.py | 61 ++++++++---- cuda_core/docs/source/support.rst | 4 +- cuda_core/pyproject.toml | 6 +- cuda_core/tests/test_bindings_floor.py | 47 +++++---- cuda_core/tests/test_build_hooks.py | 21 ++-- toolshed/check_cuda_core_bindings_floor.py | 2 +- 13 files changed, 282 insertions(+), 60 deletions(-) create mode 100644 cuda_bindings/tests/test_build_hooks.py diff --git a/cuda_bindings/AGENTS.md b/cuda_bindings/AGENTS.md index 8c544f8872a..33471b0bb47 100644 --- a/cuda_bindings/AGENTS.md +++ b/cuda_bindings/AGENTS.md @@ -51,6 +51,9 @@ the `legacy_tests` subdirectory. - `CUDA_HOME` or `CUDA_PATH` must point to a valid CUDA Toolkit for source builds. +- The toolkit's `cuda.h` must have the major.minor the generated sources came + from (`CUDA_VERSION` in `cuda/bindings/cydriver.pxd`). `build_hooks.py` + checks this before cythonize and fails with a message that names both. - `CUDA_PYTHON_PARALLEL_LEVEL` controls build parallelism. - Runtime behavior is affected by `CUDA_PYTHON_CUDA_PER_THREAD_DEFAULT_STREAM` and diff --git a/cuda_bindings/build_hooks.py b/cuda_bindings/build_hooks.py index a53438e14cf..61da79137e8 100644 --- a/cuda_bindings/build_hooks.py +++ b/cuda_bindings/build_hooks.py @@ -12,6 +12,7 @@ import functools import glob import os +import re import shutil import sys import sysconfig @@ -33,6 +34,13 @@ # Populated by _build_cuda_bindings(); consumed by setup.py. _extensions = None +# The generated sources declare the types and functions of one CUDA header set; +# cydriver.pxd records which. The install docs state the resulting build rule. +_CYDRIVER_PXD = Path(__file__).resolve().parent / "cuda" / "bindings" / "cydriver.pxd" +_GENERATED_VERSION_RE = re.compile(r"^cdef enum:\s*CUDA_VERSION\s*=\s*(\d+)\s*$") +_CUDA_H_VERSION_RE = re.compile(r"^#\s*define\s+CUDA_VERSION\s+(\d+)\s*$") +_INSTALL_URL = "https://nvidia.github.io/cuda-python/cuda-bindings/latest/install.html#installing-from-source" + # Please keep in sync with the copy in cuda_core/build_hooks.py. def _import_get_cuda_path_or_home(): @@ -80,6 +88,72 @@ def _get_cuda_path() -> str: return cuda_path +# ----------------------------------------------------------------------- +# CUDA header check + + +def _cuda_h_path(cuda_path: str) -> str: + """The cuda.h under cuda_path, with symlinks such as /usr/local/cuda resolved for messages.""" + return os.path.realpath(os.path.join(cuda_path, "include", "cuda.h")) + + +def _read_version_macro(path: str, pattern: re.Pattern) -> int | None: + """The integer on the first line of ``path`` that matches ``pattern``, or None if no line does.""" + with open(path, encoding="utf-8") as f: + for line in f: + m = pattern.match(line) + if m: + return int(m.group(1)) + return None + + +def _read_cuda_h_version(cuda_path: str) -> int: + """The CUDA_VERSION macro (e.g. 13040 for 13.4) of the cuda.h under cuda_path.""" + cuda_h = _cuda_h_path(cuda_path) + try: + version = _read_version_macro(cuda_h, _CUDA_H_VERSION_RE) + except OSError: + version = None + if version is None: + raise RuntimeError( + f"Cannot read CUDA_VERSION from {cuda_h}. " + "Ensure CUDA_PATH or CUDA_HOME points to a CUDA Toolkit with include/cuda.h." + ) + return version + + +def _generated_cuda_version() -> int: + """The CUDA_VERSION of the headers this source tree was generated from (cuda/bindings/cydriver.pxd).""" + version = _read_version_macro(str(_CYDRIVER_PXD), _GENERATED_VERSION_RE) + if version is None: + raise RuntimeError(f"Cannot read CUDA_VERSION from {_CYDRIVER_PXD}") + return version + + +def _major_minor(cuda_version: int) -> str: + """13040 -> \"13.4\".""" + return f"{cuda_version // 1000}.{cuda_version // 10 % 100}" + + +def _check_cuda_headers(cuda_path: str) -> None: + """Reject a toolkit whose cuda.h is not the major.minor this source tree was generated from. + + Against another minor, the C++ compile fails with a long list of + redefinition and undeclared-type errors that do not name the cause + (https://github.com/NVIDIA/cuda-python/issues/2783). Runs before + cythonize, which is the first step that touches the source tree. + """ + generated = _generated_cuda_version() + needed, found = _major_minor(generated), _major_minor(_read_cuda_h_version(cuda_path)) + if found != needed: + raise RuntimeError( + f"This cuda-bindings source tree needs CUDA {needed} headers, but {_cuda_h_path(cuda_path)} is " + f"CUDA {found}. This is a build-time requirement only: at run time cuda-bindings supports any " + f"CUDA {generated // 1000}.x toolkit, see {_INSTALL_URL}. Point CUDA_PATH or CUDA_HOME at a " + f"CUDA {needed} toolkit, or build from cuda-bindings {found}.x sources." + ) + + # ----------------------------------------------------------------------- # Extension preparation helpers @@ -142,6 +216,7 @@ def _build_cuda_bindings(debug=False): global _extensions cuda_path = _get_cuda_path() + _check_cuda_headers(cuda_path) if os.environ.get("PARALLEL_LEVEL") is not None: warn( diff --git a/cuda_bindings/docs/source/install.rst b/cuda_bindings/docs/source/install.rst index d77464ec91f..796ac1a2ed0 100644 --- a/cuda_bindings/docs/source/install.rst +++ b/cuda_bindings/docs/source/install.rst @@ -128,7 +128,7 @@ Requirements [^3]: The version is derived from git tags via ``setuptools-scm``, so the clone must include tags reaching back to at least the latest ``v*`` tag. Clone with ``git clone https://github.com/NVIDIA/cuda-python.git``; do not use ``--depth`` or ``--no-tags``, since a shallow clone builds without error but produces a bogus version such as ``0.1.dev1+g0d22cb444``. See `Cloning the repository `_ for details and recovery steps. -Source builds require that the provided CUDA headers are of the same major.minor version as the ``cuda.bindings`` you're trying to build. Despite this requirement, note that the minor version compatibility is still maintained. Use the ``CUDA_PATH`` (or ``CUDA_HOME``) environment variable to specify the location of your headers. If both are set, ``CUDA_PATH`` takes precedence. For example, if your headers are located in ``/usr/local/cuda/include``, then you should set ``CUDA_PATH`` with: +Source builds require that the provided CUDA headers are of the same major.minor version as the ``cuda.bindings`` you're trying to build. Despite this requirement, note that the minor version compatibility is still maintained. The build checks the header before it compiles anything; a mismatch stops it with a message that names the ``cuda.h`` it found and the version this source tree needs. Use the ``CUDA_PATH`` (or ``CUDA_HOME``) environment variable to specify the location of your headers. If both are set, ``CUDA_PATH`` takes precedence. For example, if your headers are located in ``/usr/local/cuda/include``, then you should set ``CUDA_PATH`` with: .. code-block:: console diff --git a/cuda_bindings/docs/source/release/13.5.0-notes.rst b/cuda_bindings/docs/source/release/13.5.0-notes.rst index cb1f16e8fe4..2f78b472b11 100644 --- a/cuda_bindings/docs/source/release/13.5.0-notes.rst +++ b/cuda_bindings/docs/source/release/13.5.0-notes.rst @@ -301,6 +301,7 @@ Enhancements - ``nvvm.add_module_to_program``, ``nvvm.lazy_add_module_to_program``, ``nvvm.get_compiled_result``, ``nvvm.get_program_log`` (``buffer``) - For all ``cuda-bindings`` APIs, functions that accept a struct wrapper will accept the struct wrapper directly, rather than requiring getting the ``.ptr`` property. - Creating class instances in ``cuda-bindings`` should now be faster in most cases, because it requires 1 heap allocation rather than 2. +- A source build now checks, before it compiles anything, that the CUDA Toolkit's ``cuda.h`` has the major.minor this source tree was generated from. A mismatch stops the build with a message that names the header found and the version needed. It used to surface as a long list of C++ redefinition errors. Behavior changes ---------------- diff --git a/cuda_bindings/tests/test_build_hooks.py b/cuda_bindings/tests/test_build_hooks.py new file mode 100644 index 00000000000..9b5a8923cd1 --- /dev/null +++ b/cuda_bindings/tests/test_build_hooks.py @@ -0,0 +1,98 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""build_hooks.py: the CUDA header check that runs before cythonize. + +A cuda-bindings source tree is generated from one CUDA header set and compiles +only against a toolkit of that major.minor; against another minor the C++ +compile fails with redefinition errors that do not name the cause. The check +reads both versions and fails early with a message that does. No GPU needed. + +build_hooks.py is a PEP 517 backend, not an installed module, so it is loaded +from source. It imports setuptools at the top; the ``test`` extra provides it. +""" + +import importlib.util +import os +from pathlib import Path + +import pytest +import setuptools # noqa: F401 + +# Don't call .resolve(): a symlinked checkout would make parents[1] point elsewhere. +BUILD_HOOKS = Path(__file__).parents[1] / "build_hooks.py" + + +@pytest.fixture(scope="module") +def build_hooks(): + if not BUILD_HOOKS.is_file(): + pytest.skip(f"{BUILD_HOOKS} is not in this tree; these tests need the source checkout") + spec = importlib.util.spec_from_file_location("cuda_bindings_build_hooks", BUILD_HOOKS) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def _write_cuda_h(tmp_path, cuda_version): + include = tmp_path / "include" + include.mkdir(exist_ok=True) + (include / "cuda.h").write_text(f"#define CUDA_VERSION {cuda_version}\n", encoding="utf-8") + return str(tmp_path) + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_generated_header_version_is_read_from_cydriver_pxd(build_hooks): + generated = build_hooks._generated_cuda_version() + assert generated // 1000 in (12, 13) + assert build_hooks._major_minor(13040) == "13.4" + assert build_hooks._major_minor(12090) == "12.9" + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_a_header_of_the_generated_major_minor_passes(build_hooks, tmp_path): + generated = build_hooks._generated_cuda_version() + build_hooks._check_cuda_headers(_write_cuda_h(tmp_path, generated)) + # Only major.minor matters; the last digit (13041) is a toolkit patch. + build_hooks._check_cuda_headers(_write_cuda_h(tmp_path, generated + 1)) + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +@pytest.mark.parametrize("delta", [-10, 10, -1000, 1000]) +def test_another_header_fails_and_names_both_versions(build_hooks, tmp_path, delta): + generated = build_hooks._generated_cuda_version() + cuda_path = _write_cuda_h(tmp_path, generated + delta) + with pytest.raises(RuntimeError) as excinfo: + build_hooks._check_cuda_headers(cuda_path) + message = str(excinfo.value) + needed, found = build_hooks._major_minor(generated), build_hooks._major_minor(generated + delta) + assert message.startswith(f"This cuda-bindings source tree needs CUDA {needed} headers, but ") + assert os.path.realpath(os.path.join(cuda_path, "include", "cuda.h")) in message # the resolved path + assert f" is CUDA {found}. This is a build-time requirement only" in message + assert build_hooks._INSTALL_URL in message + assert message.endswith( + f"Point CUDA_PATH or CUDA_HOME at a CUDA {needed} toolkit, or build from cuda-bindings {found}.x sources." + ) + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_an_unreadable_cuda_h_is_a_clear_error(build_hooks, tmp_path): + with pytest.raises(RuntimeError, match=r"Cannot read CUDA_VERSION from .*cuda\.h"): + build_hooks._check_cuda_headers(str(tmp_path)) # no include/cuda.h + (tmp_path / "include").mkdir() + (tmp_path / "include" / "cuda.h").write_text("/* no CUDA_VERSION macro */\n", encoding="utf-8") + with pytest.raises(RuntimeError, match=r"Cannot read CUDA_VERSION from .*cuda\.h"): + build_hooks._check_cuda_headers(str(tmp_path)) + + +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_the_build_checks_the_header_before_it_touches_the_tree(build_hooks, tmp_path, monkeypatch): + pytest.importorskip("Cython") # _build_cuda_bindings imports it first + cuda_path = _write_cuda_h(tmp_path, build_hooks._generated_cuda_version() + 10) + monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: cuda_path) + + def not_reached(): + raise AssertionError("the header check must run before the source tree is modified") + + monkeypatch.setattr(build_hooks, "_rename_architecture_specific_files", not_reached) + with pytest.raises(RuntimeError, match="source tree needs CUDA .* headers"): + build_hooks._build_cuda_bindings() diff --git a/cuda_core/AGENTS.md b/cuda_core/AGENTS.md index a749c6a9024..c5669f14412 100644 --- a/cuda_core/AGENTS.md +++ b/cuda_core/AGENTS.md @@ -34,7 +34,7 @@ This file describes `cuda_core`, the high-level Pythonic CUDA subpackage in the - `cuda_core` requires `cuda.bindings` at or above a per-major *floor* at build and run time, and a `cuda.h` of the same major.minor as that `cuda.bindings` at build time (NVIDIA/cuda-python#2783). The floors are declared once, by the - `cu12`/`cu13` extras in `pyproject.toml` (`cuda-bindings[all]>=,==.*`); + `cu12`/`cu13` extras in `pyproject.toml` (`cuda-bindings[all]>=,<`); `build_hooks.py`, the import-time check in `cuda/core/__init__.py`, the docs and CI read them from there through `cuda/core/_bindings_floor.py`. The C++ branches on `CUDA_CORE_BUILD_MAJOR` only; whether a feature is available at diff --git a/cuda_core/build_hooks.py b/cuda_core/build_hooks.py index 62676a0cacc..ab9f14eb5f3 100644 --- a/cuda_core/build_hooks.py +++ b/cuda_core/build_hooks.py @@ -176,9 +176,14 @@ def _floor_for(cuda_major) -> tuple: return floors[major] +def _cuda_h_path(cuda_path: str) -> str: + """The cuda.h under cuda_path, with symlinks such as /usr/local/cuda resolved for messages.""" + return os.path.realpath(os.path.join(cuda_path, "include", "cuda.h")) + + def _read_cuda_h_version(cuda_path: str) -> int: """The CUDA_VERSION macro (e.g. 13040 for 13.4) of the cuda.h under cuda_path.""" - cuda_h = os.path.join(cuda_path, "include", "cuda.h") + cuda_h = _cuda_h_path(cuda_path) try: with open(cuda_h, encoding="utf-8") as f: for line in f: @@ -281,12 +286,13 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: header = floor.header_minor(cuda_version) generated_from = floor.header_minor(bindings_cuda_version) if header != generated_from: + needed, found = f"{generated_from[0]}.{generated_from[1]}", f"{header[0]}.{header[1]}" raise RuntimeError( - f"cuda.h under {cuda_path} is CUDA {header[0]}.{header[1]}, but the installed cuda-bindings " - f"{bindings_version} was generated from CUDA {generated_from[0]}.{generated_from[1]} headers. " - "cuda.core must be built against the cuda.h its cuda-bindings was generated from. Point " - "CUDA_PATH or CUDA_HOME at that CUDA Toolkit, or install a matching cuda-bindings (for an " - "isolated build, constrain it with PIP_CONSTRAINT or build with --no-build-isolation)." + f"cuda.core needs CUDA {needed} headers to build with the installed cuda-bindings {bindings_version}, " + f"but {_cuda_h_path(cuda_path)} is CUDA {found}. This is a build-time requirement only: at run time " + f"cuda.core supports older CUDA {major}.x drivers and toolkits, see {floor.SUPPORT_URL}. Point " + f"CUDA_PATH or CUDA_HOME at a CUDA {needed} toolkit, or install cuda-bindings {found}.x. For an " + "isolated build, constrain cuda-bindings with PIP_CONSTRAINT or build with --no-build-isolation." ) print(f"Build configuration: CUDA {header[0]}.{header[1]} headers, cuda-bindings {bindings_version}") _write_build_info(major, cuda_version, floor_triple, bindings_version) @@ -665,7 +671,7 @@ def _get_cuda_bindings_require(): return [floor.bindings_requirement(floor_triple)] if header[0] != floor_triple[0] or header[1] < floor_triple[1]: return [floor.bindings_requirement(floor_triple)] - return [f"{floor.bindings_requirement(floor_triple)},<{header[0]}.{header[1] + 1}"] + return [floor.bindings_requirement(floor_triple, below=(header[0], header[1] + 1))] def get_requires_for_build_editable(config_settings=None): diff --git a/cuda_core/cuda/core/_bindings_floor.py b/cuda_core/cuda/core/_bindings_floor.py index 28b274fabc3..048b583ab5a 100644 --- a/cuda_core/cuda/core/_bindings_floor.py +++ b/cuda_core/cuda/core/_bindings_floor.py @@ -12,7 +12,7 @@ policy in the documentation. The floors are declared in exactly one place: the ``cu12`` and ``cu13`` extras -in ``pyproject.toml``, each of which pins ``cuda-bindings>=,==.*``. +in ``pyproject.toml``, each of which pins ``cuda-bindings>=,<``. Everything else derives from them through :func:`floors_from_extras`: - the build backend (``build_hooks.py``) checks the installed cuda-bindings @@ -36,6 +36,7 @@ from collections.abc import Mapping, Sequence __all__ = [ + "SUPPORT_URL", "bindings_requirement", "check_installed_bindings", "cuda_version_of", @@ -53,8 +54,13 @@ r"^\s*cuda-bindings(?=[\[<>=!~;\s]|$)\s*(?:\[[^\]]*\])?\s*(?P[^;]*?)\s*(?:;.*)?$" ) _FLOOR_SPEC_RE = re.compile(r"^>=(\d+)\.(\d+)\.(\d+)$") +# The upper bound: `<14` (also `<14.0`, `<14.0.0`) or the equivalent `==13.*`. +_UPPER_SPEC_RE = re.compile(r"^<(\d+)(?:\.0)*$") _MAJOR_SPEC_RE = re.compile(r"^==(\d+)\.\*$") +# The support policy section the error messages refer to. +SUPPORT_URL = "https://nvidia.github.io/cuda-python/cuda-core/latest/support.html#cuda-core-bindings-floor" + def release_triple(version: str) -> tuple[int, int, int] | None: """The leading ``major.minor.patch`` of a version string, or None. @@ -83,11 +89,12 @@ def floors_from_extras(extras: Mapping[str, Sequence[str]]) -> dict[int, tuple[i """The floor per CUDA major, from the ``cu`` extras of ``pyproject.toml``. ``extras`` is the parsed ``[project.optional-dependencies]`` table. Each - ``cu`` extra must list exactly one cuda-bindings requirement of the - form ``cuda-bindings[...]>=..,==.*`` (the order - of the two specifiers does not matter). Anything else raises ValueError - naming the extra, so a typo fails the build and the pre-commit hook - instead of shifting the floor silently. + ``cu`` extra must list exactly one cuda-bindings requirement with two + specifiers, in either order: the floor, ``>=..``, and an + upper bound that excludes the next major, ``<`` (``==.*`` is also + accepted). Anything else raises ValueError naming the extra, so a typo + fails the build and the pre-commit hook instead of shifting the floor + silently. """ floors: dict[int, tuple[int, int, int]] = {} for extra, requirements in extras.items(): @@ -103,17 +110,16 @@ def floors_from_extras(extras: Mapping[str, Sequence[str]]) -> dict[int, tuple[i requirement, requirement_match = matches[0] specifiers = [s.strip() for s in requirement_match.group("specifiers").split(",") if s.strip()] floor_matches = [fm for s in specifiers if (fm := _FLOOR_SPEC_RE.match(s)) is not None] - major_matches = [mm for s in specifiers if (mm := _MAJOR_SPEC_RE.match(s)) is not None] - if len(specifiers) != 2 or len(floor_matches) != 1 or len(major_matches) != 1: + confined = [cm for s in specifiers if (cm := _confined_major(s)) is not None] + if len(specifiers) != 2 or len(floor_matches) != 1 or len(confined) != 1: raise ValueError( - f"the {extra!r} extra must pin cuda-bindings as '>=..,==.*', " - f"found {requirement!r}" + f"the {extra!r} extra must pin cuda-bindings as '>=,<' " + f"(for example '>=13.4.1,<14'), found {requirement!r}" ) floor = (int(floor_matches[0].group(1)), int(floor_matches[0].group(2)), int(floor_matches[0].group(3))) - pinned_major = int(major_matches[0].group(1)) - if floor[0] != major or pinned_major != major: + if floor[0] != major or confined[0] != major: raise ValueError( - f"the {extra!r} extra pins cuda-bindings {requirement!r}, whose majors do not match CUDA {major}" + f"the {extra!r} extra pins cuda-bindings {requirement!r}, which does not confine it to CUDA {major}" ) floors[major] = floor if not floors: @@ -121,9 +127,23 @@ def floors_from_extras(extras: Mapping[str, Sequence[str]]) -> dict[int, tuple[i return dict(sorted(floors.items())) -def bindings_requirement(floor: tuple[int, int, int]) -> str: - """The pip requirement that pins cuda-bindings to ``floor`` and its major, without extras.""" - return f"cuda-bindings>={format_version(floor)},=={floor[0]}.*" +def _confined_major(specifier: str) -> int | None: + """The one major a specifier confines cuda-bindings to: ``<14`` -> 13, ``==13.*`` -> 13, else None.""" + if (m := _UPPER_SPEC_RE.match(specifier)) is not None: + return int(m.group(1)) - 1 + if (m := _MAJOR_SPEC_RE.match(specifier)) is not None: + return int(m.group(1)) + return None + + +def bindings_requirement(floor: tuple[int, int, int], below: tuple[int, ...] | None = None) -> str: + """The pip requirement for cuda-bindings at or above ``floor`` and below ``below``, without extras. + + ``below`` defaults to the next major: ``(13, 4, 1)`` gives ``cuda-bindings>=13.4.1,<14``; + with ``below=(13, 5)`` it gives ``cuda-bindings>=13.4.1,<13.5``. + """ + upper = format_version(below) if below is not None else str(floor[0] + 1) + return f"cuda-bindings>={format_version(floor)},<{upper}" def header_minor(cuda_version: int) -> tuple[int, int]: @@ -168,14 +188,15 @@ def check_installed_bindings( floor = format_version(build_floor) raise ImportError( f"cuda.core {core_version} requires cuda-bindings >= {floor} for CUDA {major} " - f'(found {installed_version}). Upgrade with: pip install -U "cuda-bindings>={floor},=={major}.*"' + f'(found {installed_version}). Upgrade with: pip install -U "cuda-bindings>={floor},<{major + 1}"' ) built_against, generated_from = header_minor(build_cuda_version), header_minor(installed_cuda_version) if generated_from < built_against: needed = f"{built_against[0]}.{built_against[1]}" raise ImportError( - f"cuda.core {core_version} was built against CUDA {needed} headers, but the installed cuda-bindings " - f"{installed_version} was generated from CUDA {generated_from[0]}.{generated_from[1]} headers. " - f'Install cuda-bindings {needed} or newer: pip install -U "cuda-bindings>={needed}.0,=={major}.*"' + f"cuda.core {core_version} was compiled against CUDA {needed} headers and needs cuda-bindings " + f"{needed} or newer, but cuda-bindings {installed_version} is installed. This does not require a " + f"newer CUDA driver or toolkit: cuda.core supports older CUDA {major}.x versions at run time, " + f'see {SUPPORT_URL}. Upgrade with: pip install -U "cuda-bindings>={needed}.0,<{major + 1}"' ) return installed diff --git a/cuda_core/docs/source/support.rst b/cuda_core/docs/source/support.rst index a7c554f3cab..cbb2d300551 100644 --- a/cuda_core/docs/source/support.rst +++ b/cuda_core/docs/source/support.rst @@ -98,8 +98,8 @@ there. ``cuda-bindings`` was generated from. Any other configuration fails the build with a message that names what was found and what is required. Building against an older CUDA Toolkit than the floor's minor is not supported. -- **The CUDA driver** is unaffected. Feature availability is decided by the driver alone: a - feature the installed driver lacks raises when it is used, as before. +- **The CUDA driver** is unaffected by the floor. Feature availability is decided by the driver alone: a + feature the installed driver lacks raises when it is used. A floor moves with each ``cuda-core`` release, to the newest ``cuda-bindings`` of each major at that time, and in any release whose changes need a newer ``cuda-bindings`` API. Every move is diff --git a/cuda_core/pyproject.toml b/cuda_core/pyproject.toml index 91ce246c4d1..f4fad12ef6b 100644 --- a/cuda_core/pyproject.toml +++ b/cuda_core/pyproject.toml @@ -59,10 +59,10 @@ dependencies = [ # per CUDA major, and the only place they are declared: build_hooks.py, the # import-time check, the docs and CI read them from here (see # cuda/core/_bindings_floor.py and the "Bumping the cuda-bindings floor" -# checklist in AGENTS.md). Keep the form `>=,==.*`. +# checklist in AGENTS.md). Keep the form `>=,<`. [project.optional-dependencies] -cu12 = ["cuda-bindings[all]>=12.9.8,==12.*", "cuda-toolkit==12.*"] -cu13 = ["cuda-bindings[all]>=13.4.1,==13.*", "cuda-toolkit==13.*"] +cu12 = ["cuda-bindings[all]>=12.9.8,<13", "cuda-toolkit==12.*"] +cu13 = ["cuda-bindings[all]>=13.4.1,<14", "cuda-toolkit==13.*"] [dependency-groups] test = [ diff --git a/cuda_core/tests/test_bindings_floor.py b/cuda_core/tests/test_bindings_floor.py index 38dfd7471db..4e1ee66fe8b 100644 --- a/cuda_core/tests/test_bindings_floor.py +++ b/cuda_core/tests/test_bindings_floor.py @@ -77,12 +77,15 @@ def test_floors_come_from_the_pyproject_extras(hook): @pytest.mark.agent_authored(model="claude-fable-5-1") def test_floors_from_extras_accepts_the_declared_form(): extras = { - "cu12": ["cuda-bindings[all]>=12.9.8,==12.*", "cuda-toolkit==12.*"], - "cu13": ["cuda-bindings[all]==13.*,>=13.4.1", "cuda-toolkit==13.*"], + "cu12": ["cuda-bindings[all]>=12.9.8,<13", "cuda-toolkit==12.*"], + "cu13": ["cuda-bindings[all]<14,>=13.4.1", "cuda-toolkit==13.*"], "test": ["pytest"], } assert floors_from_extras(extras) == {12: (12, 9, 8), 13: (13, 4, 1)} - assert floors_from_extras({"cu13": ['cuda-bindings>=13.4.1,==13.* ; python_version >= "3.10"']}) == {13: (13, 4, 1)} + assert floors_from_extras({"cu13": ['cuda-bindings>=13.4.1,<14 ; python_version >= "3.10"']}) == {13: (13, 4, 1)} + # Other spellings of the same upper bound. + assert floors_from_extras({"cu13": ["cuda-bindings>=13.4.1,<14.0"]}) == {13: (13, 4, 1)} + assert floors_from_extras({"cu13": ["cuda-bindings>=13.4.1,==13.*"]}) == {13: (13, 4, 1)} @pytest.mark.agent_authored(model="claude-fable-5-1") @@ -90,12 +93,15 @@ def test_floors_from_extras_accepts_the_declared_form(): ("extras", "message"), [ ({"cu13": ["cuda-toolkit==13.*"]}, "exactly one cuda-bindings requirement, found 0"), - ({"cu13": ["cuda-bindings>=13.4.1", "cuda-bindings==13.*"]}, "exactly one cuda-bindings requirement, found 2"), + ({"cu13": ["cuda-bindings>=13.4.1", "cuda-bindings<14"]}, "exactly one cuda-bindings requirement, found 2"), ({"cu13": ["cuda-bindings>=13.4.1"]}, "must pin cuda-bindings as"), - ({"cu13": ["cuda-bindings>=13.4,==13.*"]}, "must pin cuda-bindings as"), - ({"cu13": ["cuda-bindings>=13.4.1,==13.*,<14"]}, "must pin cuda-bindings as"), - ({"cu13": ["cuda-bindings>=12.9.8,==13.*"]}, "majors do not match CUDA 13"), - ({"cu13": ["cuda-bindings>=13.4.1,==12.*"]}, "majors do not match CUDA 13"), + ({"cu13": ["cuda-bindings>=13.4,<14"]}, "must pin cuda-bindings as"), + ({"cu13": ["cuda-bindings>=13.4.1,<14,==13.*"]}, "must pin cuda-bindings as"), + ({"cu13": ["cuda-bindings>=13.4.1,<13.5"]}, "must pin cuda-bindings as"), # a minor, not a major + ({"cu13": ["cuda-bindings>=12.9.8,<14"]}, "does not confine it to CUDA 13"), + ({"cu13": ["cuda-bindings>=13.4.1,<13"]}, "does not confine it to CUDA 13"), + ({"cu13": ["cuda-bindings>=13.4.1,<15"]}, "does not confine it to CUDA 13"), + ({"cu13": ["cuda-bindings>=13.4.1,==12.*"]}, "does not confine it to CUDA 13"), ({"test": ["pytest"]}, "declares no cu extra"), ], ) @@ -106,7 +112,7 @@ def test_floors_from_extras_rejects_other_forms(extras, message): @pytest.mark.agent_authored(model="claude-fable-5-1") def test_floors_from_extras_ignores_lookalike_names(): - extras = {"cu13": ["cuda-bindings-extra>=1.0", "cuda-bindings>=13.4.1,==13.*"]} + extras = {"cu13": ["cuda-bindings-extra>=1.0", "cuda-bindings>=13.4.1,<14"]} assert floors_from_extras(extras) == {13: (13, 4, 1)} @@ -133,7 +139,8 @@ def test_formatting_helpers(): assert format_version((13, 4, 1)) == "13.4.1" assert cuda_version_of((13, 4, 1)) == 13040 assert cuda_version_of((12, 9, 8)) == 12090 - assert bindings_requirement((13, 4, 1)) == "cuda-bindings>=13.4.1,==13.*" + assert bindings_requirement((13, 4, 1)) == "cuda-bindings>=13.4.1,<14" + assert bindings_requirement((13, 4, 1), below=(13, 5)) == "cuda-bindings>=13.4.1,<13.5" @pytest.mark.agent_authored(model="claude-fable-5-1") @@ -177,7 +184,7 @@ def test_rejects_older_than_the_floor_with_the_fix(self): message = str(excinfo.value) assert "requires cuda-bindings >= 13.4.1 for CUDA 13" in message assert "(found 13.3.1)" in message - assert 'pip install -U "cuda-bindings>=13.4.1,==13.*"' in message # double quotes: cmd.exe too + assert 'pip install -U "cuda-bindings>=13.4.1,<14"' in message # double quotes: cmd.exe too @pytest.mark.agent_authored(model="claude-fable-5-1") def test_rejects_bindings_generated_from_an_older_header_than_the_build(self): @@ -185,9 +192,11 @@ def test_rejects_bindings_generated_from_an_older_header_than_the_build(self): with pytest.raises(ImportError) as excinfo: self.check("13.4.1", header=self.HEADER + 10) message = str(excinfo.value) - assert "was built against CUDA 13.5 headers" in message - assert "13.4.1 was generated from CUDA 13.4 headers" in message - assert 'pip install -U "cuda-bindings>=13.5.0,==13.*"' in message + assert "was compiled against CUDA 13.5 headers and needs cuda-bindings 13.5 or newer" in message + assert "but cuda-bindings 13.4.1 is installed" in message + assert "This does not require a newer CUDA driver or toolkit" in message + assert floor_mod.SUPPORT_URL in message + assert 'pip install -U "cuda-bindings>=13.5.0,<14"' in message @pytest.mark.agent_authored(model="claude-fable-5-1") def test_header_rule_compares_headers_not_version_strings(self): @@ -196,7 +205,7 @@ def test_header_rule_compares_headers_not_version_strings(self): floor = (13, 4, 2) assert self.check("13.4.2.dev5+gabc", header=13050, floor=floor, installed_header=13050) == (13, 4, 2) # The converse, a 13.5 version string generated from 13.4 headers, is rejected. - with pytest.raises(ImportError, match="was generated from CUDA 13.4 headers"): + with pytest.raises(ImportError, match="needs cuda-bindings 13.5 or newer"): self.check("13.5.0", header=13050, floor=floor, installed_header=13040) @pytest.mark.agent_authored(model="claude-fable-5-1") @@ -208,7 +217,7 @@ def test_the_floor_comes_from_the_build_record(self): self.check("13.4.1", floor=(13, 5, 0)) message = str(excinfo.value) assert "requires cuda-bindings >= 13.5.0 for CUDA 13" in message - assert 'pip install -U "cuda-bindings>=13.5.0,==13.*"' in message + assert 'pip install -U "cuda-bindings>=13.5.0,<14"' in message @pytest.mark.agent_authored(model="claude-fable-5-1") def test_rejects_another_major_than_the_build(self): @@ -285,14 +294,14 @@ def test_below_the_floor_fails_at_import_with_the_fix(self, tmp_path): below = f"{major}.0.1" message = self._import_error(below, major * 1000, tmp_path) assert f"requires cuda-bindings >= {format_version(floor)} for CUDA {major}" in message - assert f'pip install -U "cuda-bindings>={format_version(floor)},=={major}.*"' in message + assert f'pip install -U "cuda-bindings>={format_version(floor)},<{major + 1}"' in message @pytest.mark.agent_authored(model="claude-fable-5-1") def test_bindings_from_an_older_header_fail_at_import(self, tmp_path): major, cuda_version, floor = self._build() # At the floor by version, but generated from a header one minor below the build's. message = self._import_error(format_version(floor), cuda_version - 10, tmp_path) - assert f"was built against CUDA {major}.{header_minor(cuda_version)[1]} headers" in message + assert f"was compiled against CUDA {major}.{header_minor(cuda_version)[1]} headers" in message @pytest.mark.agent_authored(model="claude-fable-5-1") def test_unparseable_version_fails_at_import(self, tmp_path): @@ -322,7 +331,7 @@ def test_a_malformed_extra_fails_the_hook(self, hook, tmp_path, capsys): core.mkdir(parents=True) shutil.copy(CUDA_CORE / "cuda" / "core" / "_bindings_floor.py", core) (tmp_path / "cuda_core" / "pyproject.toml").write_text( - '[project.optional-dependencies]\ncu13 = ["cuda-bindings[all]>=13.4,==13.*"]\n', encoding="utf-8" + '[project.optional-dependencies]\ncu13 = ["cuda-bindings[all]>=13.4,<14"]\n', encoding="utf-8" ) assert hook.main(["--repo-root", str(tmp_path)]) == 1 assert "pyproject.toml: the 'cu13' extra must pin cuda-bindings" in capsys.readouterr().err diff --git a/cuda_core/tests/test_build_hooks.py b/cuda_core/tests/test_build_hooks.py index a87d222b326..90d0ed63775 100644 --- a/cuda_core/tests/test_build_hooks.py +++ b/cuda_core/tests/test_build_hooks.py @@ -544,10 +544,19 @@ def test_bindings_of_another_major_fail(self, tmp_path, monkeypatch): @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize("header", [13030, 13050, 12090]) def test_header_minor_must_match_bindings(self, tmp_path, monkeypatch, header): + floor = self.FLOOR[13] _fake_bindings(monkeypatch, _floor_str(13)) cuda_path = _write_cuda_h(tmp_path, header) - with pytest.raises(RuntimeError, match="must be built against the cuda.h its cuda-bindings was generated from"): + with pytest.raises(RuntimeError) as excinfo: build_hooks._check_build_configuration(cuda_path, "13") + message = str(excinfo.value) + assert message.startswith( + f"cuda.core needs CUDA {floor[0]}.{floor[1]} headers to build with the installed cuda-bindings {_floor_str(13)}, but " + ) + assert os.path.realpath(os.path.join(cuda_path, "include", "cuda.h")) in message # the resolved path + assert f"is CUDA {header // 1000}.{header // 10 % 100}." in message + assert "This is a build-time requirement only" in message + assert f"Point CUDA_PATH or CUDA_HOME at a CUDA {floor[0]}.{floor[1]} toolkit" in message @pytest.mark.agent_authored(model="claude-fable-5-1") def test_header_is_read_even_when_the_major_override_is_set(self, tmp_path, monkeypatch): @@ -555,7 +564,7 @@ def test_header_is_read_even_when_the_major_override_is_set(self, tmp_path, monk monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", "13") _fake_bindings(monkeypatch, _floor_str(13)) cuda_path = _write_cuda_h(tmp_path, 13030) - with pytest.raises(RuntimeError, match="was generated from CUDA 13.4 headers"): + with pytest.raises(RuntimeError, match=r"needs CUDA 13\.\d+ headers .* is CUDA 13\.3\."): build_hooks._check_build_configuration(cuda_path, "13") @pytest.mark.agent_authored(model="claude-fable-5-1") @@ -639,7 +648,7 @@ def test_pins_the_floor_and_the_major_without_a_header(self, monkeypatch, major) build_hooks._determine_cuda_major_version.cache_clear() floor = build_hooks._bindings_floors()[int(major)] (requirement,) = build_hooks._get_cuda_bindings_require() - assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},=={major}.*" + assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},<{int(major) + 1}" @pytest.mark.agent_authored(model="claude-fable-5-1") def test_caps_at_the_headers_minor_when_cuda_h_is_readable(self, tmp_path, monkeypatch): @@ -651,7 +660,7 @@ def test_caps_at_the_headers_minor_when_cuda_h_is_readable(self, tmp_path, monke cuda_path = _write_cuda_h(tmp_path, floor[0] * 1000 + (floor[1] + 1) * 10) monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: cuda_path) (requirement,) = build_hooks._get_cuda_bindings_require() - assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},==13.*,<13.{floor[1] + 2}" + assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},<13.{floor[1] + 2}" specifiers = SpecifierSet(requirement.removeprefix("cuda-bindings")) # The dev cuda-bindings built alongside a toolkit bump still carries the old minor's version string. assert specifiers.contains(f"13.{floor[1]}.{floor[2] + 1}.dev133", prereleases=True) @@ -667,7 +676,7 @@ def test_requests_the_floor_alone_when_the_header_is_of_another_major(self, tmp_ cuda_path = _write_cuda_h(tmp_path, 12090) monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: cuda_path) (requirement,) = build_hooks._get_cuda_bindings_require() - assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},==13.*" + assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},<14" @pytest.mark.agent_authored(model="claude-fable-5-1") def test_requests_the_floor_alone_when_the_header_is_below_it(self, tmp_path, monkeypatch): @@ -678,7 +687,7 @@ def test_requests_the_floor_alone_when_the_header_is_below_it(self, tmp_path, mo cuda_path = _write_cuda_h(tmp_path, floor[0] * 1000 + (floor[1] - 1) * 10) monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: cuda_path) (requirement,) = build_hooks._get_cuda_bindings_require() - assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},==13.*" + assert requirement == f"cuda-bindings>={floor[0]}.{floor[1]}.{floor[2]},<14" @pytest.mark.agent_authored(model="claude-fable-5-1") def test_unsupported_major_names_the_supported_ones(self, monkeypatch): diff --git a/toolshed/check_cuda_core_bindings_floor.py b/toolshed/check_cuda_core_bindings_floor.py index 0fd9db023a5..2e43c5b7d52 100644 --- a/toolshed/check_cuda_core_bindings_floor.py +++ b/toolshed/check_cuda_core_bindings_floor.py @@ -9,7 +9,7 @@ checked here, as a pre-commit hook and from cuda_core/tests/test_bindings_floor.py: 1. The extras parse: each `cu` extra pins exactly - `cuda-bindings[...]>=..,==.*`. + `cuda-bindings[...]>=..,<`. 2. ci/versions.yml builds each major against a CUDA Toolkit of at least the floor's major.minor. A toolkit below the floor's minor cannot build the floor's cuda-bindings (the build requires the header cuda-bindings was From b6cbcc0ab378a0e984341f50b932c69904f6e861 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Wed, 23 Sep 2026 18:10:48 -0700 Subject: [PATCH 14/18] Copyedit the docs, comments, docstrings and messages this PR adds (#2783) A plain-language pass over the prose the PR introduces: one idea per sentence, active voice, no semicolons or parenthetical asides, one term per concept (cuda-bindings for the package, cuda.bindings for the module). Error messages keep what they state and change only how they say it; the tests that match them follow. The regenerated stubs follow their docstrings. Co-Authored-By: Claude Fable 5.1 --- ci/tools/cuda_core_bindings_floor.py | 13 +- ci/tools/env-vars | 19 +-- ci/tools/run-tests | 4 +- cuda_bindings/AGENTS.md | 7 +- cuda_bindings/build_hooks.py | 19 +-- cuda_bindings/docs/source/install.rst | 2 +- .../docs/source/release/13.5.0-notes.rst | 2 +- cuda_bindings/tests/test_build_hooks.py | 16 +-- cuda_core/AGENTS.md | 100 +++++++------- cuda_core/build_hooks.py | 130 +++++++++--------- cuda_core/cuda/core/__init__.py | 31 +++-- cuda_core/cuda/core/_bindings_floor.py | 125 +++++++++-------- cuda_core/cuda/core/_cpp/rt/DESIGN.md | 129 +++++++++-------- cuda_core/cuda/core/_cpp/rt/context.cpp | 4 +- cuda_core/cuda/core/_cpp/rt/driver_api.cpp | 6 +- cuda_core/cuda/core/_cpp/rt/driver_api.hpp | 119 ++++++++-------- cuda_core/cuda/core/_cpp/rt/error.cpp | 2 +- cuda_core/cuda/core/_cpp/rt/error.hpp | 8 +- cuda_core/cuda/core/_cpp/rt/graph.cpp | 2 +- cuda_core/cuda/core/_cpp/rt/internal.hpp | 2 +- cuda_core/cuda/core/_cpp/rt/memory.cpp | 5 +- cuda_core/cuda/core/_cpp/rt/program.cpp | 2 +- cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp | 76 +++++----- cuda_core/cuda/core/_cpp/rt/stream.cpp | 2 +- cuda_core/cuda/core/_cpp/rt/texture.cpp | 4 +- cuda_core/cuda/core/_cpp/rt/versions.hpp | 36 ++--- cuda_core/cuda/core/_device.pyx | 2 +- cuda_core/cuda/core/_device_resources.pyx | 4 +- cuda_core/cuda/core/_linker.pyx | 4 +- cuda_core/cuda/core/_memory/_buffer.pyx | 2 +- .../cuda/core/_memory/_copy_attributes.pxd | 4 +- cuda_core/cuda/core/_memory/_copy_enums.py | 4 +- cuda_core/cuda/core/_memory/_copy_ops.pyi | 2 +- cuda_core/cuda/core/_memory/_copy_ops.pyx | 4 +- .../cuda/core/_memory/_managed_buffer.py | 8 +- .../cuda/core/_memory/_managed_location.py | 6 +- .../core/_memory/_virtual_memory_resource.py | 2 +- cuda_core/cuda/core/_program.pyx | 4 +- cuda_core/cuda/core/_stream.pyx | 4 +- .../core/_utils/enum_explanations_helpers.py | 10 +- cuda_core/cuda/core/_utils/version.pyx | 11 +- cuda_core/cuda/core/checkpoint.py | 14 +- cuda_core/cuda/core/graph/_subclasses.pyi | 12 +- cuda_core/cuda/core/graph/_subclasses.pyx | 16 +-- cuda_core/cuda/core/system/_system.pyx | 16 +-- cuda_core/cuda/core/system/typing.py | 4 +- cuda_core/docs/source/api.rst | 19 ++- cuda_core/docs/source/api_nvml.rst | 4 +- cuda_core/docs/source/conf.py | 8 +- cuda_core/docs/source/install.rst | 28 ++-- cuda_core/docs/source/release/1.3.0-notes.rst | 43 +++--- cuda_core/docs/source/support.rst | 47 ++++--- cuda_core/pyproject.toml | 2 +- .../tests/memory/test_copy_single_options.py | 8 +- cuda_core/tests/test_bindings_floor.py | 71 +++++----- cuda_core/tests/test_build_hooks.py | 25 ++-- cuda_core/tests/test_checkpoint.py | 4 +- cuda_core/tests/test_driver_table.py | 27 ++-- cuda_core/tests/test_enum_coverage.py | 2 +- cuda_core/tests/test_green_context.py | 2 +- cuda_core/tests/test_program.py | 2 +- cuda_core/tests/test_rt_layout.py | 16 +-- .../cuda_python_test_helpers/arch_check.py | 7 +- toolshed/check_cuda_core_bindings_floor.py | 27 ++-- 64 files changed, 707 insertions(+), 633 deletions(-) diff --git a/ci/tools/cuda_core_bindings_floor.py b/ci/tools/cuda_core_bindings_floor.py index 0ac887b6ea0..63fc78453d9 100644 --- a/ci/tools/cuda_core_bindings_floor.py +++ b/ci/tools/cuda_core_bindings_floor.py @@ -10,13 +10,14 @@ CI installs `cuda-bindings==` next to a freshly built cuda-core wheel to test the oldest cuda-bindings that wheel supports (BINDINGS_SOURCE=floor in -ci/tools/env-vars). The floor is read from the wheel under test rather than +ci/tools/env-vars). This script reads the floor from the wheel under test, not from the checkout, so a nightly job that tests a wheel built from another commit reads that wheel's floor. -Each build records its floor in the generated cuda/core/_build_info.py (at top -level in a single-major build, under cuda/core/cu/ in the merged wheel); -this script reads the CUDA_BINDINGS_FLOOR literal out of it without running it. +Each build records its floor in the generated cuda/core/_build_info.py. A +single-major build places the file at top level. The merged wheel places it +under cuda/core/cu/. This script parses the CUDA_BINDINGS_FLOOR literal +out of it and never runs it. """ from __future__ import annotations @@ -31,7 +32,7 @@ def _literal(source: str, name: str): - """The literal assigned to `name` at module level of `source`, parsed without executing it.""" + """The literal that `source` assigns to `name` at module level. Parses the source and never runs it.""" for node in ast.parse(source, MODULE).body: if isinstance(node, ast.AnnAssign): targets = [node.target] @@ -56,7 +57,7 @@ def floor_from_wheel(wheel: Path, major: int) -> str: for candidate in (f"cuda/core/cu{major}/{MODULE}", f"cuda/core/{MODULE}"): if candidate in names: return floor_from_source(zf.read(candidate).decode("utf-8"), major) - raise SystemExit(f"{wheel.name} contains no build for CUDA {major} (no {MODULE}); is it a cuda-core wheel?") + raise SystemExit(f"{wheel.name} contains no build for CUDA {major}: it has no {MODULE}. Is it a cuda-core wheel?") def main(argv: list[str] | None = None) -> int: diff --git a/ci/tools/env-vars b/ci/tools/env-vars index 03d074b5a30..4b71d4c9082 100755 --- a/ci/tools/env-vars +++ b/ci/tools/env-vars @@ -64,22 +64,23 @@ elif [[ "${1}" == "test" ]]; then # BINDINGS_SOURCE controls which cuda-bindings to install at test time: # main — use the just-built bindings wheel from this CI run # backport — fetch bindings from the prior (N-1) branch - # floor — install the oldest cuda-bindings the cuda-core wheel under test - # supports, from PyPI (its per-major floor; see + # floor — install from PyPI the oldest cuda-bindings that the cuda-core + # wheel under test supports, its per-major floor (see # cuda_core/cuda/core/_bindings_floor.py and ci/tools/run-tests). - # Selected when the test CTK minor differs from the one the wheel - # was built against, so those rows exercise new cuda-core + floor - # bindings + older CTK libraries, the skew cuda-core supports. - # (cuda-bindings older than the floor is unsupported and fails at - # import: https://github.com/NVIDIA/cuda-python/issues/2783.) + # Selected when the test CTK minor differs from the minor the + # wheel was built against. Those rows exercise a new cuda-core + # with the floor cuda-bindings and older CTK libraries, the skew + # that cuda-core supports. cuda-bindings older than the floor is + # unsupported and fails at import + # (https://github.com/NVIDIA/cuda-python/issues/2783). # # SKIP_CUDA_BINDINGS_TEST / SKIP_CYTHON_TEST control which *tests* to run # (they do NOT affect installation — that's BINDINGS_SOURCE's job). BUILD_CUDA_MINOR="$(cut -d '.' -f 2 <<< ${BUILD_CUDA_VER})" TEST_CUDA_MINOR="$(cut -d '.' -f 2 <<< ${CUDA_VER})" - # The prior-major half of the cuda-core wheel is built against ci/versions.yml's - # prev_build toolkit (and the backport branch's bindings). + # CI builds the prior-major half of the cuda-core wheel against the prev_build + # toolkit in ci/versions.yml and the backport branch's bindings. BUILD_PREV_CUDA_VER="$(sed -n '/prev_build:/,/version:/s/.*version: *"\([^"]*\)".*/\1/p' ci/versions.yml)" BUILD_PREV_CUDA_MINOR="$(cut -d '.' -f 2 <<< ${BUILD_PREV_CUDA_VER})" diff --git a/ci/tools/run-tests b/ci/tools/run-tests index 7a9d90955a4..c12fd8100e3 100755 --- a/ci/tools/run-tests +++ b/ci/tools/run-tests @@ -74,8 +74,8 @@ elif [[ "${test_module}" == "core" || "${test_module}" == nightly-* ]]; then # Resolve bindings based on BINDINGS_SOURCE (set by env-vars): # main/backport → local wheel from artifacts dir - # floor → the oldest cuda-bindings the core wheel under test supports, - # read from that wheel, installed from PyPI + # floor → the oldest cuda-bindings that the cuda-core wheel under test + # supports, read from that wheel and installed from PyPI BINDINGS_ARGS=() if [[ "${BINDINGS_SOURCE}" == "floor" ]]; then CORE_WHL_FOR_FLOOR=("${CUDA_CORE_ARTIFACTS_DIR}"/*.whl) diff --git a/cuda_bindings/AGENTS.md b/cuda_bindings/AGENTS.md index 33471b0bb47..f0e4d9988c4 100644 --- a/cuda_bindings/AGENTS.md +++ b/cuda_bindings/AGENTS.md @@ -51,9 +51,10 @@ the `legacy_tests` subdirectory. - `CUDA_HOME` or `CUDA_PATH` must point to a valid CUDA Toolkit for source builds. -- The toolkit's `cuda.h` must have the major.minor the generated sources came - from (`CUDA_VERSION` in `cuda/bindings/cydriver.pxd`). `build_hooks.py` - checks this before cythonize and fails with a message that names both. +- The toolkit's `cuda.h` must have the same major.minor as the generated + sources, `CUDA_VERSION` in `cuda/bindings/cydriver.pxd`. `build_hooks.py` + checks this before cythonize and fails with a message that names both + versions. - `CUDA_PYTHON_PARALLEL_LEVEL` controls build parallelism. - Runtime behavior is affected by `CUDA_PYTHON_CUDA_PER_THREAD_DEFAULT_STREAM` and diff --git a/cuda_bindings/build_hooks.py b/cuda_bindings/build_hooks.py index 61da79137e8..8660522c5e2 100644 --- a/cuda_bindings/build_hooks.py +++ b/cuda_bindings/build_hooks.py @@ -34,8 +34,8 @@ # Populated by _build_cuda_bindings(); consumed by setup.py. _extensions = None -# The generated sources declare the types and functions of one CUDA header set; -# cydriver.pxd records which. The install docs state the resulting build rule. +# The generated sources declare the types and functions of one CUDA header set, +# and cydriver.pxd records which. The install docs state the build rule that results. _CYDRIVER_PXD = Path(__file__).resolve().parent / "cuda" / "bindings" / "cydriver.pxd" _GENERATED_VERSION_RE = re.compile(r"^cdef enum:\s*CUDA_VERSION\s*=\s*(\d+)\s*$") _CUDA_H_VERSION_RE = re.compile(r"^#\s*define\s+CUDA_VERSION\s+(\d+)\s*$") @@ -108,7 +108,7 @@ def _read_version_macro(path: str, pattern: re.Pattern) -> int | None: def _read_cuda_h_version(cuda_path: str) -> int: - """The CUDA_VERSION macro (e.g. 13040 for 13.4) of the cuda.h under cuda_path.""" + """The CUDA_VERSION macro of the cuda.h under cuda_path, for example 13040 for 13.4.""" cuda_h = _cuda_h_path(cuda_path) try: version = _read_version_macro(cuda_h, _CUDA_H_VERSION_RE) @@ -123,7 +123,10 @@ def _read_cuda_h_version(cuda_path: str) -> int: def _generated_cuda_version() -> int: - """The CUDA_VERSION of the headers this source tree was generated from (cuda/bindings/cydriver.pxd).""" + """The CUDA_VERSION of the headers that this source tree was generated from. + + Read from cuda/bindings/cydriver.pxd. + """ version = _read_version_macro(str(_CYDRIVER_PXD), _GENERATED_VERSION_RE) if version is None: raise RuntimeError(f"Cannot read CUDA_VERSION from {_CYDRIVER_PXD}") @@ -136,12 +139,12 @@ def _major_minor(cuda_version: int) -> str: def _check_cuda_headers(cuda_path: str) -> None: - """Reject a toolkit whose cuda.h is not the major.minor this source tree was generated from. + """Reject a toolkit whose cuda.h is not the major.minor that this source tree was generated from. Against another minor, the C++ compile fails with a long list of - redefinition and undeclared-type errors that do not name the cause - (https://github.com/NVIDIA/cuda-python/issues/2783). Runs before - cythonize, which is the first step that touches the source tree. + redefinition and undeclared-type errors that do not name the cause. See + https://github.com/NVIDIA/cuda-python/issues/2783. Runs before cythonize, + which is the first step that touches the source tree. """ generated = _generated_cuda_version() needed, found = _major_minor(generated), _major_minor(_read_cuda_h_version(cuda_path)) diff --git a/cuda_bindings/docs/source/install.rst b/cuda_bindings/docs/source/install.rst index 796ac1a2ed0..b89feb4aa89 100644 --- a/cuda_bindings/docs/source/install.rst +++ b/cuda_bindings/docs/source/install.rst @@ -128,7 +128,7 @@ Requirements [^3]: The version is derived from git tags via ``setuptools-scm``, so the clone must include tags reaching back to at least the latest ``v*`` tag. Clone with ``git clone https://github.com/NVIDIA/cuda-python.git``; do not use ``--depth`` or ``--no-tags``, since a shallow clone builds without error but produces a bogus version such as ``0.1.dev1+g0d22cb444``. See `Cloning the repository `_ for details and recovery steps. -Source builds require that the provided CUDA headers are of the same major.minor version as the ``cuda.bindings`` you're trying to build. Despite this requirement, note that the minor version compatibility is still maintained. The build checks the header before it compiles anything; a mismatch stops it with a message that names the ``cuda.h`` it found and the version this source tree needs. Use the ``CUDA_PATH`` (or ``CUDA_HOME``) environment variable to specify the location of your headers. If both are set, ``CUDA_PATH`` takes precedence. For example, if your headers are located in ``/usr/local/cuda/include``, then you should set ``CUDA_PATH`` with: +Source builds require that the provided CUDA headers are of the same major.minor version as the ``cuda.bindings`` you're trying to build. Despite this requirement, note that the minor version compatibility is still maintained. The build checks the header before it compiles anything. A mismatch stops the build with a message that names the ``cuda.h`` it found and the version this source tree needs. Use the ``CUDA_PATH`` (or ``CUDA_HOME``) environment variable to specify the location of your headers. If both are set, ``CUDA_PATH`` takes precedence. For example, if your headers are located in ``/usr/local/cuda/include``, then you should set ``CUDA_PATH`` with: .. code-block:: console diff --git a/cuda_bindings/docs/source/release/13.5.0-notes.rst b/cuda_bindings/docs/source/release/13.5.0-notes.rst index 2f78b472b11..304ad50cf64 100644 --- a/cuda_bindings/docs/source/release/13.5.0-notes.rst +++ b/cuda_bindings/docs/source/release/13.5.0-notes.rst @@ -301,7 +301,7 @@ Enhancements - ``nvvm.add_module_to_program``, ``nvvm.lazy_add_module_to_program``, ``nvvm.get_compiled_result``, ``nvvm.get_program_log`` (``buffer``) - For all ``cuda-bindings`` APIs, functions that accept a struct wrapper will accept the struct wrapper directly, rather than requiring getting the ``.ptr`` property. - Creating class instances in ``cuda-bindings`` should now be faster in most cases, because it requires 1 heap allocation rather than 2. -- A source build now checks, before it compiles anything, that the CUDA Toolkit's ``cuda.h`` has the major.minor this source tree was generated from. A mismatch stops the build with a message that names the header found and the version needed. It used to surface as a long list of C++ redefinition errors. +- Before a source build compiles anything, it now checks that the CUDA Toolkit's ``cuda.h`` has the major.minor this source tree was generated from. A mismatch stops the build with a message that names the header found and the version needed. Previously a mismatch surfaced as a long list of C++ redefinition errors. Behavior changes ---------------- diff --git a/cuda_bindings/tests/test_build_hooks.py b/cuda_bindings/tests/test_build_hooks.py index 9b5a8923cd1..f4f4480ec25 100644 --- a/cuda_bindings/tests/test_build_hooks.py +++ b/cuda_bindings/tests/test_build_hooks.py @@ -1,15 +1,15 @@ # SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""build_hooks.py: the CUDA header check that runs before cythonize. +"""Tests for the CUDA header check that build_hooks.py runs before cythonize. -A cuda-bindings source tree is generated from one CUDA header set and compiles -only against a toolkit of that major.minor; against another minor the C++ +A cuda-bindings source tree is generated from one CUDA header set. It compiles +only against a toolkit of that major.minor. Against another minor, the C++ compile fails with redefinition errors that do not name the cause. The check reads both versions and fails early with a message that does. No GPU needed. -build_hooks.py is a PEP 517 backend, not an installed module, so it is loaded -from source. It imports setuptools at the top; the ``test`` extra provides it. +build_hooks.py is a PEP 517 backend, not an installed module, so the tests load +it from source. It imports setuptools at the top, which the ``test`` extra provides. """ import importlib.util @@ -26,7 +26,7 @@ @pytest.fixture(scope="module") def build_hooks(): if not BUILD_HOOKS.is_file(): - pytest.skip(f"{BUILD_HOOKS} is not in this tree; these tests need the source checkout") + pytest.skip(f"{BUILD_HOOKS} is not in this tree. These tests need the source checkout") spec = importlib.util.spec_from_file_location("cuda_bindings_build_hooks", BUILD_HOOKS) module = importlib.util.module_from_spec(spec) spec.loader.exec_module(module) @@ -52,7 +52,7 @@ def test_generated_header_version_is_read_from_cydriver_pxd(build_hooks): def test_a_header_of_the_generated_major_minor_passes(build_hooks, tmp_path): generated = build_hooks._generated_cuda_version() build_hooks._check_cuda_headers(_write_cuda_h(tmp_path, generated)) - # Only major.minor matters; the last digit (13041) is a toolkit patch. + # Only major.minor matters. The last digit, as in 13041, is a toolkit patch. build_hooks._check_cuda_headers(_write_cuda_h(tmp_path, generated + 1)) @@ -91,7 +91,7 @@ def test_the_build_checks_the_header_before_it_touches_the_tree(build_hooks, tmp monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: cuda_path) def not_reached(): - raise AssertionError("the header check must run before the source tree is modified") + raise AssertionError("the header check must run before the build modifies the source tree") monkeypatch.setattr(build_hooks, "_rename_architecture_specific_files", not_reached) with pytest.raises(RuntimeError, match="source tree needs CUDA .* headers"): diff --git a/cuda_core/AGENTS.md b/cuda_core/AGENTS.md index c5669f14412..ac97cdac1ab 100644 --- a/cuda_core/AGENTS.md +++ b/cuda_core/AGENTS.md @@ -31,51 +31,59 @@ This file describes `cuda_core`, the high-level Pythonic CUDA subpackage in the or CUDA headers (`CUDA_HOME`/`CUDA_PATH`) and uses it for build decisions. - Source builds require CUDA headers available through `CUDA_HOME` or `CUDA_PATH`. -- `cuda_core` requires `cuda.bindings` at or above a per-major *floor* at build - and run time, and a `cuda.h` of the same major.minor as that `cuda.bindings` - at build time (NVIDIA/cuda-python#2783). The floors are declared once, by the - `cu12`/`cu13` extras in `pyproject.toml` (`cuda-bindings[all]>=,<`); - `build_hooks.py`, the import-time check in `cuda/core/__init__.py`, the docs - and CI read them from there through `cuda/core/_bindings_floor.py`. The C++ - branches on `CUDA_CORE_BUILD_MAJOR` only; whether a feature is available at - run time depends on the driver alone, never on the `cuda.bindings` version. +- `cuda_core` requires `cuda-bindings` at or above a per-major *floor* at build + time and at run time. At build time it also requires a `cuda.h` of the same + major.minor as that `cuda-bindings` (NVIDIA/cuda-python#2783). The floors are + declared once, by the `cu12`/`cu13` extras in `pyproject.toml` + (`cuda-bindings[all]>=,<`). `build_hooks.py`, the + import-time check in `cuda/core/__init__.py`, the docs and CI read them from + there through `cuda/core/_bindings_floor.py`. The C++ branches on + `CUDA_CORE_BUILD_MAJOR` only. Whether a feature is available at run time + depends on the driver alone, never on the `cuda-bindings` version. ### Bumping the cuda-bindings floor -By policy the floor of each major is the newest `cuda-bindings` release of that -major at the time of a `cuda.core` release, so the bump is a release step, not -something each `cuda-bindings` minor triggers. It is also required by the first -change that uses a `cuda-bindings` API newer than the old floor. To bump: +By policy, the floor of each major is the newest `cuda-bindings` release of +that major at the time of a `cuda.core` release. The bump is therefore a +release step. A `cuda-bindings` minor release does not trigger it. The first +change that uses a `cuda-bindings` API newer than the old floor also requires +a bump. To bump: 1. Edit the `cuda-bindings` pin in the `cu12` or `cu13` extra in `pyproject.toml`. That is the only version to type. -2. Check `ci/versions.yml`: the `build` (current major) and `prev_build` (prior - major) toolkit pins must not sit below the floors' major.minor, or CI builds a - configuration the build rejects; a toolkit ahead of the floor is the bump - window described below. The pre-commit hook `check-cuda-core-bindings-floor` - (`toolshed/check_cuda_core_bindings_floor.py`) checks this, and that no - documentation page spells a floor out by hand. -3. Add a "Breaking Changes" entry to the release notes naming the new floors. - The support-policy table in `docs/source/support.rst` reads the floors and - the release version at docs-build time; there is nothing to edit there. -4. Refresh the pixi lock files if the pins moved past what they resolve. -5. Pin `cuda-bindings` accordingly in the conda-forge `cuda-core` feedstock - (outside this repository). +2. Check `ci/versions.yml`. The `build` pin is the current major and the + `prev_build` pin is the prior major. Neither toolkit pin may sit below its + floor's major.minor, or CI builds a configuration that the build rejects. A + toolkit ahead of the floor is the bump window described below. The + pre-commit hook `check-cuda-core-bindings-floor` + (`toolshed/check_cuda_core_bindings_floor.py`) checks the pins. It also + checks that no documentation page spells a floor out by hand. +3. Add to the release notes a "Breaking Changes" entry that names the new + floors. The support-policy table in `docs/source/support.rst` reads the + floors and the release version at docs-build time. It needs no edit. +4. If the pins moved past what the pixi lock files resolve, refresh the lock + files. +5. Pin `cuda-bindings` to match in the conda-forge `cuda-core` feedstock. The + feedstock lives outside this repository. ### CUDA Toolkit minor bumps -The build compares the toolkit's `cuda.h` with the header the installed -`cuda-bindings` was generated from (`cuda.bindings.driver.CUDA_VERSION`), not -with its version string, so the commit that moves `ci/versions.yml` to a new -minor builds `cuda.core` against the `cuda-bindings` built from the same -commit (isolated builds request `cuda-bindings` no newer than the toolkit's -minor, a cap that admits that development build). The import-time check -compares headers the same way. The floor stays -where it is until a `cuda-bindings` release of the new minor exists on PyPI; -until then the CI rows that install the literal floor (`BINDINGS_SOURCE=floor`) -fail the header check and stay red. That window is accepted; do not add -fallback logic for it. Order: bump the toolkit, release `cuda-bindings` for the -new minor, then bump the floor here. +The build compares the toolkit's `cuda.h` with the header that the installed +`cuda-bindings` was generated from (`cuda.bindings.driver.CUDA_VERSION`). It +does not compare version strings. The commit that moves `ci/versions.yml` to a +new minor therefore builds `cuda.core` against the `cuda-bindings` built from +the same commit. Isolated builds request `cuda-bindings` no newer than the +toolkit's minor, a cap that admits that development build. The import-time +check compares headers the same way. + +The floor stays where it is until a `cuda-bindings` release of the new minor +exists on PyPI. Until then, the CI rows that install the literal floor +(`BINDINGS_SOURCE=floor`) fail the header check and stay red. That window is +accepted. Do not add fallback logic for it. The order is: + +1. Bump the toolkit. +2. Release `cuda-bindings` for the new minor. +3. Bump the floor here. ## Testing expectations @@ -135,20 +143,20 @@ and agents should flag violations. objects that are not meant to be shared (e.g., the thread-local `Device`) do not need such guards (see #2321). Reference-count integrity is guaranteed; cache value-identity/idempotency is not. -- **Entry points work with or without the GIL**: the helpers in `_cpp/rt/` - are called from Cython both inside and outside `with nogil` blocks. They - never require the GIL, release it around driver calls, and never acquire it - while holding a C++ lock; the only paths that acquire it are the reporting - wrappers (`pw_*`, `report_*`) and the one-time driver function-table fill +- **Entry points work with or without the GIL**: Cython calls the helpers in + `_cpp/rt/` both inside and outside `with nogil` blocks. They never require + the GIL, release it around driver calls, and never acquire it while they + hold a C++ lock. The only paths that acquire it are the reporting wrappers + (`pw_*`, `report_*`) and the one-time driver function-table fill (`ensure_fn_table()`, see `_cpp/rt/DESIGN.md`). Driver and destructor callbacks run at arbitrary times, so they take the GIL (`with gil`) and probe for interpreter shutdown before touching Python objects. - **Driver calls go through the table**: C++ calls the driver with - `DRIVER_CALL(name, args...)`, whose pointers come from cuda-bindings' - resolved table, never from the Cython wrappers. A function the installed - driver may lack is gated in Cython on `cy_driver_version()` at the version - cuda-bindings requests it at (the number in `driver_api.hpp`); the C++ - never checks a pointer for null. + `DRIVER_CALL(name, args...)`. Its pointers come from the table that + cuda-bindings resolves, never from the Cython wrappers. If the installed + driver may lack a function, Cython gates it on `cy_driver_version()` at the + version cuda-bindings requests it at (the number in `driver_api.hpp`). The + C++ never checks a pointer for null. - **Lock ordering -- release the GIL before entering the driver**: any CUDA work reachable from a host callback or a retained object's `__del__` must release the GIL before calling the driver, to avoid GIL/driver-lock deadlocks (see the diff --git a/cuda_core/build_hooks.py b/cuda_core/build_hooks.py index ab9f14eb5f3..04b3457a768 100644 --- a/cuda_core/build_hooks.py +++ b/cuda_core/build_hooks.py @@ -67,17 +67,18 @@ def _import_get_cuda_path_or_home(): def _import_cuda_bindings(): - """Import cuda.bindings, working around PEP 517 namespace shadowing. - - Same problem and same repair as _import_get_cuda_path_or_home() (see - https://github.com/NVIDIA/cuda-python/issues/1824): in an isolated build the - project's own ``cuda/`` directory is the whole ``cuda`` namespace, so the - cuda-bindings pip installed into the build environment is not importable - until its ``cuda/`` directory is added to the namespace path. Raises - ModuleNotFoundError when no cuda-bindings is installed at all. - (importlib.metadata is no alternative: pip's in-process hook runner forwards + """Import cuda.bindings and work around PEP 517 namespace shadowing. + + The problem and the repair are the same as in _import_get_cuda_path_or_home(). + See https://github.com/NVIDIA/cuda-python/issues/1824. In an isolated build, + the project's own ``cuda/`` directory is the whole ``cuda`` namespace. The + cuda-bindings that pip installed into the build environment is not importable + until this function adds its ``cuda/`` directory to the namespace path. + Raises ModuleNotFoundError when no cuda-bindings is installed. + + importlib.metadata is no alternative: pip's in-process hook runner forwards ``find_distributions`` without the requested name, so it reports this - project's own metadata for any name.) + project's own metadata for any name. """ try: import cuda.bindings @@ -101,10 +102,10 @@ def _installed_cuda_bindings() -> tuple: """(version string, CUDA_VERSION) of the cuda-bindings in the build environment. ``cuda.bindings.driver.CUDA_VERSION`` is the ``CUDA_VERSION`` macro of the - ``cuda.h`` the installed cuda-bindings was generated from (e.g. 13040). - Unlike the version string, which a development build inherits from the - previous release's tag, it is exact, so the header rule compares against - it. Raises ModuleNotFoundError when no cuda-bindings is installed. + ``cuda.h`` that the installed cuda-bindings was generated from, for example + 13040. The header rule compares against it because it is exact. The version + string is not exact: a development build inherits it from the previous + release's tag. Raises ModuleNotFoundError when no cuda-bindings is installed. """ bindings = _import_cuda_bindings() driver = importlib.import_module("cuda.bindings.driver") @@ -123,7 +124,7 @@ def _get_cuda_path() -> str: _PACKAGE_DIR = Path(__file__).parent / "cuda" / "core" -# Generated at build time by _write_build_info(); read by cuda/core/__init__.py. +# Generated at build time by _write_build_info() and read by cuda/core/__init__.py. _BUILD_INFO_PATH = _PACKAGE_DIR / "_build_info.py" @@ -132,10 +133,11 @@ def _get_cuda_path() -> str: @functools.cache def _load_bindings_floor(): - """Load cuda/core/_bindings_floor.py, the floor logic (reading, formatting, checking). + """Load cuda/core/_bindings_floor.py, the module that reads, formats and checks the floor. - Loaded by file path: the package this backend builds is not importable - during its own build, and the module is deliberately import-free. + The load is by file path: the package that this backend builds is not + importable during its own build. The module deliberately uses the standard + library only. """ path = _PACKAGE_DIR / "_bindings_floor.py" spec = importlib.util.spec_from_file_location("_cuda_core_bindings_floor", path) @@ -148,9 +150,9 @@ def _load_bindings_floor(): def _bindings_floors() -> dict: """The cuda-bindings floor per CUDA major, from the cu extras of pyproject.toml. - The extras are the single place the floors are declared; the build, the - import-time check, the docs and CI all derive from them (see - cuda/core/_bindings_floor.py). A malformed extra fails the build here. + The extras are the single place that declares the floors. The build, the + import-time check, the docs and CI all derive from them. See + cuda/core/_bindings_floor.py. A malformed extra fails the build here. """ try: import tomllib @@ -165,13 +167,13 @@ def _bindings_floors() -> dict: def _floor_for(cuda_major) -> tuple: - """The floor of `cuda_major`, or a build error naming the supported majors.""" + """The floor of `cuda_major`, or a build error that names the supported majors.""" floors = _bindings_floors() major = int(cuda_major) if major not in floors: raise RuntimeError( - f"cuda.core does not support CUDA {major}; supported CUDA major versions: " - f"{', '.join(str(m) for m in floors)}" + f"cuda.core does not support CUDA {major}. The supported CUDA major versions are " + f"{', '.join(str(m) for m in floors)}." ) return floors[major] @@ -182,7 +184,7 @@ def _cuda_h_path(cuda_path: str) -> str: def _read_cuda_h_version(cuda_path: str) -> int: - """The CUDA_VERSION macro (e.g. 13040 for 13.4) of the cuda.h under cuda_path.""" + """The CUDA_VERSION macro of the cuda.h under cuda_path, for example 13040 for 13.4.""" cuda_h = _cuda_h_path(cuda_path) try: with open(cuda_h, encoding="utf-8") as f: @@ -194,7 +196,7 @@ def _read_cuda_h_version(cuda_path: str) -> int: pass raise RuntimeError( f"Cannot read CUDA_VERSION from {cuda_h}. " - "Ensure CUDA_PATH or CUDA_HOME points to a valid CUDA installation with include/cuda.h." + "Ensure CUDA_PATH or CUDA_HOME points to a CUDA Toolkit with include/cuda.h." ) @@ -212,8 +214,8 @@ def _determine_cuda_major_version() -> str: Since CUDA_PATH or CUDA_HOME is required for the build (to provide include directories), the cuda.h header should always be available. The override - only short-circuits this detection; _check_build_configuration() still - reads the header and rejects one whose major disagrees. + only skips this detection. _check_build_configuration() still reads the + header and rejects one whose major disagrees. """ # Explicit override, e.g. in CI. cuda_major = os.environ.get("CUDA_CORE_BUILD_MAJOR") @@ -239,19 +241,20 @@ def _determine_cuda_major_version() -> str: def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: - """Reject build configurations cuda.core does not support, then record the build. - - cuda.core supports one configuration per CUDA major series: the installed - cuda-bindings is at least the series' floor (the cu extra in - pyproject.toml) and the cuda.h it compiles against is, by major.minor, the - header that cuda-bindings was generated from (its driver.CUDA_VERSION). - The pip build requirement (get_requires_for_build_*) states both, but - conda-forge, pixi and --no-build-isolation installs bypass it, so the check - lives here, where every build path passes. - - A too-old cuda-bindings used to surface late as an ImportError at module - init or as a feature that was silently compiled out; a mismatched header as - an unclear Cython error (see https://github.com/NVIDIA/cuda-python/issues/2783). + """Reject build configurations that cuda.core does not support, then record the build. + + cuda.core supports one configuration per CUDA major series. The installed + cuda-bindings is at least the floor of the series, the cu extra in + pyproject.toml. The cuda.h that it compiles against has the major.minor of + the header that cuda-bindings was generated from, its driver.CUDA_VERSION. + The pip build requirement in get_requires_for_build_* states both, but + conda-forge, pixi and --no-build-isolation installs bypass it. The check + therefore lives here, where every build path passes. + + Without this check, a too-old cuda-bindings surfaces late as an ImportError + at module init or as a feature that is silently compiled out. A mismatched + header surfaces as an unclear Cython error. See + https://github.com/NVIDIA/cuda-python/issues/2783. """ floor = _load_bindings_floor() major = int(cuda_major) @@ -262,18 +265,18 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: bindings_version, bindings_cuda_version = _installed_cuda_bindings() except ModuleNotFoundError as exc: raise RuntimeError( - f"cuda.core requires cuda-bindings to build (install '{requirement}'). " - "Isolated builds install it automatically; other builds must provide it." + "cuda.core requires cuda-bindings to build. Isolated builds install it automatically. " + f"For other builds, install '{requirement}'." ) from exc bindings = floor.release_triple(bindings_version) if bindings is None: raise RuntimeError( f"Cannot parse the installed cuda-bindings version {bindings_version!r}. " - "A shallow git clone of cuda-bindings reports a bogus version; see CONTRIBUTING.md." + "A shallow git clone of cuda-bindings reports a bogus version. See CONTRIBUTING.md." ) if bindings[0] != major: raise RuntimeError( - f"Building cuda.core for CUDA {major}, but the installed cuda-bindings is " + f"This cuda.core build is for CUDA {major}, but the installed cuda-bindings is " f"{bindings_version}. Install '{requirement}'." ) if bindings < floor_triple: @@ -299,7 +302,7 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: def _build_define_macros(cuda_major: str) -> list: - """Preprocessor macros that carry the build decision into the C++ (see _cpp/rt/versions.hpp).""" + """Preprocessor macros that carry the build decision into the C++. See _cpp/rt/versions.hpp.""" major = int(cuda_major) return [ ("CUDA_CORE_BUILD_MAJOR", str(major)), @@ -311,9 +314,9 @@ def _write_build_info(cuda_major: int, cuda_version: int, floor: tuple, bindings """Record what this build compiled against, for the import-time check. cuda/core/__init__.py reads this module before it selects the versioned - subpackage and refuses an installed cuda-bindings older than the floor or - generated from an older header minor than the build's (see - _bindings_floor.check_installed_bindings). Like _version.py, the file is + subpackage. It refuses an installed cuda-bindings that is older than the + floor or generated from an older header minor than the build's. See + _bindings_floor.check_installed_bindings. Like _version.py, the file is generated, gitignored, and shipped. """ _BUILD_INFO_PATH.write_text( @@ -321,7 +324,7 @@ def _write_build_info(cuda_major: int, cuda_version: int, floor: tuple, bindings f"CUDA_MAJOR = {cuda_major}\n" f"CUDA_VERSION = {cuda_version} # the cuda.h this build compiled against\n" f"CUDA_BINDINGS_FLOOR = {tuple(floor)!r}\n" - f"CUDA_BINDINGS_BUILD_VERSION = {bindings_version!r} # informational; not read at import\n", + f"CUDA_BINDINGS_BUILD_VERSION = {bindings_version!r} # informational, not read at import\n", encoding="utf-8", ) @@ -510,8 +513,8 @@ def module_names(): "cuda/core/_cpp", ] + all_include_dirs, - # The C++ branches on the CUDA major series only; _cpp/rt/versions.hpp - # re-checks cuda.h against both macros (see _check_build_configuration). + # The C++ branches on the CUDA major series only. _cpp/rt/versions.hpp + # re-checks cuda.h against both macros. See _check_build_configuration. define_macros=_build_define_macros(cuda_major), language="c++", extra_compile_args=extra_compile_args, @@ -652,16 +655,19 @@ def build_wheel(wheel_directory, config_settings=None, metadata_directory=None): def _get_cuda_bindings_require(): """The cuda-bindings build requirement for isolated builds. - The floor of the CUDA major being built, capped at the header's minor when - cuda.h is readable: pip would otherwise install the newest cuda-bindings of - the major, which a newer minor on PyPI turns into a header mismatch that an - isolated build cannot fix from the outside. A cap rather than a pin, because - a cuda-bindings built from main right after a toolkit bump carries the - previous release's version string (13.4.3.devN) with the new header; the - header rule in _check_build_configuration() judges it, not pip. When the - header's minor is below the floor's, the floor alone is requested so that - the configuration check reports the mismatch in its own words. Honored by - isolated builds only; the configuration check covers every build path. + The requirement is the floor of the CUDA major that the build targets, + capped at the minor of the header when cuda.h is readable. Without the cap, + pip installs the newest cuda-bindings of the major. A newer minor on PyPI + then causes a header mismatch that an isolated build cannot fix from the + outside. The cap is not a pin: a cuda-bindings built from main right after + a toolkit bump carries the previous release's version string, such as + 13.4.3.devN, with the new header. The header rule in + _check_build_configuration() judges it, not pip. + + When the minor of the header is below the minor of the floor, the + requirement is the floor alone, so that the configuration check reports the + mismatch in its own words. Only isolated builds honor the requirement. The + configuration check covers every build path. """ floor = _load_bindings_floor() floor_triple = _floor_for(_determine_cuda_major_version()) diff --git a/cuda_core/cuda/core/__init__.py b/cuda_core/cuda/core/__init__.py index 83044801c6c..b1221419428 100644 --- a/cuda_core/cuda/core/__init__.py +++ b/cuda_core/cuda/core/__init__.py @@ -6,18 +6,18 @@ def _import_versioned_module() -> None: - """Select the build for the installed cuda-bindings, after checking it is supported. + """Check that the installed cuda-bindings is supported, then select the build for it. The published wheel carries one build per CUDA major series, as the - subpackages ``cuda.core.cu12`` and ``cuda.core.cu13``; a conda or local + subpackages ``cuda.core.cu12`` and ``cuda.core.cu13``. A conda or local build carries one build at the top level. Each build records the CUDA - header it compiled against and its cuda-bindings floor in ``_build_info`` - (generated by build_hooks.py). The installed cuda-bindings must be of the - build's major, at least as new as the floor, and generated from a - ``cuda.h`` at least as new as the build's (see - ``_bindings_floor.check_installed_bindings``), or import fails here with an - actionable message instead of later with a missing C function or a - silently disabled feature. + header that it compiled against and its cuda-bindings floor in + ``_build_info``, which build_hooks.py generates. The installed cuda-bindings + must be of the build's major, at least as new as the floor, and generated + from a ``cuda.h`` at least as new as the build's. See + ``_bindings_floor.check_installed_bindings``. If it is not, import fails + here with an actionable message instead of later with a missing C function + or a silently disabled feature. """ import importlib @@ -25,11 +25,11 @@ def _import_versioned_module() -> None: from cuda import bindings except ModuleNotFoundError as exc: if exc.name in ("cuda", "cuda.bindings"): - raise ImportError("cuda.core requires cuda-bindings; install cuda-core[cu12] or cuda-core[cu13]") from None + raise ImportError("cuda.core requires cuda-bindings. Install cuda-core[cu12] or cuda-core[cu13]") from None raise def load_build_module(name: str, cuda_major: int): - # Prefer this major's build in the merged wheel; fall back to a plain build. + # Prefer this major's build in the merged wheel, then fall back to a plain build. try: return importlib.import_module(f".cu{cuda_major}.{name}", __package__) except ModuleNotFoundError as exc: @@ -38,17 +38,20 @@ def load_build_module(name: str, cuda_major: int): return importlib.import_module(f".{name}", __package__) version_str = bindings.__version__ - # The major decides which build to consult; _bindings_floor validates everything else. + # The major decides which build to consult. _bindings_floor validates everything else. try: cuda_major = int(version_str.split(".")[0]) except ValueError: - raise ImportError(f"a cuda-bindings release must be installed (found version {version_str!r})") from None + raise ImportError( + f"cuda.core requires a cuda-bindings release, but the installed cuda-bindings version is {version_str!r}" + ) from None try: floor = load_build_module("_bindings_floor", cuda_major) info = load_build_module("_build_info", cuda_major) except ModuleNotFoundError as exc: raise ImportError( - f"this cuda.core installation has no build for CUDA {cuda_major} (installed cuda-bindings: {version_str})" + f"This cuda.core installation has no build for CUDA {cuda_major}. " + f"The installed cuda-bindings is {version_str}." ) from exc # By module object: the `cuda` namespace package need not carry a `bindings` attribute. bindings_driver = importlib.import_module("cuda.bindings.driver") diff --git a/cuda_core/cuda/core/_bindings_floor.py b/cuda_core/cuda/core/_bindings_floor.py index 048b583ab5a..c8c887490ea 100644 --- a/cuda_core/cuda/core/_bindings_floor.py +++ b/cuda_core/cuda/core/_bindings_floor.py @@ -4,30 +4,29 @@ """The cuda-bindings version floor: how it is read and how it is enforced. -cuda.core supports two CUDA major series at a time and requires, for each, a -minimum cuda-bindings version (the *floor*) at build time and at run time. The -floor is the newest cuda-bindings release of that series at the time of the -cuda.core release, normally the release cuda.core's own wheels are built -against. See https://github.com/NVIDIA/cuda-python/issues/2783 and the support -policy in the documentation. - -The floors are declared in exactly one place: the ``cu12`` and ``cu13`` extras -in ``pyproject.toml``, each of which pins ``cuda-bindings>=,<``. +cuda.core supports two CUDA major series at a time. For each series, it +requires a minimum cuda-bindings version, the *floor*, at build time and at +run time. The floor is the newest cuda-bindings release of that series at the +time of the cuda.core release. Normally that is the release that the cuda.core +wheels are built against. See https://github.com/NVIDIA/cuda-python/issues/2783 +and the support policy in the documentation. + +Exactly one place declares the floors: the ``cu12`` and ``cu13`` extras in +``pyproject.toml``. Each extra pins ``cuda-bindings>=,<``. Everything else derives from them through :func:`floors_from_extras`: -- the build backend (``build_hooks.py``) checks the installed cuda-bindings - against the floor, checks that the ``cuda.h`` it compiles against is the one - that cuda-bindings was generated from, and records the floor and the header - in the generated ``_build_info.py``; +- The build backend, ``build_hooks.py``, checks the installed cuda-bindings + against the floor. It checks that the ``cuda.h`` that it compiles against is + the one that cuda-bindings was generated from. It records the floor and the + header in the generated ``_build_info.py``. - ``cuda/core/__init__.py`` checks the installed cuda-bindings against that - record with :func:`check_installed_bindings`; -- the documentation reads the floors into substitutions (``docs/source/conf.py``); -- ``toolshed/check_cuda_core_bindings_floor.py`` (a pre-commit hook) checks - that the CI toolkit pins in ``ci/versions.yml`` sit in the floors' minors. - -This module is loaded by file path during the build and by the tools above, -so it uses the standard library only and must not import anything from -``cuda``. + record with :func:`check_installed_bindings`. +- The documentation reads the floors into substitutions in ``docs/source/conf.py``. +- ``toolshed/check_cuda_core_bindings_floor.py``, a pre-commit hook, checks + that the CI toolkit pins in ``ci/versions.yml`` sit in the minors of the floors. + +The build and the tools above load this module by file path. It therefore uses +the standard library only and must not import anything from ``cuda``. """ from __future__ import annotations @@ -48,13 +47,13 @@ _RELEASE_RE = re.compile(r"^(\d+)\.(\d+)\.(\d+)") _EXTRA_RE = re.compile(r"^cu(\d+)$") # A cuda-bindings requirement, with or without extras such as `[all]`, up to -# an optional environment marker. The name must be followed by `[`, an -# operator, a marker, whitespace or the end so that other names do not match. +# an optional environment marker. After the name, the lookahead requires `[`, an +# operator, a marker, whitespace or the end, so that other names do not match. _REQUIREMENT_RE = re.compile( r"^\s*cuda-bindings(?=[\[<>=!~;\s]|$)\s*(?:\[[^\]]*\])?\s*(?P[^;]*?)\s*(?:;.*)?$" ) _FLOOR_SPEC_RE = re.compile(r"^>=(\d+)\.(\d+)\.(\d+)$") -# The upper bound: `<14` (also `<14.0`, `<14.0.0`) or the equivalent `==13.*`. +# The upper bound: `<14`, `<14.0` or `<14.0.0`, or the equivalent `==13.*`. _UPPER_SPEC_RE = re.compile(r"^<(\d+)(?:\.0)*$") _MAJOR_SPEC_RE = re.compile(r"^==(\d+)\.\*$") @@ -65,10 +64,10 @@ def release_triple(version: str) -> tuple[int, int, int] | None: """The leading ``major.minor.patch`` of a version string, or None. - Pre-release and development suffixes are ignored, so ``13.4.1``, + The function ignores pre-release and development suffixes, so ``13.4.1``, ``13.4.1a0`` and ``13.4.2.dev249+gabcdef`` yield (13, 4, 1), (13, 4, 1) - and (13, 4, 2). A string without three leading integers (for example the - ``0.1.dev1`` that setuptools-scm reports for a shallow clone) yields None. + and (13, 4, 2). A string without three leading integers yields None, for + example the ``0.1.dev1`` that setuptools-scm reports for a shallow clone. """ m = _RELEASE_RE.match(version.strip()) if m is None: @@ -81,7 +80,7 @@ def format_version(triple: tuple[int, ...]) -> str: def cuda_version_of(triple: tuple[int, int, int]) -> int: - """The ``CUDA_VERSION`` macro value (e.g. 13040) for a version triple's major.minor.""" + """The ``CUDA_VERSION`` macro value for the major.minor of a version triple, for example 13040.""" return triple[0] * 1000 + triple[1] * 10 @@ -90,11 +89,11 @@ def floors_from_extras(extras: Mapping[str, Sequence[str]]) -> dict[int, tuple[i ``extras`` is the parsed ``[project.optional-dependencies]`` table. Each ``cu`` extra must list exactly one cuda-bindings requirement with two - specifiers, in either order: the floor, ``>=..``, and an - upper bound that excludes the next major, ``<`` (``==.*`` is also - accepted). Anything else raises ValueError naming the extra, so a typo - fails the build and the pre-commit hook instead of shifting the floor - silently. + specifiers, in either order. One is the floor, ``>=..``. + The other is an upper bound that excludes the next major, ``<`` or + ``==.*``. Anything else raises ValueError with the name of the extra, so + a typo fails the build and the pre-commit hook instead of a silent shift of + the floor. """ floors: dict[int, tuple[int, int, int]] = {} for extra, requirements in extras.items(): @@ -113,8 +112,8 @@ def floors_from_extras(extras: Mapping[str, Sequence[str]]) -> dict[int, tuple[i confined = [cm for s in specifiers if (cm := _confined_major(s)) is not None] if len(specifiers) != 2 or len(floor_matches) != 1 or len(confined) != 1: raise ValueError( - f"the {extra!r} extra must pin cuda-bindings as '>=,<' " - f"(for example '>=13.4.1,<14'), found {requirement!r}" + f"the {extra!r} extra must pin cuda-bindings as '>=,<', " + f"for example '>=13.4.1,<14', but it lists {requirement!r}" ) floor = (int(floor_matches[0].group(1)), int(floor_matches[0].group(2)), int(floor_matches[0].group(3))) if floor[0] != major or confined[0] != major: @@ -139,8 +138,8 @@ def _confined_major(specifier: str) -> int | None: def bindings_requirement(floor: tuple[int, int, int], below: tuple[int, ...] | None = None) -> str: """The pip requirement for cuda-bindings at or above ``floor`` and below ``below``, without extras. - ``below`` defaults to the next major: ``(13, 4, 1)`` gives ``cuda-bindings>=13.4.1,<14``; - with ``below=(13, 5)`` it gives ``cuda-bindings>=13.4.1,<13.5``. + ``below`` defaults to the next major: ``(13, 4, 1)`` gives ``cuda-bindings>=13.4.1,<14``. + With ``below=(13, 5)`` it gives ``cuda-bindings>=13.4.1,<13.5``. """ upper = format_version(below) if below is not None else str(floor[0] + 1) return f"cuda-bindings>={format_version(floor)},<{upper}" @@ -159,36 +158,48 @@ def check_installed_bindings( build_floor: tuple[int, int, int], core_version: str, ) -> tuple[int, int, int]: - """Validate the installed cuda-bindings against a build; return its triple. - - ``installed_version`` and ``installed_cuda_version`` are the installed - cuda-bindings' ``__version__`` and ``driver.CUDA_VERSION`` (the ``cuda.h`` - it was generated from, e.g. 13040); ``build_cuda_major``, - ``build_cuda_version`` and ``build_floor`` are the build's record in - ``_build_info.py``. Raises ImportError with an actionable message when the - installed cuda-bindings is not a release, is not of the major this build - was compiled for, is older than the floor, or was generated from an older - ``cuda.h`` minor than the build compiled against: the driver function - table the C++ layer looks up in cuda-bindings is keyed by that header's - macros, so an older minor may lack entries. Headers are compared as - ``CUDA_VERSION`` values, not version strings, because a development build - of cuda-bindings carries the previous release's version string. + """Validate the installed cuda-bindings against a build and return its triple. + + ``installed_version`` and ``installed_cuda_version`` are the ``__version__`` + and ``driver.CUDA_VERSION`` of the installed cuda-bindings. The latter is + the ``cuda.h`` that it was generated from, for example 13040. + ``build_cuda_major``, ``build_cuda_version`` and ``build_floor`` are the + build's record in ``_build_info.py``. + + Raises ImportError with an actionable message unless the installed + cuda-bindings passes all of these checks: + + - It is a release. + - It is of the major that this build was compiled for. + - It is at least the floor. + - It was generated from a ``cuda.h`` minor at least as new as the one that + the build compiled against. The driver function table that the C++ layer + looks up in cuda-bindings is keyed by the macros of that header, so an + older minor may lack entries. + + The check compares headers as ``CUDA_VERSION`` values, not version strings, + because a development build of cuda-bindings carries the previous release's + version string. """ installed = release_triple(installed_version) if installed is None: - raise ImportError(f"a cuda-bindings {build_cuda_major}.x release is required (found {installed_version})") + raise ImportError( + f"cuda.core requires a cuda-bindings {build_cuda_major}.x release, " + f"but the installed cuda-bindings version is {installed_version}" + ) major = installed[0] if major != build_cuda_major: raise ImportError( - f"this cuda.core {core_version} build is for CUDA {build_cuda_major}, but the installed " - f"cuda-bindings is {installed_version}. Install cuda-bindings {build_cuda_major}.x " - f'(pip install "cuda-bindings=={build_cuda_major}.*"), or a cuda.core build for CUDA {major} if one exists.' + f"This cuda.core {core_version} build is for CUDA {build_cuda_major}, but the installed " + f"cuda-bindings is {installed_version}. Install cuda-bindings {build_cuda_major}.x with: " + f'pip install "cuda-bindings=={build_cuda_major}.*". If a cuda.core build for CUDA {major} exists, ' + "install it instead." ) if installed < tuple(build_floor): floor = format_version(build_floor) raise ImportError( - f"cuda.core {core_version} requires cuda-bindings >= {floor} for CUDA {major} " - f'(found {installed_version}). Upgrade with: pip install -U "cuda-bindings>={floor},<{major + 1}"' + f"cuda.core {core_version} requires cuda-bindings >= {floor} for CUDA {major}, but cuda-bindings " + f'{installed_version} is installed. Upgrade with: pip install -U "cuda-bindings>={floor},<{major + 1}"' ) built_against, generated_from = header_minor(build_cuda_version), header_minor(installed_cuda_version) if generated_from < built_against: diff --git a/cuda_core/cuda/core/_cpp/rt/DESIGN.md b/cuda_core/cuda/core/_cpp/rt/DESIGN.md index 783dc53a4df..36b238a9e8a 100644 --- a/cuda_core/cuda/core/_cpp/rt/DESIGN.md +++ b/cuda_core/cuda/core/_cpp/rt/DESIGN.md @@ -168,17 +168,19 @@ avoiding the duplicate state problem. ## CUDA driver function pointers from cuda-bindings **Problem**: cuda.core cannot link against `libcuda.so` at build time, and it -must not load the driver itself: cuda-bindings owns driver loading and symbol -resolution (with `cuGetProcAddress`, which also selects the ABI variant and the -per-thread-default-stream variant). Until #2783, the C++ called cuda-bindings' -*Cython wrappers*, extracted from `cydriver.__pyx_capi__`. When the driver -lacked a function, a wrapper raised a Python exception that C++ never saw and -returned a sentinel `CUresult`; the exception surfaced later as `SystemError`. -And a wrapper could be absent when the installed cuda-bindings was older than -the build, so some pointers were optional and probed for null. - -**Solution**: `driver_api.hpp` lists every driver function the C++ calls, with -the CUDA version cuda-bindings requests it at: +must not load the driver itself. cuda-bindings owns driver loading and symbol +resolution. It resolves symbols with `cuGetProcAddress`, which also selects the +ABI variant and the per-thread-default-stream variant. + +Until #2783, the C++ called cuda-bindings' *Cython wrappers*, extracted from +`cydriver.__pyx_capi__`. When the driver lacked a function, a wrapper raised a +Python exception that C++ never saw and returned a sentinel `CUresult`. The +exception surfaced later as `SystemError`. A wrapper could also be absent when +the installed cuda-bindings was older than the build, so some pointers were +optional and probed for null. + +**Solution**: `driver_api.hpp` lists every driver function that the C++ calls, +with the CUDA version cuda-bindings requests it at: ```cpp #define CUDA_CORE_DRIVER_FUNCTIONS(X) \ @@ -188,63 +190,70 @@ the CUDA version cuda-bindings requests it at: ``` The list declares the `p_cuXxx` pointers and builds a table of -`{key, name, slot, introduced}` entries. The key is `"__"` plus the symbol -`cuda.h` maps the name to (`cuStreamDestroy` -> `"__cuStreamDestroy_v2"`), -which is how cuda-bindings names the slot in +`{key, name, slot, introduced}` entries. The key is `"__"` plus the symbol that +`cuda.h` maps the name to (`cuStreamDestroy` -> `"__cuStreamDestroy_v2"`). That +is how cuda-bindings names the slot in `cuda.bindings._internal.driver._inspect_function_pointers()`. Because the key follows the header's macros, the build requires the header's major.minor to equal cuda-bindings' (see "Build-time version guards"). -The table is filled lazily by `ensure_fn_table()` (`py_driver_fns.cpp`), the -first time a `DRIVER_CALL(name, args...)` finds its pointer null, so -`import cuda.core` never touches the driver. The fill acquires the GIL, calls +`ensure_fn_table()` (`py_driver_fns.cpp`) fills the table lazily, the first +time a `DRIVER_CALL(name, args...)` finds its pointer null. `import cuda.core` +therefore never touches the driver. The fill acquires the GIL, calls `_inspect_function_pointers()`, and copies every entry's address into its `p_` pointer under a mutex that is never held across a Python call. It then checks that every function introduced at or before the CUDA major series' -first release is present; a null one means the driver is older than the series -and the fill fails with that message. - -After the fill a pointer is either the driver's entry point or null because -the installed driver does not provide a newer function. Those functions are -gated on the driver version in Cython (`cy_driver_version()`), never by a null -check in C++. A `DRIVER_CALL` that still finds null after the fill is a gate -bug or a failed fill: it reports through `report_message()` and returns -`CUDA_ERROR_NOT_INITIALIZED` from a trampoline of the right signature, so it -never dereferences null and never throws, which makes it safe in `noexcept` -deleters. The `pw_` wrappers go through the same path. Owning handle -constructors whose deleter calls the driver but which do not call it -themselves (`create_graph_handle`, ...) call `ensure_fn_table()` so the fill -never happens in a deleter. NVRTC, NVVM and nvJitLink have one table each, -filled by `create_*_handle` after the library has loaded. - -Two rules follow. A `DRIVER_CALL` must not be made while a C++ lock is held, -because the fill acquires the GIL; resolve the table before the lock -(`ensure_fn_table(FnTable::driver)`) and use the raw pointer inside it, marked -`// raw:` (see `deviceptr_import_ipc`). And a driver function the Cython layer -gates must be gated at the version cuda-bindings requests it at (the number in -the table), not the version the driver first shipped it. - -`tests/test_rt_layout.py` checks the table against cuda-bindings' loader and -that no raw `p_` call exists outside the machinery and the marked lines. +first release is present. A null one means that the driver is older than the +series, and the fill fails with that message. + +After the fill, a pointer is either the driver's entry point or null. A null +pointer means that the installed driver does not provide that newer function. +Cython gates those functions on the driver version (`cy_driver_version()`). The +C++ never gates them with a null check. + +A `DRIVER_CALL` that still finds null after the fill is a gate bug or a failed +fill. It reports through `report_message()` and returns +`CUDA_ERROR_NOT_INITIALIZED` from a trampoline of the right signature. It never +dereferences null and never throws, so it is safe in `noexcept` deleters. The +`pw_` wrappers go through the same path. + +Some owning handle constructors do not call the driver, but their deleter does +(`create_graph_handle`, ...). Those constructors call `ensure_fn_table()` so +that the fill never happens in a deleter. NVRTC, NVVM and nvJitLink have one +table each. `create_*_handle` fills it after the library has loaded. + +Two rules follow: + +1. Do not make a `DRIVER_CALL` while the caller holds a C++ lock, because the + fill acquires the GIL. Resolve the table before the lock + (`ensure_fn_table(FnTable::driver)`) and use the raw pointer inside it, + marked `// raw:` (see `deviceptr_import_ipc`). +2. If the Cython layer gates a driver function, gate it at the version + cuda-bindings requests it at (the number in the table), not at the version + the driver first shipped it. + +`tests/test_rt_layout.py` checks the table against cuda-bindings' loader. It +also checks that no raw `p_` call exists outside the machinery and the marked +lines. ## Build-time version guards -cuda.core supports one build configuration per CUDA major series: the `cuda.h` +cuda.core supports one build configuration per CUDA major series. The `cuda.h` it compiles against has the same major.minor as the cuda-bindings it is built with, and that cuda-bindings is at or above the series' floor (`cuda/core/_bindings_floor.py`). `build_hooks.py` enforces both before -compiling and defines `CUDA_CORE_BUILD_MAJOR` and `CUDA_CORE_MIN_CUDA_VERSION` -for the C++ compiler; `versions.hpp`, the first include of the tree, re-checks -`cuda.h` against them with `#error`. +compilation and defines `CUDA_CORE_BUILD_MAJOR` and +`CUDA_CORE_MIN_CUDA_VERSION` for the C++ compiler. `versions.hpp`, the first +include of the tree, re-checks `cuda.h` against them with `#error`. The C++ branches on `CUDA_CORE_BUILD_MAJOR` only, and only where the two major series differ. Minor-version fences (`#if CUDA_VERSION >= 130x0`) are not -allowed: they compiled features out of source builds against an older header -while the run-time checks, which looked at the bindings and the driver, never -noticed (https://github.com/NVIDIA/cuda-python/issues/2783). Whether the -*driver* provides a function is decided by the driver-version gates in Cython, -never by the C++ layer. `tests/test_rt_layout.py` enforces that `versions.hpp` -is the only file under `_cpp/` that names `CUDA_VERSION`. +allowed. They compiled features out of source builds against an older header. +The run-time checks looked at cuda-bindings and the driver, so they never +noticed (https://github.com/NVIDIA/cuda-python/issues/2783). The driver-version +gates in Cython, never the C++ layer, decide whether the *driver* provides a +function. `tests/test_rt_layout.py` enforces that `versions.hpp` is the only +file under `_cpp/` that names `CUDA_VERSION`. ## Key Implementation Details @@ -274,13 +283,13 @@ Handle destructors may run from any thread. The implementation includes RAII gua - Handle Python finalization gracefully (avoid GIL operations during shutdown) - Ensure Python object manipulation happens with GIL held -The handle API functions may be called with or without the GIL held (Cython -calls most of them from `with nogil` blocks and some with the GIL held). They -never require it and never take a C++ lock while acquiring it. They release the -GIL (if necessary) before calling CUDA driver API functions. The only places +The handle API functions work with or without the GIL held. Cython calls most +of them from `with nogil` blocks and some with the GIL held. They never require +the GIL and never acquire it while they hold a C++ lock. If necessary, they +release the GIL before they call CUDA driver API functions. The only places that acquire the GIL are the reporting paths (`pw_*`, `report_*`) and the -one-time function-table fill (`ensure_fn_table()`), neither of which may run -while a C++ lock is held. +one-time function-table fill (`ensure_fn_table()`). Neither may run while a +C++ lock is held. **The GIL is the outermost lock.** Code that holds a C++ lock (a registry's mutex, `ipc_import_mutex`, any `std::mutex`) must not acquire or reacquire the @@ -467,8 +476,8 @@ The resource handle design: 2. **Encodes lifetimes structurally** via embedded handle dependencies. 3. **Uses Cython's `cimport` mechanism** to share C++ code across modules without duplicate static/thread-local state. -4. **Resolves CUDA driver symbols** through the driver entry points cuda-bindings resolves - (`_inspect_function_pointers()`), filled lazily by `ensure_fn_table()` on first use. +4. **Resolves CUDA driver symbols** through the driver entry points that cuda-bindings + resolves (`_inspect_function_pointers()`), filled lazily by `ensure_fn_table()` on first use. 5. **Provides overloaded accessors** (`as_cu`, `as_intptr`, `as_py`) since handles cannot have attributes without unnecessary Python object wrappers. diff --git a/cuda_core/cuda/core/_cpp/rt/context.cpp b/cuda_core/cuda/core/_cpp/rt/context.cpp index b0b6d162ab2..47678502989 100644 --- a/cuda_core/cuda/core/_cpp/rt/context.cpp +++ b/cuda_core/cuda/core/_cpp/rt/context.cpp @@ -227,8 +227,8 @@ ContextHandle get_primary_context(int device_id) { [device_id](const ContextBox* b) { context_registry.unregister_handle(b->resource); // During interpreter shutdown, leave primary-context cleanup to - // process teardown (an unavailable table entry would need Python - // to report itself). + // process teardown. An unavailable table entry would need Python + // to report itself. if (Py_IsInitialized() && !py_is_finalizing()) { GILReleaseGuard gil; DRIVER_CALL(cuDevicePrimaryCtxRelease, device_id); diff --git a/cuda_core/cuda/core/_cpp/rt/driver_api.cpp b/cuda_core/cuda/core/_cpp/rt/driver_api.cpp index 83b6b92530e..1d222573617 100644 --- a/cuda_core/cuda/core/_cpp/rt/driver_api.cpp +++ b/cuda_core/cuda/core/_cpp/rt/driver_api.cpp @@ -8,7 +8,7 @@ namespace cuda_core::rt { -// The pointers. Null until ensure_fn_table() fills the table; see driver_api.hpp. +// The pointers. Null until ensure_fn_table() fills the table. See driver_api.hpp. #define CUDA_CORE_DEFINE_DRIVER_FN(name, introduced) decltype(&name) p_##name = nullptr; CUDA_CORE_DRIVER_FUNCTIONS(CUDA_CORE_DEFINE_DRIVER_FN) #undef CUDA_CORE_DEFINE_DRIVER_FN @@ -22,8 +22,8 @@ namespace { #define CUDA_CORE_STR(x) #x #define CUDA_CORE_XSTR(x) CUDA_CORE_STR(x) -// "__" + the symbol cuda.h maps the public name to (macro-expanded), which is -// how cuda-bindings keys its table; #name is the public name, unexpanded. +// "__" + the macro-expanded symbol that cuda.h maps the public name to, which +// is how cuda-bindings keys its table. #name is the public name, unexpanded. #define CUDA_CORE_DRIVER_FN_ENTRY(name, introduced) \ {"__" CUDA_CORE_XSTR(name), #name, reinterpret_cast(&p_##name), introduced}, diff --git a/cuda_core/cuda/core/_cpp/rt/driver_api.hpp b/cuda_core/cuda/core/_cpp/rt/driver_api.hpp index 08f17380441..615f3cde34f 100644 --- a/cuda_core/cuda/core/_cpp/rt/driver_api.hpp +++ b/cuda_core/cuda/core/_cpp/rt/driver_api.hpp @@ -16,40 +16,42 @@ namespace cuda_core::rt { // // The C++ under _cpp/rt/ calls the CUDA driver through the p_cuXxx pointers // below. They hold the driver's own entry points, taken from the table that -// cuda-bindings builds when it loads the driver (cuGetProcAddress for each -// symbol, choosing the ABI variant and the per-thread-default-stream variant) -// and exposes as cuda.bindings._internal.driver._inspect_function_pointers(). -// cuda.core never loads the driver or resolves a symbol itself, and it never -// calls cuda-bindings' Cython wrappers from C++: those raise a Python -// exception when the driver lacks a function, which C++ cannot see -// (https://github.com/NVIDIA/cuda-python/issues/2783). +// cuda-bindings builds when it loads the driver. cuda-bindings calls +// cuGetProcAddress for each symbol and chooses the ABI variant and the +// per-thread-default-stream variant. It exposes the table as +// cuda.bindings._internal.driver._inspect_function_pointers(). +// cuda.core never loads the driver or resolves a symbol itself. It never calls +// cuda-bindings' Cython wrappers from C++: those raise a Python exception when +// the driver lacks a function, and C++ cannot see that exception. See +// https://github.com/NVIDIA/cuda-python/issues/2783. // -// The table is filled lazily, the first time a DRIVER_CALL finds its pointer -// null, so that `import cuda.core` never touches the driver. The fill acquires -// the GIL and runs Python (see py_driver_fns.cpp), so a DRIVER_CALL must not -// be made while a C++ lock is held; call ensure_fn_table() before taking the -// lock and use the raw pointer inside it (see deviceptr_import_ipc). +// The first DRIVER_CALL that finds its pointer null fills the table, so that +// `import cuda.core` never touches the driver. The fill acquires the GIL and +// runs Python (see py_driver_fns.cpp), so a DRIVER_CALL must not run while the +// thread holds a C++ lock. Call ensure_fn_table() before you take the lock. +// Inside the lock, use the raw pointer (see deviceptr_import_ipc). // // After the fill a pointer is either the driver's entry point or null because // the installed driver does not provide that function. Functions introduced -// at or before the first release of the CUDA major series being built are -// present in every driver cuda.core supports; the fill checks them and -// rejects an older driver. A newer function can legitimately be null, and the -// Cython layer gates its use on the driver version (cy_driver_version()), so -// the C++ never checks a pointer for null before a call. A DRIVER_CALL that -// still finds null after the fill is therefore a gate bug (or a failed fill); -// it reports an internal error and returns CUDA_ERROR_NOT_INITIALIZED from a -// trampoline of the right signature instead of dereferencing null. It never -// throws, so it is safe in noexcept deleters and cleanup paths. +// at or before the first release of the build's CUDA major series are present +// in every driver cuda.core supports. The fill checks them and rejects an +// older driver. A newer function can be null. The Cython layer gates its use +// on the driver version (cy_driver_version()), so the C++ never checks a +// pointer for null before a call. A DRIVER_CALL that still finds null after +// the fill is therefore a gate bug or a failed fill. It reports an internal +// error and returns CUDA_ERROR_NOT_INITIALIZED from a trampoline of the right +// signature, and it never dereferences null. It never throws, so it is safe in +// noexcept deleters and cleanup paths. // ============================================================================ // Each X(name, introduced) names a driver function cuda.core calls and the CUDA -// version cuda-bindings requests it at (cuGetProcAddress's cudaVersion; the -// ABI's introduction). tests/test_rt_layout.py checks the list against -// cuda-bindings' loader. `name` is the public name; cuda.h may map it to a -// versioned symbol (cuStreamDestroy -> cuStreamDestroy_v2), and the table key -// follows that mapping, so the header cuda.core compiles against must be the one -// cuda-bindings was generated from (enforced by build_hooks.py). +// version cuda-bindings requests it at: the cudaVersion it passes to +// cuGetProcAddress, which is the ABI's introduction. tests/test_rt_layout.py +// checks the list against cuda-bindings' loader. `name` is the public name. +// cuda.h may map it to a versioned symbol, for example +// cuStreamDestroy -> cuStreamDestroy_v2, and the table key follows that +// mapping. The header cuda.core compiles against must therefore be the one +// cuda-bindings was generated from. build_hooks.py enforces this. #define CUDA_CORE_DRIVER_FUNCTIONS(X) \ /* Error formatting */ \ X(cuGetErrorName, 6000) \ @@ -112,7 +114,7 @@ namespace cuda_core::rt { X(cuGraphChildGraphNodeGetGraph, 10000) \ /* Linker */ \ X(cuLinkDestroy, 5050) \ - /* Graphics interop; cuda-bindings requests 7000 (PTDS) or 3000 (legacy) */ \ + /* Graphics interop. cuda-bindings requests 7000 (PTDS) or 3000 (legacy) */ \ X(cuGraphicsUnmapResources, 7000) \ X(cuGraphicsUnregisterResource, 3000) \ /* Texture / surface / array (PR #467) */ \ @@ -130,9 +132,10 @@ namespace cuda_core::rt { CUDA_CORE_DRIVER_FUNCTIONS(CUDA_CORE_DECLARE_DRIVER_FN) #undef CUDA_CORE_DECLARE_DRIVER_FN -// Compiler-library entry points, one table per library so that filling one -// does not load the others (NVVM and nvJitLink are optional at run time). -// Types are spelled out where the library header is not included; they match +// Compiler-library entry points, one table per library, so that a fill of one +// table does not load the other libraries. NVVM and nvJitLink are optional at +// run time. Where this file does not include the library header, the +// declaration spells out the type. The types match // `nvvmResult nvvmDestroyProgram(nvvmProgram*)` and // `nvJitLinkResult nvJitLinkDestroy(nvJitLinkHandle*)` as int-sized enums. extern decltype(&nvrtcDestroyProgram) p_nvrtcDestroyProgram; @@ -157,29 +160,33 @@ struct FnEntry { // The entries of a table. Implemented in driver_api.cpp. const FnEntry* fn_table_entries(FnTable table, std::size_t* count) noexcept; -// Whether a table has been filled (acquire: a true result orders the slots). +// Whether a table is filled. An acquire load: a true result orders the slots. bool fn_table_ready(FnTable table) noexcept; -// Fill a table from cuda-bindings if it is not ready. Acquires the GIL (never -// call with a C++ lock held), imports cuda.bindings._internal., calls -// _inspect_function_pointers(), and stores every entry's pointer. Returns -// false, records the reason (fn_table_error) and reports it through -// report_message() when cuda-bindings cannot load the library, a key is -// missing (the installed cuda-bindings does not match the header this build -// compiled against), or a baseline driver function is null (the driver is -// older than the CUDA major series supports). A failed fill is latched: later -// calls return false at once without retrying. Preserves a pending Python -// exception and never leaves one set. Implemented in py_driver_fns.cpp. +// Fill a table from cuda-bindings if it is not ready. The fill acquires the +// GIL, imports cuda.bindings._internal., calls +// _inspect_function_pointers(), and stores every entry's pointer. Never call +// it with a C++ lock held. It returns false, records the reason for +// fn_table_error(), and reports the reason through report_message() when: +// - cuda-bindings cannot load the library +// - a key is missing, because the installed cuda-bindings does not match +// the header this build compiled against +// - a baseline driver function is null, because the driver is older than +// the CUDA major series supports +// A failed fill is latched: later calls return false at once and do not +// retry. The fill preserves a pending Python exception and never leaves one +// set. Implemented in py_driver_fns.cpp. bool ensure_fn_table(FnTable table) noexcept; -// Copy the reason the fill of `table` failed into `buffer`; false if it did -// not fail. Implemented in py_driver_fns.cpp. +// Copy the reason the fill of `table` failed into `buffer`. Return false if +// it did not fail. Implemented in py_driver_fns.cpp. bool fn_table_error(FnTable table, char* buffer, std::size_t size) noexcept; -// Record that `name` was called while unavailable. After a failed fill the -// fill's reason is attached to the error the caller raises (it was reported -// when the fill failed); a null entry in a filled table is a gate bug and is -// reported once per table. Implemented in py_driver_fns.cpp. +// Record a call to `name` while it is unavailable. After a failed fill, this +// function attaches the fill's reason to the error the caller raises. The fill +// reported that reason when it failed. A null entry in a filled table is a gate +// bug, which this function reports once per table. Implemented in +// py_driver_fns.cpp. void report_unavailable_fn(FnTable table, const char* name) noexcept; namespace detail { @@ -202,7 +209,8 @@ struct UnavailableStatus { // A function of the same signature as an unavailable entry point, so that a // call site never dereferences null. Its status flows to the caller's normal -// error handling; the cause has already been reported. +// error handling. The fill or report_unavailable_fn() already reported the +// cause. template struct Unavailable; template @@ -210,8 +218,8 @@ struct Unavailable { static R call(A...) noexcept { return UnavailableStatus::value; } }; -// The resolved pointer, filling the table on first use; the trampoline when -// the function is unavailable after the fill. Never throws. +// Fills the table on first use and returns the resolved pointer, or the +// trampoline when the function is unavailable after the fill. Never throws. template inline F fn_or_unavailable(F& slot, FnTable table, const char* name) noexcept { if (!fn_table_ready(table)) { @@ -228,10 +236,11 @@ inline F fn_or_unavailable(F& slot, FnTable table, const char* name) noexcept { } // namespace cuda_core::rt // A driver function's table entry, resolved on first use. DRIVER_CALL(name, args...) -// calls it; use these for every driver call in the C++ layer except under a C++ -// lock (see the header comment; such a call is marked `// raw:`). Each macro -// pastes its own parameter: passing `name` through another macro would let -// cuda.h's versioning macros rewrite it (cuMemFree -> cuMemFree_v2) first. +// calls it. Use these macros for every driver call in the C++ layer, except under +// a C++ lock (see the header comment). Such a call carries a `// raw:` marker. +// Each macro pastes its own parameter: if it passed `name` through another +// macro, cuda.h's versioning macros would rewrite it first, for example +// cuMemFree -> cuMemFree_v2. #define DRIVER_FN(name) \ (::cuda_core::rt::detail::fn_or_unavailable(::cuda_core::rt::p_##name, ::cuda_core::rt::FnTable::driver, #name)) #define DRIVER_CALL(name, ...) \ diff --git a/cuda_core/cuda/core/_cpp/rt/error.cpp b/cuda_core/cuda/core/_cpp/rt/error.cpp index 8ebc30945f8..0761c5eb7dc 100644 --- a/cuda_core/cuda/core/_cpp/rt/error.cpp +++ b/cuda_core/cuda/core/_cpp/rt/error.cpp @@ -38,7 +38,7 @@ void format_cuda_error(char* buffer, size_t size, const char* operation, CUresul const char* detail) noexcept { const char* error_name = nullptr; const char* error_description = nullptr; - // With the table unavailable the trampolines fail and the numeric fallback is used. + // If the table is unavailable, the trampolines fail and the numeric branch below runs. bool decoded = DRIVER_CALL(cuGetErrorName, status, &error_name) == CUDA_SUCCESS && DRIVER_CALL(cuGetErrorString, status, &error_description) == CUDA_SUCCESS; const char* outcome = detail ? detail : "failed"; diff --git a/cuda_core/cuda/core/_cpp/rt/error.hpp b/cuda_core/cuda/core/_cpp/rt/error.hpp index b3edbab744a..4c4905411c7 100644 --- a/cuda_core/cuda/core/_cpp/rt/error.hpp +++ b/cuda_core/cuda/core/_cpp/rt/error.hpp @@ -66,10 +66,10 @@ void attach_rollback_failure(const char* operation, CUresult status, const char* const char* take_last_error_detail(CUresult status) noexcept; void clear_last_error_detail() noexcept; -// Record why the driver function table is unavailable as the detail of the -// CUDA_ERROR_NOT_INITIALIZED the trampoline is about to return (see -// driver_api.hpp), so the raised CUDAError explains the failed fill instead -// of suggesting that cuInit() was not called. Implemented in error.cpp. +// Record why the driver function table is unavailable. The reason becomes the +// detail of the CUDA_ERROR_NOT_INITIALIZED that the trampoline is about to +// return (see driver_api.hpp), so the raised CUDAError explains the failed +// fill and does not suggest a missed cuInit() call. Implemented in error.cpp. void note_driver_table_failure(const char* reason) noexcept; // Tests only: make the next context restoration on this thread fail with diff --git a/cuda_core/cuda/core/_cpp/rt/graph.cpp b/cuda_core/cuda/core/_cpp/rt/graph.cpp index 26a9229424f..47ffc80b941 100644 --- a/cuda_core/cuda/core/_cpp/rt/graph.cpp +++ b/cuda_core/cuda/core/_cpp/rt/graph.cpp @@ -296,7 +296,7 @@ struct PreparedChildGraphUpdateState { }; GraphHandle create_graph_handle(CUgraph graph) { - ensure_fn_table(FnTable::driver); // the deleter calls the driver; resolve before it can run + ensure_fn_table(FnTable::driver); // the deleter calls the driver: resolve the table before it can run if (!graph) { return {}; } diff --git a/cuda_core/cuda/core/_cpp/rt/internal.hpp b/cuda_core/cuda/core/_cpp/rt/internal.hpp index 6b22acefc74..fb2a47b4c58 100644 --- a/cuda_core/cuda/core/_cpp/rt/internal.hpp +++ b/cuda_core/cuda/core/_cpp/rt/internal.hpp @@ -80,7 +80,7 @@ class WarnOnFailure { // The first argument is the resource being released; it is named in the // report so that independent failures are not collapsed by the warning // registry (see format_operation). The call goes through the function - // table like DRIVER_CALL: an unavailable entry is reported and yields an + // table like DRIVER_CALL: it reports an unavailable entry and yields an // error status, never a null dereference. template auto operator()(First&& first, Rest&&... rest) const noexcept { diff --git a/cuda_core/cuda/core/_cpp/rt/memory.cpp b/cuda_core/cuda/core/_cpp/rt/memory.cpp index 8831fc4191f..e990f7176e7 100644 --- a/cuda_core/cuda/core/_cpp/rt/memory.cpp +++ b/cuda_core/cuda/core/_cpp/rt/memory.cpp @@ -404,8 +404,9 @@ DevicePtrHandle deviceptr_import_ipc(const MemoryPoolHandle& h_pool, const void* auto data = const_cast( reinterpret_cast(export_data)); - // Resolve the table before any lock is taken: a fill acquires the GIL, and - // nothing under ipc_import_mutex may (#2840). The raw p_ calls below rely on it. + // Resolve the table before you take any lock: a fill acquires the GIL, and + // nothing under ipc_import_mutex may acquire it (#2840). The raw p_ calls + // below rely on this. if (!ensure_fn_table(FnTable::driver)) { report_unavailable_fn(FnTable::driver, "cuMemPoolImportPointer"); err = CUDA_ERROR_NOT_INITIALIZED; diff --git a/cuda_core/cuda/core/_cpp/rt/program.cpp b/cuda_core/cuda/core/_cpp/rt/program.cpp index 2a41f8fb125..1f32c07bf71 100644 --- a/cuda_core/cuda/core/_cpp/rt/program.cpp +++ b/cuda_core/cuda/core/_cpp/rt/program.cpp @@ -127,7 +127,7 @@ struct NvrtcProgramBox { NvrtcProgramHandle create_nvrtc_program_handle(nvrtcProgram prog) { // Resolve the table now, while the library that created `prog` is loaded, - // so the deleter never has to. + // so that the deleter never triggers a fill. ensure_fn_table(FnTable::nvrtc); auto box = std::shared_ptr( new NvrtcProgramBox{prog}, diff --git a/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp b/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp index d0c2b548d1e..0b3139bddf8 100644 --- a/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp +++ b/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp @@ -2,29 +2,29 @@ // // SPDX-License-Identifier: Apache-2.0 -// Filling the driver and compiler-library function tables from cuda-bindings. +// This file fills the driver and compiler-library function tables from cuda-bindings. // -// cuda-bindings loads each library and resolves its symbols once (for the -// driver, with cuGetProcAddress). cuda.bindings._internal. +// cuda-bindings loads each library and resolves its symbols once. For the +// driver it uses cuGetProcAddress. cuda.bindings._internal. // ._inspect_function_pointers() returns that table as {name: address}, where a // zero address means the library does not provide the symbol. This file copies // the entries cuda.core uses into the p_ pointers declared in driver_api.hpp. // -// The fill runs Python, so it acquires the GIL and must never run under a C++ -// lock (the GIL is the outermost lock; see DESIGN.md). The slot stores happen -// under fill_mutex with no Python call inside, and the ready flag is published -// with release semantics after them, so readers that see the flag see the -// pointers. Two threads may both compute the table; they store identical +// The fill runs Python, so it acquires the GIL. It must never run under a C++ +// lock, because the GIL is the outermost lock (see DESIGN.md). The slot stores +// happen under fill_mutex with no Python call inside. The fill then publishes +// the ready flag with release semantics, so a reader that sees the flag sees +// the pointers. Two threads may both compute the table. They store identical // values, one after the other. // -// Failures never propagate as exceptions and never leave a Python error set: -// they are recorded, reported through report_message(), latched (a failed -// table is not retried, so no call re-imports or re-warns), and every -// affected DRIVER_CALL then returns an error status from a trampoline -// (driver_api.hpp) with the reason attached to the raised error as a note. -// A fill can run while a Python exception is propagating (a deleter making its -// first driver call during unwinding), so the pending exception is saved -// around the Python calls and restored afterwards. +// A failure never propagates as an exception and never leaves a Python error +// set. The fill records it, reports it through report_message(), and latches +// it: no later call retries the table, re-imports, or re-warns. Every +// affected DRIVER_CALL then returns an error status from a trampoline in +// driver_api.hpp, and the raised error carries the reason as a note. +// A fill can run while a Python exception propagates, for example when a +// deleter makes its first driver call during unwinding. The fill saves the +// pending exception before its Python calls and restores it afterwards. #include "py.hpp" #include "driver_api.hpp" @@ -49,8 +49,8 @@ std::mutex fill_mutex; char fill_error[kTables][512] = {}; // guarded by fill_mutex // Saves the pending Python exception on construction and restores it on -// destruction, so the Python calls in between start from a clean error state -// and the caller's exception survives. Requires the GIL. +// destruction. The Python calls in between start from a clean error state, and +// the caller's exception survives. Requires the GIL. class PendingExceptionGuard { public: PendingExceptionGuard() noexcept { @@ -103,11 +103,11 @@ const char* library_name(FnTable table) noexcept { return "library"; } -// Copy the pending Python exception's text into buf and clear it. Returns +// Copies the pending Python exception's text into buf and clears it. Returns // true when the exception says nothing about cuda-bindings or the driver: an -// interruption (KeyboardInterrupt, SystemExit) or exhaustion (MemoryError, -// RecursionError). Such a failure is reported but not latched; the next call -// tries again. +// interruption such as KeyboardInterrupt or SystemExit, or exhaustion such as +// MemoryError or RecursionError. The fill reports such a failure but does not +// latch it, so the next call tries again. bool take_python_error(char* buf, std::size_t size) noexcept { const bool transient = PyErr_Occurred() && (!PyErr_ExceptionMatches(PyExc_Exception) || PyErr_ExceptionMatches(PyExc_MemoryError) @@ -164,7 +164,7 @@ bool ensure_fn_table(FnTable table) noexcept { return true; } if (table_failed[idx].load(std::memory_order_acquire)) { - return false; // latched: the reason was reported when the fill failed + return false; // latched: the fill reported its reason when it failed } std::size_t count = 0; const FnEntry* entries = fn_table_entries(table, &count); @@ -173,7 +173,7 @@ bool ensure_fn_table(FnTable table) noexcept { return false; } if (!Py_IsInitialized() || py_is_finalizing()) { - record_failure(table, "cuda.core cannot resolve driver functions while the interpreter is shutting down"); + record_failure(table, "cuda.core cannot resolve driver functions during interpreter shutdown"); return false; } @@ -183,7 +183,7 @@ bool ensure_fn_table(FnTable table) noexcept { GILAcquireGuard gil; if (!gil.acquired()) { - record_failure(table, "cuda.core cannot resolve driver functions while the interpreter is shutting down"); + record_failure(table, "cuda.core cannot resolve driver functions during interpreter shutdown"); return false; } PendingExceptionGuard pending; @@ -217,8 +217,8 @@ bool ensure_fn_table(FnTable table) noexcept { if (item == nullptr) { Py_DECREF(pointers); std::snprintf(message, sizeof(message), - "the installed cuda-bindings has no entry for %s (%s). cuda.core was compiled against a " - "cuda.h that names this symbol differently than the cuda-bindings in use; install the " + "the installed cuda-bindings has no entry for %s (%s). This cuda.core build used a " + "cuda.h that names this symbol differently from the installed cuda-bindings. Install the " "cuda-bindings this cuda.core requires", entries[i].name, entries[i].key); record_failure(table, message); @@ -239,15 +239,15 @@ bool ensure_fn_table(FnTable table) noexcept { Py_DECREF(pointers); // Every function introduced at or before the first release of the CUDA - // major series is present in every driver cuda.core supports; a null one - // means the driver is older than that. Newer functions may be null and are - // gated on the driver version in Cython. + // major series is present in every driver cuda.core supports. A null one + // means the driver is older than that. A newer function may be null, and + // the Cython layer gates its use on the driver version. if (table == FnTable::driver) { for (std::size_t i = 0; i < count; ++i) { if (values[i] == nullptr && entries[i].introduced <= CUDA_CORE_BUILD_MAJOR * 1000) { std::snprintf(message, sizeof(message), - "the installed CUDA driver lacks %s, which every CUDA %d driver provides " - "(introduced in CUDA %d.%d); this cuda.core build needs a newer driver", + "the installed CUDA driver lacks %s, which every CUDA %d driver provides. " + "CUDA %d.%d introduced this function. This cuda.core build needs a newer driver", entries[i].name, CUDA_CORE_BUILD_MAJOR, entries[i].introduced / 1000, entries[i].introduced / 10 % 100); record_failure(table, message); @@ -277,19 +277,19 @@ void report_unavailable_fn(FnTable table, const char* name) noexcept { std::snprintf(message, sizeof(message), "cuda.core could not call %s: %s", name, reason); } else { std::snprintf(message, sizeof(message), - "internal cuda.core error, please report: %s was called but the installed %s does not " - "provide it; a feature gate is missing or wrong. The call returned an error instead.", + "internal cuda.core error, please report: cuda.core called %s but the installed %s does not " + "provide it. A feature gate is missing or wrong. The call returned an error instead.", name, library_name(table)); } if (table == FnTable::driver) { - // The trampoline returns CUDA_ERROR_NOT_INITIALIZED; the Cython error - // path attaches this as a note to the CUDAError it raises for it. + // The trampoline returns CUDA_ERROR_NOT_INITIALIZED. The Cython error + // path attaches this message as a note to the CUDAError it raises. note_driver_table_failure(message); } if (failed_fill) { - return; // the fill reported its reason when it failed; the note carries it to each raised error + return; // the fill reported its reason when it failed, and the note carries it to each raised error } - // A gate bug: warn once per table, the first unavailable call is the informative one. + // A gate bug: warn once per table. The first unavailable call is the informative one. if (unavailable_reported[index_of(table)].exchange(true)) { return; } diff --git a/cuda_core/cuda/core/_cpp/rt/stream.cpp b/cuda_core/cuda/core/_cpp/rt/stream.cpp index 1cff809e50a..0497c0fabb5 100644 --- a/cuda_core/cuda/core/_cpp/rt/stream.cpp +++ b/cuda_core/cuda/core/_cpp/rt/stream.cpp @@ -70,7 +70,7 @@ StreamHandle create_stream_handle(const ContextHandle& h_ctx, unsigned int flags CUstream stream = nullptr; GreenCtxHandle h_green = get_context_green_ctx(h_ctx); if (h_green) { - // Gated in Cython on driver >= 12.5 (cuGreenCtxStreamCreate's introduction). + // Cython gates this on driver >= 12.5, which introduced cuGreenCtxStreamCreate. err = DRIVER_CALL(cuGreenCtxStreamCreate, &stream, as_cu(h_green), flags, priority); } else { err = invoke_in_context_or_undo( diff --git a/cuda_core/cuda/core/_cpp/rt/texture.cpp b/cuda_core/cuda/core/_cpp/rt/texture.cpp index b3f34e08cd5..309e34aa4c9 100644 --- a/cuda_core/cuda/core/_cpp/rt/texture.cpp +++ b/cuda_core/cuda/core/_cpp/rt/texture.cpp @@ -27,7 +27,7 @@ struct GraphicsResourceBox { } // namespace GraphicsResourceHandle create_graphics_resource_handle(CUgraphicsResource resource) { - ensure_fn_table(FnTable::driver); // the deleter calls the driver; resolve before it can run + ensure_fn_table(FnTable::driver); // the deleter calls the driver: resolve the table before it can run auto box = std::shared_ptr( new GraphicsResourceBox{resource}, [](const GraphicsResourceBox* b) { @@ -131,7 +131,7 @@ OpaqueArrayHandle create_array_handle_ref(CUarray arr) { } OpaqueArrayHandle create_array_handle_owning(CUarray arr) { - ensure_fn_table(FnTable::driver); // the deleter calls the driver; resolve before it can run + ensure_fn_table(FnTable::driver); // the deleter calls the driver: resolve the table before it can run if (!arr) { return {}; } diff --git a/cuda_core/cuda/core/_cpp/rt/versions.hpp b/cuda_core/cuda/core/_cpp/rt/versions.hpp index eda836cb507..5f37c0033ad 100644 --- a/cuda_core/cuda/core/_cpp/rt/versions.hpp +++ b/cuda_core/cuda/core/_cpp/rt/versions.hpp @@ -6,30 +6,30 @@ // The one place the C++ under _cpp/ consults CUDA_VERSION. // -// cuda.core supports one build configuration per CUDA major series: the -// cuda.h it compiles against has the same major.minor as the cuda-bindings it -// is built with, and that cuda-bindings is at or above the series' floor -// (cuda/core/_bindings_floor.py; https://github.com/NVIDIA/cuda-python/issues/2783). -// build_hooks.py enforces both before compiling and passes the decision down -// as two macros: +// cuda.core supports one build configuration per CUDA major series. The +// cuda.h it compiles against has the same major.minor as the cuda-bindings +// present at build time, and that cuda-bindings is at or above the series' +// floor (see cuda/core/_bindings_floor.py and +// https://github.com/NVIDIA/cuda-python/issues/2783). build_hooks.py enforces +// both before it compiles and passes the decision down as two macros: // -// CUDA_CORE_BUILD_MAJOR the CUDA major series being built (12 or 13); -// the only version the C++ may branch on, as +// CUDA_CORE_BUILD_MAJOR the CUDA major series of the build, 12 or 13. +// The only version the C++ may branch on, as // `#if CUDA_CORE_BUILD_MAJOR >= 13`, and always // for a difference between major series. // CUDA_CORE_MIN_CUDA_VERSION the floor's major.minor as a CUDA_VERSION -// value (e.g. 13040). +// value, for example 13040. // -// This header re-checks the header against both macros so that a build that +// This file checks cuda.h against both macros again, so that a build that // bypasses build_hooks.py still cannot compile against an unsupported header. -// Minor-version fences (`#if CUDA_VERSION >= 130x0`) are not allowed anywhere -// else: they compiled features out of source builds against an older header -// while the run-time checks, which looked at the bindings and the driver, -// never noticed. tests/test_rt_layout.py enforces that this is the only file -// that names CUDA_VERSION. +// No other file may use a minor-version fence such as +// `#if CUDA_VERSION >= 130x0`. Such fences compiled features out of source +// builds against an older header, and the run-time checks, which looked at +// cuda-bindings and the driver, never noticed. tests/test_rt_layout.py +// enforces that this is the only file that names CUDA_VERSION. // // Downstream Cython code that cimports _rt includes this header without the -// macros; it then only learns the major from cuda.h and skips the floor check. +// macros. It then learns only the major from cuda.h and skips the floor check. #include @@ -38,11 +38,11 @@ #endif #if (CUDA_VERSION / 1000) != CUDA_CORE_BUILD_MAJOR -#error "cuda.h does not belong to the CUDA major series cuda.core is being built for (CUDA_CORE_BUILD_MAJOR)" +#error "cuda.h does not belong to the CUDA major series this cuda.core build targets (CUDA_CORE_BUILD_MAJOR)" #endif #ifdef CUDA_CORE_MIN_CUDA_VERSION #if CUDA_VERSION < CUDA_CORE_MIN_CUDA_VERSION -#error "cuda.h is older than the minimum this cuda.core release supports for its CUDA major series (see the cuda.core support policy)" +#error "cuda.h is older than the floor of this cuda.core release for its CUDA major series (see the cuda.core support policy)" #endif #endif diff --git a/cuda_core/cuda/core/_device.pyx b/cuda_core/cuda/core/_device.pyx index f7901081c98..8611cd3d819 100644 --- a/cuda_core/cuda/core/_device.pyx +++ b/cuda_core/cuda/core/_device.pyx @@ -1641,7 +1641,7 @@ cdef inline int Device_ensure_cuda_initialized() except? -1: HANDLE_RETURN(cydriver.cuInit(0)) _is_cuInit = True IF CUDA_CORE_BUILD_MAJOR >= 13: - # Added in cuda-bindings 13.3; absent from the 12.x line. + # cuda-bindings 13.3 added this function. The 12.x line does not have it. from cuda.bindings.utils import warn_if_cuda_major_version_mismatch warn_if_cuda_major_version_mismatch() return 0 diff --git a/cuda_core/cuda/core/_device_resources.pyx b/cuda_core/cuda/core/_device_resources.pyx index 0f6e94bf6b3..dcd6aa2014b 100644 --- a/cuda_core/cuda/core/_device_resources.pyx +++ b/cuda_core/cuda/core/_device_resources.pyx @@ -212,9 +212,9 @@ cdef inline unsigned int _to_sm_count(object value) except? 0: cdef int _structured_split_checked = 0 cdef inline bint _can_use_structured_sm_split(): - """Whether the driver provides cuDevSmResourceSplit (13.1+). Cached. + """Whether the driver provides cuDevSmResourceSplit, a 13.1 driver API. Cached. - cuda-bindings 13.4+ (the floor) always exports it; only the driver can lack it.""" + Every cuda-bindings at or above the 13.4 floor exports it. Only the driver can lack it.""" global _structured_split_checked if _structured_split_checked != 0: return _structured_split_checked == 1 diff --git a/cuda_core/cuda/core/_linker.pyx b/cuda_core/cuda/core/_linker.pyx index f30a3fc30a7..e5deeb07b10 100644 --- a/cuda_core/cuda/core/_linker.pyx +++ b/cuda_core/cuda/core/_linker.pyx @@ -790,8 +790,8 @@ def _decide_nvjitlink_or_driver() -> bool: " For best results, consider upgrading to a recent version of" ) - # cuda.bindings.nvjitlink is present in every cuda-bindings cuda.core accepts; - # only the nvJitLink library itself can be missing or too old. + # Every cuda-bindings that cuda.core accepts provides cuda.bindings.nvjitlink. + # Only the nvJitLink library itself can be missing or too old. from cuda.bindings._internal import nvjitlink try: diff --git a/cuda_core/cuda/core/_memory/_buffer.pyx b/cuda_core/cuda/core/_memory/_buffer.pyx index 903a7ab0264..b1ee356c245 100644 --- a/cuda_core/cuda/core/_memory/_buffer.pyx +++ b/cuda_core/cuda/core/_memory/_buffer.pyx @@ -225,7 +225,7 @@ cdef void _dispatch_buffer_copy( else: _reject_unsupported_during_api_call( options.src_access_order, - "the CUDA 13 build of cuda.core and a driver reporting CUDA 13.2 or newer " + "the CUDA 13 build of cuda.core and a driver that reports CUDA 13.2 or newer " "(cuMemcpyWithAttributesAsync is unavailable here)", ) # STREAM and ANY never require access sooner than stream order, so diff --git a/cuda_core/cuda/core/_memory/_copy_attributes.pxd b/cuda_core/cuda/core/_memory/_copy_attributes.pxd index e85d1b71a99..aab29cdf801 100644 --- a/cuda_core/cuda/core/_memory/_copy_attributes.pxd +++ b/cuda_core/cuda/core/_memory/_copy_attributes.pxd @@ -12,8 +12,8 @@ from cuda.core._utils.version cimport cy_driver_version # no-cython-lint IF CUDA_CORE_BUILD_MAJOR >= 13: cdef inline bint _with_attributes_available(): - # cuMemcpyWithAttributesAsync is a 13.2 driver API; cuda-bindings 13.4+ - # (the floor) always exports it, so only the driver can lack it. + # cuMemcpyWithAttributesAsync is a 13.2 driver API. Every cuda-bindings at + # or above the 13.4 floor exports it, so only the driver can lack it. return cy_driver_version() >= (13, 2, 0) ELSE: cdef inline bint _with_attributes_available(): diff --git a/cuda_core/cuda/core/_memory/_copy_enums.py b/cuda_core/cuda/core/_memory/_copy_enums.py index 8b53708ee26..880d21e2d93 100644 --- a/cuda_core/cuda/core/_memory/_copy_enums.py +++ b/cuda_core/cuda/core/_memory/_copy_enums.py @@ -123,8 +123,8 @@ def _to_driver_flags(self) -> int: return _OVERLAP_MODE_TO_DRIVER[MemcpyOverlapMode(self.overlap_mode)] -# CUmemcpySrcAccessOrder and CUmemcpyFlags were added in CUDA 12.8; every -# cuda-bindings cuda.core accepts has them. Keyed by ``str``: under +# CUDA 12.8 added CUmemcpySrcAccessOrder and CUmemcpyFlags. Every +# cuda-bindings that cuda.core accepts has them. Keyed by ``str``: under # ``python_version = "3.10"`` mypy resolves StrEnum to the unstubbed backports # shim and so infers the members as plain ``str``. StrEnum members are ``str`` # instances, so this holds on every version. The values are wrapped in diff --git a/cuda_core/cuda/core/_memory/_copy_ops.pyi b/cuda_core/cuda/core/_memory/_copy_ops.pyi index cb30bedb294..3cb871c5f70 100644 --- a/cuda_core/cuda/core/_memory/_copy_ops.pyi +++ b/cuda_core/cuda/core/_memory/_copy_ops.pyi @@ -69,7 +69,7 @@ def copy_batch(stream: Stream, srcs: Sequence[Buffer], dsts: Sequence[Buffer], * Notes ----- Batching through ``cuMemcpyBatchAsync`` requires both the CUDA 13 build of - ``cuda.core`` and a driver reporting CUDA 13.0 or newer + ``cuda.core`` and a driver that reports CUDA 13.0 or newer (``cuDriverGetVersion() >= 13000``). ``cuda.bindings`` binds only the CUDA 13.0 revision of the entry point, so a driver that predates it is refused even where it implements the earlier CUDA 12.8 signature. diff --git a/cuda_core/cuda/core/_memory/_copy_ops.pyx b/cuda_core/cuda/core/_memory/_copy_ops.pyx index 3143b4cc158..73824e6c1f4 100644 --- a/cuda_core/cuda/core/_memory/_copy_ops.pyx +++ b/cuda_core/cuda/core/_memory/_copy_ops.pyx @@ -144,7 +144,7 @@ def copy_batch( Notes ----- Batching through ``cuMemcpyBatchAsync`` requires both the CUDA 13 build of - ``cuda.core`` and a driver reporting CUDA 13.0 or newer + ``cuda.core`` and a driver that reports CUDA 13.0 or newer (``cuDriverGetVersion() >= 13000``). ``cuda.bindings`` binds only the CUDA 13.0 revision of the entry point, so a driver that predates it is refused even where it implements the earlier CUDA 12.8 signature. @@ -240,7 +240,7 @@ cdef void _reject_during_api_call_fallback(tuple attr_tuple): for i in range(len(attr_tuple)): _reject_unsupported_during_api_call( (attr_tuple[i]).src_access_order, - "the CUDA 13 build of cuda.core and a driver reporting CUDA 13.0 or newer " + "the CUDA 13 build of cuda.core and a driver that reports CUDA 13.0 or newer " "(cuMemcpyBatchAsync is unavailable here)", index=i, ) diff --git a/cuda_core/cuda/core/_memory/_managed_buffer.py b/cuda_core/cuda/core/_memory/_managed_buffer.py index 4353fb7975d..42acc5f2315 100644 --- a/cuda_core/cuda/core/_memory/_managed_buffer.py +++ b/cuda_core/cuda/core/_memory/_managed_buffer.py @@ -215,13 +215,13 @@ def preferred_location(self) -> Device | Host | None: as ``Host()``. """ # The v2 path uses CU_MEM_RANGE_ATTRIBUTE_PREFERRED_LOCATION_{TYPE,ID}, - # both added in CUDA 13: it exists in the CUDA 13 build only, and the - # runtime driver must be 13.0+ too; otherwise fall back to the legacy + # both added in CUDA 13. The path exists in the CUDA 13 build only, and + # the runtime driver must be 13.0+. Otherwise fall back to the legacy # device-ordinal path. See PR #2054 / #2064 for prior regressions. if BUILD_CUDA_MAJOR >= 13 and driver_version() >= (13, 0, 0): return _read_preferred_location_v2(self) - # CUDA 12 legacy path (no NUMA info available; also taken by a CUDA 13 - # build when the runtime driver is still 12.x). + # CUDA 12 legacy path. No NUMA info is available. A CUDA 13 build also + # takes this path when the runtime driver is still 12.x. loc_id = _get_int_attr(self, _ATTR_PREFERRED) if loc_id == -2: return None diff --git a/cuda_core/cuda/core/_memory/_managed_location.py b/cuda_core/cuda/core/_memory/_managed_location.py index e77f242de7f..fc89ac162a3 100644 --- a/cuda_core/cuda/core/_memory/_managed_location.py +++ b/cuda_core/cuda/core/_memory/_managed_location.py @@ -39,9 +39,9 @@ def _reject_numa_host_on_cuda12(spec: _LocSpec) -> None: ``TypeError`` at the call boundary with actionable wording. """ # The host-NUMA kinds map to CU_MEM_LOCATION_TYPE_HOST_NUMA{,_CURRENT}, - # both added in CUDA 13: the CUDA 13 build passes them to the v2 driver - # entry points, which need a 13.0+ runtime driver as well (a build check - # alone is insufficient; PR #2054 / #2064 precedent). + # both added in CUDA 13. The CUDA 13 build passes them to the v2 driver + # entry points, which need a 13.0+ runtime driver as well. A build check + # alone is insufficient. See PR #2054 / #2064 for the precedent. if BUILD_CUDA_MAJOR >= 13 and driver_version() >= (13, 0, 0): return if spec.kind in ("host_numa", "host_numa_current"): diff --git a/cuda_core/cuda/core/_memory/_virtual_memory_resource.py b/cuda_core/cuda/core/_memory/_virtual_memory_resource.py index bdc57595f53..e643b0f3050 100644 --- a/cuda_core/cuda/core/_memory/_virtual_memory_resource.py +++ b/cuda_core/cuda/core/_memory/_virtual_memory_resource.py @@ -112,7 +112,7 @@ class VirtualMemoryResourceOptions: VirtualMemoryLocationType.HOST_NUMA_CURRENT: _l.CU_MEM_LOCATION_TYPE_HOST_NUMA_CURRENT, } _t = driver.CUmemAllocationType - # CUDA 13 added MANAGED to CUmemAllocationType; the CUDA 12 build has no such member. + # CUDA 13 added MANAGED to CUmemAllocationType. The CUDA 12 build has no such member. _allocation_type = {VirtualMemoryAllocationType.PINNED: _t.CU_MEM_ALLOCATION_TYPE_PINNED} # noqa: RUF012 if BUILD_CUDA_MAJOR >= 13: _allocation_type[VirtualMemoryAllocationType.MANAGED] = _t.CU_MEM_ALLOCATION_TYPE_MANAGED diff --git a/cuda_core/cuda/core/_program.pyx b/cuda_core/cuda/core/_program.pyx index f28f63ff067..c39b03d0cd7 100644 --- a/cuda_core/cuda/core/_program.pyx +++ b/cuda_core/cuda/core/_program.pyx @@ -758,8 +758,8 @@ def _get_nvvm_module() -> object: raise RuntimeError("NVVM module is not available (previous import attempt failed)") try: - # cuda.bindings.nvvm is present in every cuda-bindings cuda.core accepts; - # the probe checks that libnvvm itself can be loaded. + # Every cuda-bindings that cuda.core accepts provides cuda.bindings.nvvm. + # The probe checks that libnvvm itself loads. nvvm = _optional_cuda_import( "cuda.bindings.nvvm", probe_function=lambda module: module.version(), # probe triggers libnvvm load diff --git a/cuda_core/cuda/core/_stream.pyx b/cuda_core/cuda/core/_stream.pyx index 2ed1e402be4..6980981581e 100644 --- a/cuda_core/cuda/core/_stream.pyx +++ b/cuda_core/cuda/core/_stream.pyx @@ -173,8 +173,8 @@ cdef class Stream: # C++ creates the stream and returns owning handle with context dependency. # For green contexts, the C++ layer auto-dispatches to cuGreenCtxStreamCreate, - # a 12.5 driver API (cuGreenCtxCreate itself is 12.4); the driver alone - # decides availability, and the gate lives here, not in C++. + # a 12.5 driver API. cuGreenCtxCreate itself is 12.4. The driver alone + # decides availability. The gate lives here, not in C++. if context.is_green and cy_driver_version() < (12, 5, 0): raise RuntimeError( "Green context stream creation requires CUDA driver 12.5 or newer " diff --git a/cuda_core/cuda/core/_utils/enum_explanations_helpers.py b/cuda_core/cuda/core/_utils/enum_explanations_helpers.py index e5e46145a83..5adf21b7622 100644 --- a/cuda_core/cuda/core/_utils/enum_explanations_helpers.py +++ b/cuda_core/cuda/core/_utils/enum_explanations_helpers.py @@ -4,13 +4,13 @@ """Internal support for error-enum explanations. Driver and runtime error enums in ``cuda-bindings`` carry per-member -``__doc__`` text (since 12.9.6 in the 12.x line and 13.2.0 in the 13.x line; -every ``cuda-bindings`` that ``cuda.core`` accepts has it). This module -normalizes those generated docstrings so user-facing ``CUDAError`` messages -stay presentable. +``__doc__`` text. ``cuda-bindings`` added the text in 12.9.6 on the 12.x line +and in 13.2.0 on the 13.x line. Every ``cuda-bindings`` that ``cuda.core`` +accepts has it. This module normalizes those generated docstrings so that +user-facing ``CUDAError`` messages stay presentable. The cleanup rules here were derived while validating generated enum docstrings -in PR #1805. Keep them narrow and remove them when the codegen quirks are gone. +in PR #1805. Keep them narrow. When the codegen quirks are gone, remove them. """ from __future__ import annotations diff --git a/cuda_core/cuda/core/_utils/version.pyx b/cuda_core/cuda/core/_utils/version.pyx index ae396af940d..b92fc5952e3 100644 --- a/cuda_core/cuda/core/_utils/version.pyx +++ b/cuda_core/cuda/core/_utils/version.pyx @@ -8,11 +8,12 @@ import re from cuda.core._utils.cuda_utils import driver, handle_return -# The CUDA major series this build of cuda.core targets (12 or 13), from the -# compile-time environment build_hooks.py sets. The installed cuda-bindings has -# the same major (cuda/core/__init__.py enforces it at import). Python modules -# that must branch on the series, where `IF CUDA_CORE_BUILD_MAJOR` is not -# available, read this instead of comparing binding_version(). +# The CUDA major series that this build of cuda.core targets, 12 or 13. The +# value comes from the compile-time environment that build_hooks.py sets. The +# installed cuda-bindings has the same major, which cuda/core/__init__.py +# enforces at import. Python modules that must branch on the series, where +# `IF CUDA_CORE_BUILD_MAJOR` is not available, read this constant rather than +# compare binding_version(). BUILD_CUDA_MAJOR: int = CUDA_CORE_BUILD_MAJOR diff --git a/cuda_core/cuda/core/checkpoint.py b/cuda_core/cuda/core/checkpoint.py index 08a0f0189c0..579f46f73d2 100644 --- a/cuda_core/cuda/core/checkpoint.py +++ b/cuda_core/cuda/core/checkpoint.py @@ -2,11 +2,11 @@ # # SPDX-License-Identifier: Apache-2.0 -"""CUDA process checkpointing (Linux). +"""CUDA process checkpointing on Linux. -Requires the CUDA 13 build of cuda.core (the driver structures it uses are -CUDA 13 types) and a CUDA driver of version 12.8 or newer with checkpoint API -support. +This module requires the CUDA 13 build of cuda.core, because the driver +structures it uses are CUDA 13 types. It also requires a CUDA driver of version +12.8 or newer with checkpoint API support. """ import ctypes as _ctypes @@ -125,10 +125,10 @@ def _get_driver() -> Any: if _driver_capability_checked: return _driver - # Restoring onto other GPUs uses CUcheckpointGpuPair, a CUDA 13 type that - # the CUDA 12 build's cuda-bindings does not have. + # A restore onto other GPUs uses CUcheckpointGpuPair, a CUDA 13 type that + # cuda-bindings 12.x does not have. if _BUILD_CUDA_MAJOR < 13: - raise RuntimeError("CUDA checkpointing requires the CUDA 13 build of cuda.core (cuda-core[cu13]).") + raise RuntimeError("CUDA checkpointing requires the CUDA 13 build of cuda.core. Install cuda-core[cu13].") driver_ver = _driver_version() if driver_ver < _REQUIRED_DRIVER_VERSION: diff --git a/cuda_core/cuda/core/graph/_subclasses.pyi b/cuda_core/cuda/core/graph/_subclasses.pyi index dcc526cbc54..e380307b670 100644 --- a/cuda_core/cuda/core/graph/_subclasses.pyi +++ b/cuda_core/cuda/core/graph/_subclasses.pyi @@ -138,9 +138,9 @@ class MemsetNode(GraphNode): only accompany a raw-address ``dst``. With drivers from CUDA 12.2 through 13.1, the node's intended CUDA - context must be current when this method is called. With the CUDA 13 - build of ``cuda.core`` and a driver of CUDA 13.2 or newer, the recorded - context is preserved automatically. + context must be current when this method runs. With the CUDA 13 build + of ``cuda.core`` and a driver of CUDA 13.2 or newer, this method + preserves the recorded context. .. warning:: @@ -190,9 +190,9 @@ class MemcpyNode(GraphNode): not supported. With drivers from CUDA 12.2 through 13.1, the node's intended CUDA - context must be current when this method is called. With the CUDA 13 - build of ``cuda.core`` and a driver of CUDA 13.2 or newer, the recorded - context is preserved automatically. + context must be current when this method runs. With the CUDA 13 build + of ``cuda.core`` and a driver of CUDA 13.2 or newer, this method + preserves the recorded context. .. warning:: diff --git a/cuda_core/cuda/core/graph/_subclasses.pyx b/cuda_core/cuda/core/graph/_subclasses.pyx index cf2d704facb..540849ee97a 100644 --- a/cuda_core/cuda/core/graph/_subclasses.pyx +++ b/cuda_core/cuda/core/graph/_subclasses.pyx @@ -203,9 +203,9 @@ cdef void _set_executable_node_enabled( cdef bint _check_node_get_params(): - """Whether cuGraphNodeGetParams (CUDA 13.2) can be called. + """Whether cuGraphNodeGetParams, a 13.2 driver API, is available. - The CUDA 13 build always has the binding; only the driver can lack it.""" + The CUDA 13 build always has the binding. Only the driver can lack it.""" IF CUDA_CORE_BUILD_MAJOR >= 13: return cy_driver_version() >= (13, 2, 0) ELSE: @@ -651,9 +651,9 @@ cdef class MemsetNode(GraphNode): only accompany a raw-address ``dst``. With drivers from CUDA 12.2 through 13.1, the node's intended CUDA - context must be current when this method is called. With the CUDA 13 - build of ``cuda.core`` and a driver of CUDA 13.2 or newer, the recorded - context is preserved automatically. + context must be current when this method runs. With the CUDA 13 build + of ``cuda.core`` and a driver of CUDA 13.2 or newer, this method + preserves the recorded context. .. warning:: @@ -841,9 +841,9 @@ cdef class MemcpyNode(GraphNode): not supported. With drivers from CUDA 12.2 through 13.1, the node's intended CUDA - context must be current when this method is called. With the CUDA 13 - build of ``cuda.core`` and a driver of CUDA 13.2 or newer, the recorded - context is preserved automatically. + context must be current when this method runs. With the CUDA 13 build + of ``cuda.core`` and a driver of CUDA 13.2 or newer, this method + preserves the recorded context. .. warning:: diff --git a/cuda_core/cuda/core/system/_system.pyx b/cuda_core/cuda/core/system/_system.pyx index f962613c82d..fbae1c01b40 100644 --- a/cuda_core/cuda/core/system/_system.pyx +++ b/cuda_core/cuda/core/system/_system.pyx @@ -3,18 +3,18 @@ # SPDX-License-Identifier: Apache-2.0 -# cuda.core.system uses NVML through cuda.bindings.nvml, which every -# cuda-bindings cuda.core accepts provides (cuda/core/_bindings_floor.py). -# Loading the NVML library itself happens in initialize(), on first use, so -# this module stays importable without CUDA or NVML installed. +# cuda.core.system uses NVML through cuda.bindings.nvml. Every cuda-bindings +# that cuda.core accepts provides the module. See cuda/core/_bindings_floor.py. +# initialize() loads the NVML library itself on first use, so this module +# stays importable without CUDA or NVML installed. from typing import TYPE_CHECKING -# Always True since the cuda-bindings floor made NVML support unconditional; -# kept for callers that read it. Assigned in a runtime-only block so that the -# generated stub keeps the bare annotation the public API had (the API check -# reports a changed attribute value otherwise). +# Always True, because the cuda-bindings floor made NVML support unconditional. +# Kept for callers that read it. The assignment sits in a runtime-only block so +# that the generated stub keeps the bare annotation the public API had. The API +# check reports a changed attribute value otherwise. CUDA_BINDINGS_NVML_IS_COMPATIBLE: bool if not TYPE_CHECKING: CUDA_BINDINGS_NVML_IS_COMPATIBLE = True diff --git a/cuda_core/cuda/core/system/typing.py b/cuda_core/cuda/core/system/typing.py index a083163379b..208e38876d9 100644 --- a/cuda_core/cuda/core/system/typing.py +++ b/cuda_core/cuda/core/system/typing.py @@ -325,8 +325,8 @@ class ThermalTarget(StrEnum): ThermalTarget.VCD_OUTLET.__doc__ = "Visual Computing Device Outlet temperature requires visual computing device handle." -# DeviceArch values are derived from cuda.bindings.nvml at definition time. -# An IntEnum rather than a StrEnum because the ordering of the values is +# DeviceArch takes its values from cuda.bindings.nvml at definition time. +# It is an IntEnum rather than a StrEnum because the order of the values is # meaningful, e.g. Kepler "or later". class DeviceArch(enum.IntEnum): """ diff --git a/cuda_core/docs/source/api.rst b/cuda_core/docs/source/api.rst index ec898c039ec..5831172bc35 100644 --- a/cuda_core/docs/source/api.rst +++ b/cuda_core/docs/source/api.rst @@ -166,15 +166,15 @@ Parameter-bearing definition nodes expose subclass-specific ``update()`` methods: :class:`~graph.KernelNode`, :class:`~graph.MemcpyNode`, :class:`~graph.MemsetNode`, :class:`~graph.ChildGraphNode`, :class:`~graph.EventRecordNode`, :class:`~graph.EventWaitNode`, and -:class:`~graph.HostCallbackNode`. These methods require a CUDA driver of -version 12.2 or newer. Updates affect future graph +:class:`~graph.HostCallbackNode`. These methods require CUDA driver 12.2 or +newer. Updates affect future graph instantiations; executable graphs that were already instantiated continue using their previous parameters and retained resources. Omitted optional arguments preserve their current values where supported. -With drivers from CUDA 12.2 through 13.1, the intended CUDA context must be -current when updating memcpy or memset nodes. With the CUDA 13 build of -``cuda.core`` and a driver of CUDA 13.2 or newer, the recorded context is -preserved automatically. +With CUDA driver 12.2 through 13.1, an update of a memcpy or memset node +requires the intended CUDA context to be current. With the CUDA 13 build of +``cuda.core`` and CUDA driver 13.2 or newer, updates preserve the recorded +context. Multidimensional or array-backed memcpy nodes and clustered or cooperative kernel nodes cannot currently be updated. Clustered and cooperative kernel nodes also cannot currently be constructed explicitly. @@ -218,8 +218,7 @@ Memcpy and memset updates use the current CUDA context, which must match the original node context. Kernel, memcpy, and memset views also provide ``is_enabled``, ``enable()``, and -``disable()``. Executable-node updates require a CUDA driver of version 12.2 -or newer. +``disable()``. Executable-node updates require CUDA driver 12.2 or newer. .. autosummary:: :toctree: generated/ @@ -335,8 +334,8 @@ CUDA process checkpointing The :mod:`cuda.core.checkpoint` module wraps the CUDA driver process checkpoint APIs. These APIs are intended for Linux process checkpoint and -restore workflows, and require the CUDA 13 build of ``cuda.core`` and a CUDA -driver of version 12.8 or newer with checkpoint API support. +restore workflows. They require the CUDA 13 build of ``cuda.core`` and CUDA +driver 12.8 or newer with checkpoint API support. Checkpointing is typically driven by a coordinator process acting on a target CUDA process, similar to attaching a debugger or sending a signal. The target diff --git a/cuda_core/docs/source/api_nvml.rst b/cuda_core/docs/source/api_nvml.rst index 06ceecfa515..af49e176c0c 100644 --- a/cuda_core/docs/source/api_nvml.rst +++ b/cuda_core/docs/source/api_nvml.rst @@ -11,8 +11,8 @@ through the NVIDIA Management Library (NVML). .. note:: ``cuda.core.system`` uses NVML through ``cuda-bindings``. It has no requirement beyond the - ``cuda-bindings`` floor of the release (see :ref:`cuda-core-bindings-floor`); the NVML library - itself is loaded on first use, so importing the module needs neither CUDA nor NVML installed. + ``cuda-bindings`` floor of the release. See :ref:`cuda-core-bindings-floor`. The NVML library + loads on first use, so ``import cuda.core.system`` needs neither CUDA nor NVML installed. Basic functions --------------- diff --git a/cuda_core/docs/source/conf.py b/cuda_core/docs/source/conf.py index fd62b3877ef..4e73974e167 100644 --- a/cuda_core/docs/source/conf.py +++ b/cuda_core/docs/source/conf.py @@ -21,9 +21,11 @@ def _bindings_floor_substitutions() -> str: - """|cuda-bindings-floor-cu12| and |cuda-bindings-floor-cu13|, read from the - cu12/cu13 extras of pyproject.toml, the single place the floors are declared - (see cuda/core/_bindings_floor.py). Used by support.rst.""" + """Return the rst_prolog that defines |cuda-bindings-floor-cu12| and |cuda-bindings-floor-cu13|. + + The cu12 and cu13 extras of pyproject.toml are the single place that declares the floors. + See cuda/core/_bindings_floor.py. support.rst uses the substitutions. + """ import importlib.util import tomllib diff --git a/cuda_core/docs/source/install.rst b/cuda_core/docs/source/install.rst index f241793b318..320d8179982 100644 --- a/cuda_core/docs/source/install.rst +++ b/cuda_core/docs/source/install.rst @@ -45,9 +45,9 @@ Starting ``cuda-core`` 0.4.0, **experimental** packages for the `free-threaded i Installing from PyPI -------------------- -``cuda.core`` works with ``cuda-bindings`` (part of ``cuda-python``) 12 or 13, at or above the -release's per-major floor (see :ref:`cuda-core-bindings-floor`); the ``cu12`` and ``cu13`` extras -install a compatible version. Test dependencies now use the ``cuda-toolkit`` metapackage for improved dependency resolution. For example with CUDA 12: +``cuda.core`` works with ``cuda-bindings`` (part of ``cuda-python``) 12 or 13 at or above the +release's floor for that major. See :ref:`cuda-core-bindings-floor`. The ``cu12`` and ``cu13`` +extras install a compatible version. Test dependencies now use the ``cuda-toolkit`` metapackage for improved dependency resolution. For example with CUDA 12: .. code-block:: console @@ -55,8 +55,8 @@ install a compatible version. Test dependencies now use the ``cuda-toolkit`` met and likewise use ``[cu13]`` for CUDA 13. -Upgrading ``cuda-core`` on its own can leave an older ``cuda-bindings`` installed than the new -release requires; ``import cuda.core`` then reports the required version and the ``pip`` command +If you upgrade ``cuda-core`` alone, the installed ``cuda-bindings`` can be older than the new +release requires. ``import cuda.core`` then reports the required version and the ``pip`` command that installs it. @@ -71,9 +71,8 @@ Same as above, ``cuda.core`` can be installed in a CUDA 12 or 13 environment. Fo and likewise use ``cuda-version=13`` for CUDA 13. -The conda-forge package depends on ``cuda-bindings`` of the same CUDA major; the -``cuda-bindings`` floor of the release (see :ref:`cuda-core-bindings-floor`) applies to it as -well. +The conda-forge package depends on ``cuda-bindings`` of the same CUDA major. The +``cuda-bindings`` floor of the release also applies to it. See :ref:`cuda-core-bindings-floor`. Development environment @@ -160,15 +159,16 @@ Installing from Source $ cd cuda-python/cuda_core $ pip install . -A source build requires two things to agree (see :ref:`cuda-core-bindings-floor`): +A source build has two requirements. See :ref:`cuda-core-bindings-floor`. - ``cuda-bindings`` 12.x or 13.x at or above the release's floor for that major. An isolated - build (the default ``pip install``) installs one no newer than the toolkit's minor; other - builds must provide it. + build, the default for ``pip install``, installs a ``cuda-bindings`` no newer than the + toolkit's minor. Other builds must provide it. - A CUDA Toolkit, located through ``CUDA_PATH`` or ``CUDA_HOME``, whose ``cuda.h`` has the same - major.minor as the header that ``cuda-bindings`` was generated from. The build fails early - otherwise. To build against a particular ``cuda-bindings`` in an isolated build, constrain it - with ``PIP_CONSTRAINT``; or use ``--no-build-isolation`` with it installed. + major.minor as the header that ``cuda-bindings`` was generated from. If the versions differ, + the build fails early. To build against a specific ``cuda-bindings`` in an isolated build, + constrain it with ``PIP_CONSTRAINT``. A build with ``--no-build-isolation`` uses the installed + ``cuda-bindings``. .. note:: diff --git a/cuda_core/docs/source/release/1.3.0-notes.rst b/cuda_core/docs/source/release/1.3.0-notes.rst index f403ee7e94e..5329b2db033 100644 --- a/cuda_core/docs/source/release/1.3.0-notes.rst +++ b/cuda_core/docs/source/release/1.3.0-notes.rst @@ -9,27 +9,32 @@ Breaking Changes ---------------- -- ``cuda.core`` now requires a minimum ``cuda-bindings`` version per CUDA major, at build time and - at run time: 12.9.8 for CUDA 12 and 13.4.1 for CUDA 13 (see the - :ref:`support policy `). ``import cuda.core`` with an older - ``cuda-bindings`` fails with a message that names the required version and how to install it; - ``pip install cuda-core[cu12]`` / ``[cu13]`` installs a compatible version. A source build must - also use a ``cuda.h`` of the same major.minor as its ``cuda-bindings``; it fails early otherwise. - Supported CUDA drivers and CUDA Toolkit libraries are unchanged. Previously any - ``cuda-bindings`` of the right major was accepted, and an older one produced import errors for - missing C functions, silently disabled features, or crashes. +- ``cuda.core`` now requires a minimum ``cuda-bindings`` version per CUDA major, at build and run + time: 12.9.8 for CUDA 12 and 13.4.1 for CUDA 13. See the + :ref:`support policy `. ``import cuda.core`` with an older + ``cuda-bindings`` fails with a message that names the required version and how to install it. + ``pip install cuda-core[cu12]`` or ``pip install cuda-core[cu13]`` installs a compatible + version. Supported CUDA drivers and CUDA Toolkit libraries are unchanged. Previously + ``cuda.core`` accepted any ``cuda-bindings`` of the right major, and an older one produced import + errors for missing C functions, silently disabled features, or crashes. (https://github.com/NVIDIA/cuda-python/issues/2783) -- With the ``cuda-bindings`` floor in place, whether a feature is available now depends on the - CUDA driver alone (and on the CUDA major of the ``cuda.core`` build). Checks that also inspected - the ``cuda-bindings`` version are gone, and the error messages they produced with them; error - messages that name a minimum now name a driver version. ``cuda.core.system`` always uses NVML - through ``cuda-bindings``, so ``cuda.core.system.CUDA_BINDINGS_NVML_IS_COMPATIBLE`` is always - ``True``. :mod:`cuda.core.checkpoint` requires the CUDA 13 build of - ``cuda.core`` (it did in effect before: the CUDA 12 ``cuda-bindings`` lack a type it uses) and - now says so. The C++ layer calls the driver through the entry points ``cuda-bindings`` resolves - rather than through its Cython wrappers, so a driver function that the installed driver lacks - can no longer surface as ``SystemError`` from a C++ call. +- A source build of ``cuda.core`` must use a ``cuda.h`` of the same major.minor as its + ``cuda-bindings``. If they differ, the build fails early. + (https://github.com/NVIDIA/cuda-python/issues/2783) + +- With the ``cuda-bindings`` floor in place, whether a feature is available now depends only on + the CUDA driver and the ``cuda.core`` build's CUDA major. The checks that also inspected the + ``cuda-bindings`` version are gone, and so are their error messages. Error messages that name a + minimum now name a driver version. ``cuda.core.system`` always uses NVML through + ``cuda-bindings``, so ``cuda.core.system.CUDA_BINDINGS_NVML_IS_COMPATIBLE`` is always ``True``. + :mod:`cuda.core.checkpoint` requires the CUDA 13 build of ``cuda.core`` and now says so. That + requirement was already in effect, because the CUDA 12 ``cuda-bindings`` lack a type it uses. + (https://github.com/NVIDIA/cuda-python/issues/2783) + +- The C++ layer calls the driver through the entry points ``cuda-bindings`` resolves rather than + through its Cython wrappers. As a result, a driver function that the installed driver lacks can + no longer surface as ``SystemError`` from a C++ call. (https://github.com/NVIDIA/cuda-python/issues/2783) New features diff --git a/cuda_core/docs/source/support.rst b/cuda_core/docs/source/support.rst index cbb2d300551..36dffdf2c75 100644 --- a/cuda_core/docs/source/support.rst +++ b/cuda_core/docs/source/support.rst @@ -45,10 +45,10 @@ CUDA Version Support example, ``cuda.core`` 1.x supports CUDA 12 and 13. In particular, what this entails is that all CUDA minor versions within the two major releases -(12.x, 13.x) are supported by the same ``cuda-core`` package, at run time: any CUDA driver and any -CUDA Toolkit libraries of a supported major work with the same ``cuda-core`` wheel. The one input -this does not extend to is ``cuda-bindings``, which has a per-release minimum (see -:ref:`cuda-core-bindings-floor` below). +(12.x, 13.x) are supported by the same ``cuda-core`` package at run time. Any CUDA driver and any +CUDA Toolkit libraries of a supported major work with the same ``cuda-core`` wheel. The exception +is ``cuda-bindings``, which has a minimum version per release. See +:ref:`cuda-core-bindings-floor` below. When a new CUDA major version is released and support for the oldest major version is dropped, ``cuda.core`` will release a new major version (e.g., 1.x → 2.0.0). @@ -69,12 +69,11 @@ CUDA library or CUDA driver versions. Refer to the individual module documentati ``cuda-bindings`` Version Requirements ************************************** -Each ``cuda-core`` release declares, for each supported CUDA major version, a minimum -``cuda-bindings`` version, its *floor*: the newest ``cuda-bindings`` release of that major at the -time of the ``cuda-core`` release, which is the version the published wheels are built against. -The floors of the current release are declared by the ``cu12``/``cu13`` extras of ``cuda-core`` -(in ``pyproject.toml``); the build, the import-time check, this page and CI all read them from -there. +For each supported CUDA major version, each ``cuda-core`` release declares a minimum +``cuda-bindings`` version, its *floor*. The floor is the newest ``cuda-bindings`` release of that +major at the time of the ``cuda-core`` release. The published wheels are built against it. The +``cu12`` and ``cu13`` extras of ``cuda-core`` in ``pyproject.toml`` declare the floors of the +current release. The build, the import-time check, this page, and CI all read them from there. .. list-table:: ``cuda-bindings`` floors :header-rows: 1 @@ -86,24 +85,24 @@ there. - ``cuda-bindings`` >= |cuda-bindings-floor-cu12| - ``cuda-bindings`` >= |cuda-bindings-floor-cu13| -- **At run time**, ``import cuda.core`` requires an installed ``cuda-bindings`` of the same major - as the ``cuda-core`` build in use, at least as new as that build's floor, and generated from a - ``cuda.h`` at least as new (by major.minor) as the one the build compiled against; the published - wheels are built against the floor's header, so the floor alone satisfies them. An older - ``cuda-bindings`` fails at import with a message that names the version found, the version - required, and the ``pip`` command that fixes it. A newer ``cuda-bindings`` of the same major is - supported. -- **At build time**, a source build requires ``cuda-bindings`` at or above the floor and a - ``cuda.h`` (``CUDA_PATH`` or ``CUDA_HOME``) of the same major.minor as the header that - ``cuda-bindings`` was generated from. Any other configuration fails the build with a message - that names what was found and what is required. Building against an older CUDA Toolkit than - the floor's minor is not supported. +- **At run time**, ``import cuda.core`` requires an installed ``cuda-bindings`` that meets three + conditions. It has the same major as the ``cuda-core`` build in use and is at least as new as + that build's floor. It was generated from a ``cuda.h`` at least as new, by major.minor, as the + one the build compiled against. The published wheels are built against the floor's header, so + the floor alone satisfies them. An older ``cuda-bindings`` fails at import with a message that + names the version found, the version required, and the ``pip`` command that fixes it. + ``cuda.core`` supports a newer ``cuda-bindings`` of the same major. +- **At build time**, a source build requires ``cuda-bindings`` at or above the floor. It also + requires a ``cuda.h``, located through ``CUDA_PATH`` or ``CUDA_HOME``, of the same major.minor + as the header that ``cuda-bindings`` was generated from. Any other configuration fails the + build with a message that names what was found and what is required. ``cuda.core`` does not + support a build against a CUDA Toolkit older than the floor's minor. - **The CUDA driver** is unaffected by the floor. Feature availability is decided by the driver alone: a feature the installed driver lacks raises when it is used. A floor moves with each ``cuda-core`` release, to the newest ``cuda-bindings`` of each major at -that time, and in any release whose changes need a newer ``cuda-bindings`` API. Every move is -listed under "Breaking Changes" in the :doc:`release notes `. +that time. It also moves in any release whose changes need a newer ``cuda-bindings`` API. The +:doc:`release notes ` list every move under "Breaking Changes". Python Version Support ---------------------- diff --git a/cuda_core/pyproject.toml b/cuda_core/pyproject.toml index f4fad12ef6b..6971f68a7b2 100644 --- a/cuda_core/pyproject.toml +++ b/cuda_core/pyproject.toml @@ -56,7 +56,7 @@ dependencies = [ ] # The cuda-bindings pins below are the cuda-bindings *floors* of this release, one -# per CUDA major, and the only place they are declared: build_hooks.py, the +# per CUDA major. This is the only place that declares them. build_hooks.py, the # import-time check, the docs and CI read them from here (see # cuda/core/_bindings_floor.py and the "Bumping the cuda-bindings floor" # checklist in AGENTS.md). Keep the form `>=,<`. diff --git a/cuda_core/tests/memory/test_copy_single_options.py b/cuda_core/tests/memory/test_copy_single_options.py index 3dc17ebd4a6..044c5a4c118 100644 --- a/cuda_core/tests/memory/test_copy_single_options.py +++ b/cuda_core/tests/memory/test_copy_single_options.py @@ -19,10 +19,10 @@ def _options_honored(): """True when cuMemcpyWithAttributesAsync will actually be used for options. - Mirrors _with_attributes_available() in _copy_attributes.pxd. CI runs a - matrix that includes CUDA 12 builds and pre-13.2 drivers (see - ci/test-matrix.yml), where this is False and the DURING_API_CALL tests - below must expect a RuntimeError instead of a successful copy. + Mirrors _with_attributes_available() in _copy_attributes.pxd. The CI + matrix in ci/test-matrix.yml includes CUDA 12 builds and pre-13.2 + drivers. On those runs this returns False, and the DURING_API_CALL + tests below must expect a RuntimeError instead of a successful copy. """ return BUILD_CUDA_MAJOR >= 13 and driver_version() >= (13, 2, 0) diff --git a/cuda_core/tests/test_bindings_floor.py b/cuda_core/tests/test_bindings_floor.py index 4e1ee66fe8b..deca684591e 100644 --- a/cuda_core/tests/test_bindings_floor.py +++ b/cuda_core/tests/test_bindings_floor.py @@ -2,14 +2,15 @@ # # SPDX-License-Identifier: Apache-2.0 -"""The cuda-bindings version floor (cuda/core/_bindings_floor.py): reading it -from the pyproject extras, the import-time check built on it, and the -consistency hook that guards ci/versions.yml and the docs. +"""Tests for the cuda-bindings version floor in cuda/core/_bindings_floor.py. -Source-tree properties and pure functions only: no GPU, so this file also runs -with --noconftest (conftest.py initializes CUDA). The consistency tests read -pyproject.toml and ci/versions.yml from the checkout, so they need the source -tree next to the tests, which every CI job that runs tests/ has. +The tests cover the floor as read from the pyproject extras and the import-time check +built on it. They also cover the consistency hook that guards ci/versions.yml and the docs. + +The tests check source-tree properties and pure functions and need no GPU. The file +also runs with --noconftest, which skips the CUDA setup in conftest.py. The consistency +tests read pyproject.toml and ci/versions.yml from the checkout. They need the source +tree next to the tests. Every CI job that runs tests/ has it. pytest tests/test_bindings_floor.py -v --noconftest """ @@ -52,13 +53,13 @@ def _load(name, path): def hook(): """toolshed/check_cuda_core_bindings_floor.py lives outside cuda_core/, so an sdist tree lacks it.""" if not HOOK.is_file(): - pytest.skip(f"{HOOK} is not in this tree; the hook tests need the monorepo checkout") + pytest.skip(f"{HOOK} is not in this tree. The hook tests need the monorepo checkout") return _load("check_cuda_core_bindings_floor", HOOK) @pytest.mark.agent_authored(model="claude-fable-5-1") def test_floor_module_is_import_free(): - """build_hooks.py, conf.py and the hook load it by file path; it must stay standard-library only.""" + """build_hooks.py, conf.py and the hook load it by file path, so it must stay standard-library only.""" source = Path(floor_mod.__file__).read_text(encoding="utf-8") imports = re.findall(r"^\s*(?:from|import)\s+(\w+)", source, re.M) assert set(imports) <= {"__future__", "collections", "re"} @@ -66,7 +67,7 @@ def test_floor_module_is_import_free(): @pytest.mark.agent_authored(model="claude-fable-5-1") def test_floors_come_from_the_pyproject_extras(hook): - """The extras are the single source; reading them back gives one release triple per major.""" + """The extras are the single source. read_floors() returns one release triple per major.""" floors = hook.read_floors(REPO) assert sorted(floors) == [12, 13] for major, floor in floors.items(): @@ -156,7 +157,7 @@ class TestCheckInstalledBindings: def check(self, installed, major=13, header=None, floor=None, installed_header=None): """installed_header defaults to the header of the installed version's major.minor, - as a release of that version would have been generated from.""" + the header that a release of that version was generated from.""" if installed_header is None: triple = release_triple(installed) or (major, 0, 0) installed_header = cuda_version_of(triple) @@ -170,8 +171,8 @@ def check(self, installed, major=13, header=None, floor=None, installed_header=N [ "13.4.1", "13.4.2", - "13.4.2.dev249+gabcdef0", # main-built bindings in CI - "13.5.0b1", # newer bindings than the build: supported + "13.4.2.dev249+gabcdef0", # main-built cuda-bindings in CI + "13.5.0b1", # newer cuda-bindings than the build: supported ], ) def test_accepts_the_floor_and_newer(self, installed): @@ -183,12 +184,12 @@ def test_rejects_older_than_the_floor_with_the_fix(self): self.check("13.3.1") message = str(excinfo.value) assert "requires cuda-bindings >= 13.4.1 for CUDA 13" in message - assert "(found 13.3.1)" in message - assert 'pip install -U "cuda-bindings>=13.4.1,<14"' in message # double quotes: cmd.exe too + assert "but cuda-bindings 13.3.1 is installed" in message + assert 'pip install -U "cuda-bindings>=13.4.1,<14"' in message # double quotes work in cmd.exe too @pytest.mark.agent_authored(model="claude-fable-5-1") def test_rejects_bindings_generated_from_an_older_header_than_the_build(self): - # Built against 13.5 headers; a 13.4-generated cuda-bindings lacks table entries. + # The build uses 13.5 headers. A cuda-bindings generated from 13.4 lacks table entries. with pytest.raises(ImportError) as excinfo: self.check("13.4.1", header=self.HEADER + 10) message = str(excinfo.value) @@ -200,17 +201,17 @@ def test_rejects_bindings_generated_from_an_older_header_than_the_build(self): @pytest.mark.agent_authored(model="claude-fable-5-1") def test_header_rule_compares_headers_not_version_strings(self): - # A development cuda-bindings carries the previous release's version string - # (13.4.2.dev5) but was generated from the new 13.5 header: accepted. + # A development cuda-bindings carries the previous release's version string, + # 13.4.2.dev5, but was generated from the new 13.5 header. The check accepts it. floor = (13, 4, 2) assert self.check("13.4.2.dev5+gabc", header=13050, floor=floor, installed_header=13050) == (13, 4, 2) - # The converse, a 13.5 version string generated from 13.4 headers, is rejected. + # The check rejects the converse, a 13.5 version string generated from 13.4 headers. with pytest.raises(ImportError, match="needs cuda-bindings 13.5 or newer"): self.check("13.5.0", header=13050, floor=floor, installed_header=13040) @pytest.mark.agent_authored(model="claude-fable-5-1") def test_the_floor_comes_from_the_build_record(self): - # A build recorded with a lower floor (and header) accepts what the default floor rejects. + # A build recorded with a lower floor and header accepts what the default floor rejects. assert self.check("13.3.0", header=cuda_version_of((13, 2, 0)), floor=(13, 2, 0)) == (13, 3, 0) # A higher recorded floor rejects what the default floor accepts. with pytest.raises(ImportError) as excinfo: @@ -227,27 +228,29 @@ def test_rejects_another_major_than_the_build(self): @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize("installed", ["11.8.0", "14.0.0"]) def test_another_major_names_only_the_fix_it_knows(self, installed): - """A plain (single-build) install cannot know which other builds exist, so the message - must not promise one.""" + """A single-build install cannot know which other builds exist, so the message must not + promise one.""" with pytest.raises(ImportError) as excinfo: self.check(installed, major=12, header=12090, floor=(12, 9, 8)) message = str(excinfo.value) - assert 'Install cuda-bindings 12.x (pip install "cuda-bindings==12.*")' in message - assert f"build for CUDA {installed.split('.')[0]} if one exists" in message + assert 'Install cuda-bindings 12.x with: pip install "cuda-bindings==12.*"' in message + assert f"If a cuda.core build for CUDA {installed.split('.')[0]} exists, install it instead." in message @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize("installed", ["0.1.dev1+g0d22cb444", "garbage"]) def test_rejects_unparseable_versions(self, installed): with pytest.raises( - ImportError, match=rf"a cuda-bindings 13\.x release is required \(found {re.escape(installed)}\)" + ImportError, + match=rf"requires a cuda-bindings 13\.x release, " + rf"but the installed cuda-bindings version is {re.escape(installed)}", ): self.check(installed) class TestImportTimeCheck: """`import cuda.core` runs check_installed_bindings against the installed build's record - before importing any extension module. A fake cuda.bindings in a child interpreter - exercises the reject paths end to end (issue #2783 asked for this test).""" + before it imports any extension module. A fake cuda.bindings in a child interpreter + exercises the reject paths end to end. Issue #2783 asked for this test.""" _CHILD = textwrap.dedent(""" import sys, types @@ -307,17 +310,21 @@ def test_bindings_from_an_older_header_fail_at_import(self, tmp_path): def test_unparseable_version_fails_at_import(self, tmp_path): major, cuda_version, floor = self._build() # A major but no release triple. The build's major keeps the merged wheel on its cu - # build; a foreign major (a shallow clone's 0.1.dev1) stops earlier there with "no build for CUDA 0". + # build. A foreign major, such as a shallow clone's 0.1.dev1, stops earlier there with + # "no build for CUDA 0". no_triple = f"{major}.4" message = self._import_error(no_triple, cuda_version, tmp_path) - assert f"a cuda-bindings {major}.x release is required (found {no_triple})" in message + assert ( + f"requires a cuda-bindings {major}.x release, but the installed cuda-bindings version is {no_triple}" + in message + ) # No major at all. message = self._import_error("garbage", cuda_version, tmp_path) - assert "a cuda-bindings release must be installed (found version 'garbage')" in message + assert "requires a cuda-bindings release, but the installed cuda-bindings version is 'garbage'" in message class TestConsistencyHook: - """toolshed/check_cuda_core_bindings_floor.py, also the pre-commit hook.""" + """Tests for toolshed/check_cuda_core_bindings_floor.py, which is also the pre-commit hook.""" @pytest.mark.agent_authored(model="claude-fable-5-1") def test_the_checkout_is_consistent(self, hook): @@ -326,7 +333,7 @@ def test_the_checkout_is_consistent(self, hook): @pytest.mark.agent_authored(model="claude-fable-5-1") def test_a_malformed_extra_fails_the_hook(self, hook, tmp_path, capsys): - # check() returns before reading ci/versions.yml or the docs, so the tree needs only these two files. + # check() returns before it reads ci/versions.yml or the docs, so the tree needs only these two files. core = tmp_path / "cuda_core" / "cuda" / "core" core.mkdir(parents=True) shutil.copy(CUDA_CORE / "cuda" / "core" / "_bindings_floor.py", core) diff --git a/cuda_core/tests/test_build_hooks.py b/cuda_core/tests/test_build_hooks.py index 90d0ed63775..4f2874703f3 100644 --- a/cuda_core/tests/test_build_hooks.py +++ b/cuda_core/tests/test_build_hooks.py @@ -232,8 +232,8 @@ def fake_cythonize(ext_modules, **kwargs): # Builds resolve the CTK for include dirs; stub it so the test runs # where no toolkit is installed (e.g. the wheels CI jobs). monkeypatch.setattr(build_hooks, "_get_cuda_path", lambda: "/nonexistent-cuda") - # The configuration check reads that header and the installed cuda-bindings; - # it has its own tests (TestBuildConfigurationCheck). + # The configuration check reads that header and the installed cuda-bindings. + # TestBuildConfigurationCheck covers it. monkeypatch.setattr(build_hooks, "_check_build_configuration", lambda *_: None) monkeypatch.setattr(build_hooks, "cythonize", fake_cythonize) monkeypatch.setenv("CUDA_CORE_BUILD_MAJOR", cuda_major) @@ -457,8 +457,9 @@ def test_serial_builds_and_compilers_without_the_hook_keep_the_stock_path(self, def _fake_bindings(monkeypatch, version, cuda_version=None): - """Make the build see an installed cuda-bindings of `version` (None: not installed), - generated from the header `cuda_version` (default: the header of its major.minor).""" + """Make the build see an installed cuda-bindings of `version`, generated from the header + `cuda_version`. None for `version` means not installed. `cuda_version` defaults to the + header of the version's major.minor.""" def installed_cuda_bindings(): if version is None: @@ -486,7 +487,7 @@ class TestBuildConfigurationCheck: """_check_build_configuration() accepts exactly one configuration per CUDA major: cuda-bindings at or above the floor, and a cuda.h of the same major.minor as that cuda-bindings. Anything else is a build error that - names what was found and what is required.""" + names what the check found and what it requires.""" FLOOR = build_hooks._bindings_floors() @@ -511,7 +512,7 @@ def test_floor_bindings_and_matching_header_pass_and_are_recorded(self, tmp_path assert floor[0] * 1000 + floor[1] * 10 == info.CUDA_VERSION assert floor == info.CUDA_BINDINGS_FLOOR assert version == info.CUDA_BINDINGS_BUILD_VERSION - # ci/tools/cuda_core_bindings_floor.py reads this record out of the wheel (BINDINGS_SOURCE=floor). + # ci/tools/cuda_core_bindings_floor.py reads this record out of the wheel when BINDINGS_SOURCE=floor. tool_path = Path(__file__).resolve().parents[2] / "ci" / "tools" / "cuda_core_bindings_floor.py" if tool_path.is_file(): # absent from an sdist tree spec = importlib.util.spec_from_file_location("cuda_core_bindings_floor_tool", tool_path) @@ -537,7 +538,8 @@ def test_bindings_of_another_major_fail(self, tmp_path, monkeypatch): _fake_bindings(monkeypatch, _floor_str(13)) cuda_path = _write_cuda_h(tmp_path, 12090) with pytest.raises( - RuntimeError, match=f"Building cuda.core for CUDA 12, but the installed cuda-bindings is {_floor_str(13)}" + RuntimeError, + match=f"This cuda.core build is for CUDA 12, but the installed cuda-bindings is {_floor_str(13)}", ): build_hooks._check_build_configuration(cuda_path, "12") @@ -601,7 +603,7 @@ def test_unsupported_major_is_a_build_error(self, tmp_path, monkeypatch): class TestBindingsFloorsFromPyproject: - """_bindings_floors() reads the cu extras of pyproject.toml; a malformed extra fails the build.""" + """_bindings_floors() reads the cu extras of pyproject.toml. A malformed extra fails the build.""" @pytest.mark.agent_authored(model="claude-fable-5-1") def test_malformed_extra_is_a_build_error(self, tmp_path, monkeypatch): @@ -633,8 +635,9 @@ def test_the_checkout_declares_both_majors(self): class TestBuildRequirement: - """get_requires_for_build_wheel pins cuda-bindings for isolated builds: the floor, and the - header's minor when cuda.h is readable, so pip cannot pick a newer minor than the toolkit.""" + """get_requires_for_build_wheel pins cuda-bindings for isolated builds. It pins the floor. + When cuda.h is readable, it also pins the header's minor, so pip cannot pick a newer minor + than the toolkit.""" @staticmethod def _no_cuda_path(): @@ -698,7 +701,7 @@ def test_unsupported_major_names_the_supported_ones(self, monkeypatch): class TestDefineMacros: - """The C++ learns the build decision through two macros (see _cpp/rt/versions.hpp).""" + """The C++ learns the build decision through two macros. See _cpp/rt/versions.hpp.""" @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize("major", ["12", "13"]) diff --git a/cuda_core/tests/test_checkpoint.py b/cuda_core/tests/test_checkpoint.py index d2ce7491532..a52d1b7dabf 100644 --- a/cuda_core/tests/test_checkpoint.py +++ b/cuda_core/tests/test_checkpoint.py @@ -409,10 +409,10 @@ def test_pid_is_read_only(self): from cuda.bindings import driver as _bindings_driver from cuda.core._utils.version import BUILD_CUDA_MAJOR -# The helpers build CUcheckpointGpuPair, a CUDA 13 type the CUDA 12 bindings lack. +# The helpers build CUcheckpointGpuPair, a CUDA 13 type that cuda-bindings 12.x lacks. needs_checkpoint_bindings = pytest.mark.skipif( BUILD_CUDA_MAJOR < 13, - reason="the checkpoint helpers use CUDA 13 binding types", + reason="the checkpoint helpers require the CUDA 13 build", ) diff --git a/cuda_core/tests/test_driver_table.py b/cuda_core/tests/test_driver_table.py index 523898e20ac..774c2060e41 100644 --- a/cuda_core/tests/test_driver_table.py +++ b/cuda_core/tests/test_driver_table.py @@ -2,17 +2,18 @@ # # SPDX-License-Identifier: Apache-2.0 -"""The C++ driver function table (cuda/core/_cpp/rt/py_driver_fns.cpp) when its fill fails. +"""Tests for the C++ driver function table in cuda/core/_cpp/rt/py_driver_fns.cpp when its fill fails. -The table is filled from ``cuda.bindings._internal.driver._inspect_function_pointers()`` on -the first driver call. A child interpreter replaces that function so the fill fails in a -controlled way, then makes driver calls through ``cuda.core`` and reports what happened. -Expected: the failure is reported once as a :class:`CUDAWarning`, every affected call raises -:class:`CUDAError` with the reason attached as a note, and the failure is latched (no second -warning, no retry). +The first driver call fills the table from ``cuda.bindings._internal.driver._inspect_function_pointers()``. +A child interpreter replaces that function so that the fill fails in a controlled way. It then +makes driver calls through ``cuda.core`` and reports what happened. The expected outcome: + +- The fill reports the failure once as a :class:`CUDAWarning`. +- Every affected call raises :class:`CUDAError` with the reason attached as a note. +- The failure latches: there is no second warning and no retry. The child needs a loadable CUDA driver and a visible device, so this module skips without them. -Runs with ``--noconftest``. +The module runs with ``--noconftest``. """ import os @@ -50,8 +51,8 @@ def _gpu_available() -> bool: def _table_keys() -> list[str]: - """The keys the fill looks up: "__" + the symbol cuda-bindings' loader requests for each - name in driver_api.hpp (cuStreamDestroy -> __cuStreamDestroy_v2).""" + """The keys the fill looks up: "__" + the symbol that the cuda-bindings loader requests for + each name in driver_api.hpp. For example, cuStreamDestroy maps to __cuStreamDestroy_v2.""" if not LOADER.is_file(): pytest.skip("needs the cuda_bindings source tree next to cuda_core") names = re.findall( @@ -73,7 +74,7 @@ def _table_keys() -> list[str]: KEYS = {keys!r} def fake_inspect_function_pointers(): - table = {{key: 1 for key in KEYS}} # placeholder addresses; the fill fails before any call + table = {{key: 1 for key in KEYS}} # placeholder addresses: the fill fails before any call {mutation} return table @@ -85,7 +86,7 @@ def fake_inspect_function_pointers(): warnings.simplefilter("always") for attempt in range(2): try: - # cuInit and the device query are Cython calls; the primary context + # cuInit and the device query are Cython calls. The primary context # retain is the first call through the C++ table. Device(0).set_current() except CUDAError as exc: @@ -146,5 +147,5 @@ def test_missing_table_entry_names_the_mismatch(tmp_path): for line in attempts: assert "CUDAError: CUDA_ERROR_NOT_INITIALIZED" in line assert "has no entry for cuDevicePrimaryCtxRetain" in line - assert "install the cuda-bindings this cuda.core requires" in line + assert "Install the cuda-bindings this cuda.core requires" in line assert "CUDAWARNINGS 1" in lines diff --git a/cuda_core/tests/test_enum_coverage.py b/cuda_core/tests/test_enum_coverage.py index 6e941b1dda7..38badecd29b 100644 --- a/cuda_core/tests/test_enum_coverage.py +++ b/cuda_core/tests/test_enum_coverage.py @@ -312,7 +312,7 @@ } -# CUdevWorkqueueConfigScope was added to the CUDA driver in 13.1; on the +# CUDA 13.1 added CUdevWorkqueueConfigScope to the CUDA driver. On the # CUDA 12 build, WorkqueueSharingScopeType has no driver-side counterpart to # check against. if BUILD_CUDA_MAJOR >= 13: diff --git a/cuda_core/tests/test_green_context.py b/cuda_core/tests/test_green_context.py index bf336f8e0f2..773e0f4f42a 100644 --- a/cuda_core/tests/test_green_context.py +++ b/cuda_core/tests/test_green_context.py @@ -156,7 +156,7 @@ def test_memory_node_updates_preserve_green_context( green_ctx, ): if BUILD_CUDA_MAJOR < 13 or driver_version() < (13, 2, 0): - pytest.skip("generic graph node parameter queries require the CUDA 13 build and driver 13.2+") + pytest.skip("cuGraphNodeGetParams requires the CUDA 13 build and driver 13.2+") memory_resource = LegacyPinnedMemoryResource() src = memory_resource.allocate(4) diff --git a/cuda_core/tests/test_program.py b/cuda_core/tests/test_program.py index d3d15eefa83..3e737caef43 100644 --- a/cuda_core/tests/test_program.py +++ b/cuda_core/tests/test_program.py @@ -38,7 +38,7 @@ def _is_nvvm_available(): return False -nvvm_available = pytest.mark.skipif(not _is_nvvm_available(), reason="NVVM not available (libNVVM not found)") +nvvm_available = pytest.mark.skipif(not _is_nvvm_available(), reason="NVVM not available: libNVVM not found") def _get_nvrtc_version_for_tests(): diff --git a/cuda_core/tests/test_rt_layout.py b/cuda_core/tests/test_rt_layout.py index d9c611a11fc..2c721074e34 100644 --- a/cuda_core/tests/test_rt_layout.py +++ b/cuda_core/tests/test_rt_layout.py @@ -101,8 +101,8 @@ def test_consumer_closure_is_types_and_the_python_seam(): @pytest.mark.agent_authored(model="claude-fable-5-1") def test_cuda_version_is_named_only_in_versions_hpp(): """The C++ may branch on CUDA_CORE_BUILD_MAJOR only. A `#if CUDA_VERSION >= 130x0` - fence compiled a feature out of source builds against an older header while the - run-time checks never noticed (https://github.com/NVIDIA/cuda-python/issues/2783); + fence compiled a feature out of source builds against an older header, and the + run-time checks never noticed. See https://github.com/NVIDIA/cuda-python/issues/2783. versions.hpp checks the header once and is the only file allowed to name it.""" cpp = CORE / "_cpp" files = sorted(p for p in cpp.rglob("*") if p.suffix in (".hpp", ".h", ".cpp")) @@ -114,9 +114,9 @@ def test_cuda_version_is_named_only_in_versions_hpp(): @pytest.mark.agent_authored(model="claude-fable-5-1") def test_driver_function_table_matches_the_cuda_bindings_loader(): - """driver_api.hpp lists each driver function with the CUDA version cuda-bindings - requests it at; that number decides which functions every supported driver - must provide. Check it against the loader cuda-bindings generates.""" + """driver_api.hpp lists each driver function with the CUDA version that cuda-bindings + requests it at. That number decides which functions every supported driver must + provide. Check it against the loader that cuda-bindings generates.""" loader = CORE.parents[2] / "cuda_bindings" / "cuda" / "bindings" / "_internal" / "driver_linux.pyx" if not loader.is_file(): pytest.skip("cuda-bindings source is not next to cuda_core") @@ -136,10 +136,10 @@ def test_driver_function_table_matches_the_cuda_bindings_loader(): @pytest.mark.agent_authored(model="claude-fable-5-1") def test_driver_calls_go_through_the_table(): - """Every driver call uses DRIVER_CALL (or a pw_ wrapper), which resolves the + """Every driver call uses DRIVER_CALL or a pw_ wrapper, which resolves the table on first use and never dereferences null. The only raw p_ calls are - the table's own machinery and the sites under ipc_import_mutex, where the - table is resolved before the lock and marked `// raw:`.""" + the table's own machinery and the sites under ipc_import_mutex. Those sites + resolve the table before the lock and carry the `// raw:` mark.""" machinery = {"driver_api.hpp", "driver_api.cpp", "py_driver_fns.cpp", "internal.hpp"} raw_call = re.compile(r"\bp_(cu|nv)\w+\b") # calls and null checks alike offenders = [] diff --git a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py index 0193c6ca003..1db4d210693 100644 --- a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py +++ b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py @@ -30,13 +30,16 @@ def hardware_supports_nvml(): def _should_skip_nvml_tests() -> bool: - """Return True if NVML tests should be skipped on this system (a hardware-level NVML probe).""" + """Return True if the NVML tests should skip on this system. + + The check is a hardware-level NVML probe. + """ return not hardware_supports_nvml() skip_if_nvml_unsupported = pytest.mark.skipif( _should_skip_nvml_tests(), - reason="NVML support requires hardware that supports NVML", + reason="this hardware does not support NVML", ) diff --git a/toolshed/check_cuda_core_bindings_floor.py b/toolshed/check_cuda_core_bindings_floor.py index 2e43c5b7d52..929c4ae5bbd 100644 --- a/toolshed/check_cuda_core_bindings_floor.py +++ b/toolshed/check_cuda_core_bindings_floor.py @@ -5,17 +5,18 @@ The floor of each CUDA major is declared once, by the `cu12`/`cu13` extras in cuda_core/pyproject.toml (see cuda_core/cuda/core/_bindings_floor.py). Most -consumers read it from there, but two constraints cannot be derived and are -checked here, as a pre-commit hook and from cuda_core/tests/test_bindings_floor.py: +consumers read it from there. These constraints cannot be derived from it, so +this script checks them, as a pre-commit hook and from +cuda_core/tests/test_bindings_floor.py: 1. The extras parse: each `cu` extra pins exactly `cuda-bindings[...]>=..,<`. 2. ci/versions.yml builds each major against a CUDA Toolkit of at least the floor's major.minor. A toolkit below the floor's minor cannot build the - floor's cuda-bindings (the build requires the header cuda-bindings was - generated from); a toolkit ahead of the floor is the toolkit-bump window - described in cuda_core/AGENTS.md. -3. No documentation page spells a floor out by hand; docs/source/conf.py + floor's cuda-bindings, because the build requires the header that + cuda-bindings was generated from. A toolkit ahead of the floor is the + toolkit-bump window described in cuda_core/AGENTS.md. +3. No documentation page spells a floor out by hand. docs/source/conf.py provides |cuda-bindings-floor-cu12| and |cuda-bindings-floor-cu13|. Release notes are history and are exempt. @@ -55,14 +56,14 @@ def load_floor_module(repo_root: Path): def read_floors(repo_root: Path) -> dict[int, tuple[int, int, int]]: - """The floors declared by the pyproject extras; ValueError if they do not parse.""" + """The floors declared by the pyproject extras. Raises ValueError if they do not parse.""" with open(repo_root / PYPROJECT, "rb") as f: extras = tomllib.load(f)["project"]["optional-dependencies"] return load_floor_module(repo_root).floors_from_extras(extras) def ci_pin_problems(floors: dict[int, tuple[int, int, int]], versions_yml: str) -> list[str]: - """Toolkit pins in ci/versions.yml whose major.minor is not the floor's.""" + """Toolkit pins in ci/versions.yml that sit below the floor's major.minor.""" pins = {int(major): (int(major), int(minor), key) for key, major, minor in _CI_PIN_RE.findall(versions_yml)} problems = [] for major, floor in floors.items(): @@ -73,8 +74,8 @@ def ci_pin_problems(floors: dict[int, tuple[int, int, int]], versions_yml: str) if pinned_major != floor[0] or pinned_minor < floor[1]: problems.append( f"{CI_VERSIONS}: cuda.{key}.version is {pinned_major}.{pinned_minor} but the CUDA {major} floor " - f"is cuda-bindings {'.'.join(map(str, floor))}; the toolkit must not sit below the floor's " - "major.minor (ahead of it is the toolkit-bump window, see cuda_core/AGENTS.md)" + f"is cuda-bindings {'.'.join(map(str, floor))}. The toolkit must not sit below the floor's " + "major.minor. A toolkit ahead of the floor is the toolkit-bump window. See cuda_core/AGENTS.md" ) for major in pins: if major not in floors: @@ -83,7 +84,7 @@ def ci_pin_problems(floors: dict[int, tuple[int, int, int]], versions_yml: str) def docs_problems(repo_root: Path) -> list[str]: - """Documentation pages (release notes excepted) that spell out a floor by hand.""" + """Documentation pages that spell out a floor by hand. Release notes are exempt.""" problems = [] for path in sorted((repo_root / DOCS_SOURCE).rglob("*.rst")): if "release" in path.relative_to(repo_root / DOCS_SOURCE).parts: @@ -91,8 +92,8 @@ def docs_problems(repo_root: Path) -> list[str]: for number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1): if _HAND_WRITTEN_FLOOR_RE.search(line): problems.append( - f"{path.relative_to(repo_root).as_posix()}:{number}: spells out a cuda-bindings floor; " - "use the |cuda-bindings-floor-cu| substitution from conf.py" + f"{path.relative_to(repo_root).as_posix()}:{number}: spells out a cuda-bindings floor. " + "Use the |cuda-bindings-floor-cu| substitution from conf.py" ) return problems From 5b087f43ed24f828693a86c4fdd8231b2e35092a Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Wed, 30 Sep 2026 07:19:35 -0700 Subject: [PATCH 15/18] cuda-bindings header check: review wording for the run-time note (#2783) Per review: "supports any CUDA 13.x toolkit" overstates the claim, and "this cuda-bindings build" names what the note is about. Co-Authored-By: Claude Fable 5.1 --- cuda_bindings/build_hooks.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/cuda_bindings/build_hooks.py b/cuda_bindings/build_hooks.py index 8660522c5e2..715c9493825 100644 --- a/cuda_bindings/build_hooks.py +++ b/cuda_bindings/build_hooks.py @@ -151,9 +151,9 @@ def _check_cuda_headers(cuda_path: str) -> None: if found != needed: raise RuntimeError( f"This cuda-bindings source tree needs CUDA {needed} headers, but {_cuda_h_path(cuda_path)} is " - f"CUDA {found}. This is a build-time requirement only: at run time cuda-bindings supports any " - f"CUDA {generated // 1000}.x toolkit, see {_INSTALL_URL}. Point CUDA_PATH or CUDA_HOME at a " - f"CUDA {needed} toolkit, or build from cuda-bindings {found}.x sources." + f"CUDA {found}. This is a build-time requirement only: at run time this cuda-bindings build can be " + f"used with any CUDA {generated // 1000}.x toolkit, see {_INSTALL_URL}. Point CUDA_PATH or CUDA_HOME " + f"at a CUDA {needed} toolkit, or build from cuda-bindings {found}.x sources." ) From 100cb8eaa0baeba3d8f83773b14fa3183c925533 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Wed, 30 Sep 2026 14:21:08 -0700 Subject: [PATCH 16/18] Compile the build record; verify the compiler's cuda.h against the Python read; review follow-ups (#2783) - cuda/core/_build_info.pyx replaces the generated _build_info.py. It exposes CUDA_VERSION from the cuda.h the compiler resolved and the major and floor from the Cython compile-time environment, so the record cannot disagree with the binaries. cuda/core/__init__.py reads the extension. - build_hooks.py passes the cuda-bindings header version as a third macro; versions.hpp fails the compile unless the resolved cuda.h has the same major.minor. The Python read of cuda.h stays for the decisions that precede any compile, and can no longer silently differ from the compiler's. - ci/tools/cuda_core_bindings_floor.py reads the floor from the wheel's METADATA (the cu extra) instead of parsing a generated module. - __init__.py imports _bindings_floor directly; the merge tool keeps it at the top level. A missing _build_info now distinguishes "no build for this CUDA major" from "not built". - ci/tools/env-vars reads ci/versions.yml with yq. - cuda_python/setup.py no longer pins cuda-core. Co-Authored-By: Claude Fable 5.1 --- .gitignore | 2 - ci/tools/cuda_core_bindings_floor.py | 57 ++++++------ ci/tools/env-vars | 6 +- ci/tools/merge_cuda_core_wheels.py | 9 +- .../tests/test_cuda_core_bindings_floor.py | 89 +++++++++++-------- cuda_core/build_hooks.py | 62 ++++++------- cuda_core/cuda/core/__init__.py | 40 ++++++--- cuda_core/cuda/core/_bindings_floor.py | 7 +- cuda_core/cuda/core/_build_info.pyi | 13 +++ cuda_core/cuda/core/_build_info.pyx | 19 ++++ cuda_core/cuda/core/_cpp/rt/DESIGN.md | 13 ++- cuda_core/cuda/core/_cpp/rt/versions.hpp | 18 +++- cuda_core/tests/test_bindings_floor.py | 18 ++++ cuda_core/tests/test_build_hooks.py | 55 +++++------- cuda_python/setup.py | 5 +- 15 files changed, 253 insertions(+), 160 deletions(-) create mode 100644 cuda_core/cuda/core/_build_info.pyi create mode 100644 cuda_core/cuda/core/_build_info.pyx diff --git a/.gitignore b/.gitignore index 96a059748ed..5616c05cc44 100644 --- a/.gitignore +++ b/.gitignore @@ -41,8 +41,6 @@ cuda_bindings/cuda/bindings/utils/_get_handle.pyx # Version files from setuptools_scm _version.py -# Generated by cuda_core/build_hooks.py at build time (see cuda/core/__init__.py). -cuda_core/cuda/core/_build_info.py # Distribution / packaging .Python diff --git a/ci/tools/cuda_core_bindings_floor.py b/ci/tools/cuda_core_bindings_floor.py index 63fc78453d9..5f4dadc8cf0 100644 --- a/ci/tools/cuda_core_bindings_floor.py +++ b/ci/tools/cuda_core_bindings_floor.py @@ -14,50 +14,51 @@ from the checkout, so a nightly job that tests a wheel built from another commit reads that wheel's floor. -Each build records its floor in the generated cuda/core/_build_info.py. A -single-major build places the file at top level. The merged wheel places it -under cuda/core/cu/. This script parses the CUDA_BINDINGS_FLOOR literal -out of it and never runs it. +The floors are declared once, by the `cu12`/`cu13` extras in +cuda_core/pyproject.toml, and the wheel's METADATA carries them as +`Requires-Dist: cuda-bindings[all]<14,>=13.4.1; extra == "cu13"`. This script +reads that line. """ from __future__ import annotations import argparse -import ast +import re import sys import zipfile from pathlib import Path -MODULE = "_build_info.py" +# The cuda-bindings requirement of one `cu` extra in METADATA. The name +# is followed by `[`, an operator or whitespace so that other names do not match. +_REQUIRES_DIST_RE = re.compile( + r"^Requires-Dist:\s*cuda-bindings(?=[\[<>=!~\s])(?P[^;]*);\s*extra\s*==\s*['\"]cu(?P\d+)['\"]" +) +_FLOOR_RE = re.compile(r">=\s*(\d+)\.(\d+)\.(\d+)") -def _literal(source: str, name: str): - """The literal that `source` assigns to `name` at module level. Parses the source and never runs it.""" - for node in ast.parse(source, MODULE).body: - if isinstance(node, ast.AnnAssign): - targets = [node.target] - elif isinstance(node, ast.Assign): - targets = node.targets - else: +def floor_from_metadata(metadata: str, major: int) -> str: + """The floor that the `cu` extra of a wheel's METADATA pins cuda-bindings to.""" + for line in metadata.splitlines(): + m = _REQUIRES_DIST_RE.match(line) + if m is None or int(m.group("major")) != major: continue - if node.value is not None and any(isinstance(t, ast.Name) and t.id == name for t in targets): - return ast.literal_eval(node.value) - raise SystemExit(f"{MODULE} does not assign {name}") - - -def floor_from_source(source: str, major: int) -> str: - if _literal(source, "CUDA_MAJOR") != major: - raise SystemExit(f"{MODULE} records a CUDA {_literal(source, 'CUDA_MAJOR')} build, not CUDA {major}") - return ".".join(str(part) for part in _literal(source, "CUDA_BINDINGS_FLOOR")) + floor = _FLOOR_RE.search(m.group("spec")) + if floor is None: + raise SystemExit(f"METADATA pins cuda-bindings without a floor for the cu{major} extra: {line.strip()!r}") + if int(floor.group(1)) != major: + raise SystemExit( + f"METADATA pins a cuda-bindings floor of another major for the cu{major} extra: {line.strip()!r}" + ) + return ".".join(floor.groups()) + raise SystemExit(f"METADATA declares no cuda-bindings requirement for the cu{major} extra") def floor_from_wheel(wheel: Path, major: int) -> str: with zipfile.ZipFile(wheel) as zf: - names = set(zf.namelist()) - for candidate in (f"cuda/core/cu{major}/{MODULE}", f"cuda/core/{MODULE}"): - if candidate in names: - return floor_from_source(zf.read(candidate).decode("utf-8"), major) - raise SystemExit(f"{wheel.name} contains no build for CUDA {major}: it has no {MODULE}. Is it a cuda-core wheel?") + metadata = [name for name in zf.namelist() if name.endswith(".dist-info/METADATA")] + if len(metadata) != 1: + raise SystemExit(f"{wheel.name} contains {len(metadata)} METADATA files, expected 1. Is it a wheel?") + return floor_from_metadata(zf.read(metadata[0]).decode("utf-8"), major) def main(argv: list[str] | None = None) -> int: diff --git a/ci/tools/env-vars b/ci/tools/env-vars index 4b71d4c9082..9e55a988487 100755 --- a/ci/tools/env-vars +++ b/ci/tools/env-vars @@ -81,7 +81,11 @@ elif [[ "${1}" == "test" ]]; then TEST_CUDA_MINOR="$(cut -d '.' -f 2 <<< ${CUDA_VER})" # CI builds the prior-major half of the cuda-core wheel against the prev_build # toolkit in ci/versions.yml and the backport branch's bindings. - BUILD_PREV_CUDA_VER="$(sed -n '/prev_build:/,/version:/s/.*version: *"\([^"]*\)".*/\1/p' ci/versions.yml)" + if ! command -v yq >/dev/null 2>&1; then + echo "Error: yq is required to read ci/versions.yml" >&2 + exit 1 + fi + BUILD_PREV_CUDA_VER="$(yq -r '.cuda.prev_build.version' ci/versions.yml)" BUILD_PREV_CUDA_MINOR="$(cut -d '.' -f 2 <<< ${BUILD_PREV_CUDA_VER})" if [[ ${BUILD_CUDA_MAJOR} != ${TEST_CUDA_MAJOR} ]]; then diff --git a/ci/tools/merge_cuda_core_wheels.py b/ci/tools/merge_cuda_core_wheels.py index 6e7f51e73b6..374ea1f53ea 100644 --- a/ci/tools/merge_cuda_core_wheels.py +++ b/ci/tools/merge_cuda_core_wheels.py @@ -148,10 +148,11 @@ def merge_wheels(wheels: list[Path], output_dir: Path, show_wheel_contents: bool print("\n=== Removing files from cuda/core/ directory ===", file=sys.stderr) # Only what cuda/core/__init__.py uses before it rewrites __path__ to the - # versioned subpackage stays at top level: it imports _version, then - # redirects every later import into the versioned tree. Anything else - # left at top level is a dead copy that nothing imports. - items_to_keep = {"__init__.py", "_version.py", *versioned_dirs} + # versioned subpackage stays at top level: it imports _version and the + # build-independent floor logic in _bindings_floor, then redirects every + # later import into the versioned tree. Anything else left at top level + # is a dead copy that nothing imports. + items_to_keep = {"__init__.py", "_version.py", "_bindings_floor.py", *versioned_dirs} all_items = os.scandir(base_wheel / base_dir) removed_count = 0 for f in all_items: diff --git a/ci/tools/tests/test_cuda_core_bindings_floor.py b/ci/tools/tests/test_cuda_core_bindings_floor.py index 487e31418a1..5f4b18dc97d 100644 --- a/ci/tools/tests/test_cuda_core_bindings_floor.py +++ b/ci/tools/tests/test_cuda_core_bindings_floor.py @@ -20,63 +20,80 @@ def _load(name, path): tool = _load("cuda_core_bindings_floor", TOOLS / "cuda_core_bindings_floor.py") -FLOORS = {12: (12, 9, 8), 13: (13, 4, 1)} +# METADATA as setuptools writes it: pip's normalized specifier order, extras in quotes. +METADATA = """\ +Metadata-Version: 2.4 +Name: cuda-core +Version: 1.3.0 +Requires-Dist: cuda-pathfinder>=1.4.2 +Requires-Dist: numpy +Provides-Extra: cu12 +Requires-Dist: cuda-bindings[all]<13,>=12.9.8; extra == "cu12" +Requires-Dist: cuda-toolkit==12.*; extra == "cu12" +Provides-Extra: cu13 +Requires-Dist: cuda-bindings[all]<14,>=13.4.1; extra == "cu13" +Requires-Dist: cuda-toolkit==13.*; extra == "cu13" +""" + + +def _wheel(tmp_path, metadata=METADATA): + path = tmp_path / "cuda_core-1.3.0-cp312-cp312-linux_x86_64.whl" + with zipfile.ZipFile(path, "w") as zf: + zf.writestr("cuda/core/__init__.py", "") + if metadata is not None: + zf.writestr("cuda_core-1.3.0.dist-info/METADATA", metadata) + return path -def _build_info(major): - floor = FLOORS[major] - return ( - "# Generated by build_hooks.py at build time. Do not edit or commit.\n" - f"CUDA_MAJOR = {major}\n" - f"CUDA_VERSION = {major * 1000 + floor[1] * 10} # the cuda.h this build compiled against\n" - f"CUDA_BINDINGS_FLOOR = {floor!r}\n" - f"CUDA_BINDINGS_BUILD_VERSION = '{major}.{floor[1]}.{floor[2]}'\n" - ) +@pytest.mark.agent_authored(model="claude-fable-5-1") +@pytest.mark.parametrize(("major", "floor"), [(12, "12.9.8"), (13, "13.4.1")]) +def test_reads_the_floor_of_each_extra(tmp_path, major, floor): + assert tool.floor_from_wheel(_wheel(tmp_path), major) == floor -def _wheel(tmp_path, entries): - path = tmp_path / "cuda_core-1.3.0-cp312-cp312-linux_x86_64.whl" - with zipfile.ZipFile(path, "w") as zf: - for name, major in entries.items(): - zf.writestr(name, _build_info(major)) - return path +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_accepts_other_spellings_of_the_requirement(): + metadata = 'Requires-Dist: cuda-bindings>=13.4.1,==13.*; extra == "cu13"\n' + assert tool.floor_from_metadata(metadata, 13) == "13.4.1" + metadata = "Requires-Dist: cuda-bindings [all] >= 13.4.1, < 14 ; extra == 'cu13'\n" + assert tool.floor_from_metadata(metadata, 13) == "13.4.1" @pytest.mark.agent_authored(model="claude-fable-5-1") -@pytest.mark.parametrize("major", [12, 13]) -def test_reads_the_merged_wheel_layout(tmp_path, major): - wheel = _wheel(tmp_path, {"cuda/core/cu12/_build_info.py": 12, "cuda/core/cu13/_build_info.py": 13}) - assert tool.floor_from_wheel(wheel, major) == ".".join(map(str, FLOORS[major])) +def test_ignores_lookalike_names_and_other_extras(): + metadata = ( + 'Requires-Dist: cuda-bindings-extra>=1.0; extra == "cu13"\n' + 'Requires-Dist: cuda-bindings[all]<13,>=12.9.8; extra == "cu12"\n' + 'Requires-Dist: cuda-bindings[all]<14,>=13.4.1; extra == "cu13"\n' + ) + assert tool.floor_from_metadata(metadata, 13) == "13.4.1" @pytest.mark.agent_authored(model="claude-fable-5-1") -def test_reads_a_single_major_wheel(tmp_path): - wheel = _wheel(tmp_path, {"cuda/core/_build_info.py": 13}) - assert tool.floor_from_wheel(wheel, 13) == "13.4.1" +def test_rejects_a_missing_extra(tmp_path): + with pytest.raises(SystemExit, match="no cuda-bindings requirement for the cu11 extra"): + tool.floor_from_wheel(_wheel(tmp_path), 11) @pytest.mark.agent_authored(model="claude-fable-5-1") -def test_rejects_a_single_major_wheel_of_another_major(tmp_path): - wheel = _wheel(tmp_path, {"cuda/core/_build_info.py": 13}) - with pytest.raises(SystemExit, match="records a CUDA 13 build, not CUDA 12"): - tool.floor_from_wheel(wheel, 12) +def test_rejects_a_requirement_without_a_floor(): + with pytest.raises(SystemExit, match="without a floor for the cu13 extra"): + tool.floor_from_metadata('Requires-Dist: cuda-bindings==13.*; extra == "cu13"\n', 13) @pytest.mark.agent_authored(model="claude-fable-5-1") -def test_rejects_a_wheel_without_the_build_record(tmp_path): - wheel = _wheel(tmp_path, {}) - with pytest.raises(SystemExit, match="contains no build for CUDA 13"): - tool.floor_from_wheel(wheel, 13) +def test_rejects_a_floor_of_another_major(): + with pytest.raises(SystemExit, match="floor of another major for the cu13 extra"): + tool.floor_from_metadata('Requires-Dist: cuda-bindings>=12.9.8,<14; extra == "cu13"\n', 13) @pytest.mark.agent_authored(model="claude-fable-5-1") -def test_rejects_a_build_record_without_the_floor(): - with pytest.raises(SystemExit, match="does not assign CUDA_BINDINGS_FLOOR"): - tool.floor_from_source("CUDA_MAJOR = 13\n", 13) +def test_rejects_a_zip_that_is_not_a_wheel(tmp_path): + with pytest.raises(SystemExit, match="contains 0 METADATA files"): + tool.floor_from_wheel(_wheel(tmp_path, metadata=None), 13) @pytest.mark.agent_authored(model="claude-fable-5-1") def test_cli_prints_the_floor(tmp_path, capsys): - wheel = _wheel(tmp_path, {"cuda/core/_build_info.py": 13}) - assert tool.main(["--wheel", str(wheel), "--major", "13"]) == 0 + assert tool.main(["--wheel", str(_wheel(tmp_path)), "--major", "13"]) == 0 assert capsys.readouterr().out.strip() == "13.4.1" diff --git a/cuda_core/build_hooks.py b/cuda_core/build_hooks.py index fd7d942e090..f1285521d9a 100644 --- a/cuda_core/build_hooks.py +++ b/cuda_core/build_hooks.py @@ -133,11 +133,6 @@ def _get_cuda_path() -> str: _PACKAGE_DIR = Path(__file__).parent / "cuda" / "core" - -# Generated at build time by _write_build_info() and read by cuda/core/__init__.py. -_BUILD_INFO_PATH = _PACKAGE_DIR / "_build_info.py" - - _PYPROJECT_PATH = Path(__file__).parent / "pyproject.toml" @@ -495,8 +490,11 @@ def _determine_cuda_major_version() -> str: return cuda_major -def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: - """Reject build configurations that cuda.core does not support, then record the build. +def _check_build_configuration(cuda_path: str, cuda_major: str) -> int: + """Reject build configurations that cuda.core does not support. + + Returns the ``CUDA_VERSION`` of the header that the installed cuda-bindings + was generated from, for _build_define_macros(). cuda.core supports one configuration per CUDA major series. The installed cuda-bindings is at least the floor of the series, the cu extra in @@ -553,35 +551,26 @@ def _check_build_configuration(cuda_path: str, cuda_major: str) -> None: "isolated build, constrain cuda-bindings with PIP_CONSTRAINT or build with --no-build-isolation." ) print(f"Build configuration: CUDA {header[0]}.{header[1]} headers, cuda-bindings {bindings_version}") - _write_build_info(major, cuda_version, floor_triple, bindings_version) + return bindings_cuda_version + +def _build_define_macros(cuda_major: str, bindings_cuda_version: int | None = None) -> list: + """Preprocessor macros that carry the build decision into the C++. See _cpp/rt/versions.hpp. -def _build_define_macros(cuda_major: str) -> list: - """Preprocessor macros that carry the build decision into the C++. See _cpp/rt/versions.hpp.""" + ``bindings_cuda_version`` is the header that the installed cuda-bindings was + generated from, as _check_build_configuration() returns it. versions.hpp + fails the compile unless the ``cuda.h`` the compiler resolved has the same + major.minor, so the Python read of ``cuda.h`` can never silently disagree + with the compiler's. + """ major = int(cuda_major) - return [ + macros = [ ("CUDA_CORE_BUILD_MAJOR", str(major)), ("CUDA_CORE_MIN_CUDA_VERSION", str(_load_bindings_floor().cuda_version_of(_floor_for(major)))), ] - - -def _write_build_info(cuda_major: int, cuda_version: int, floor: tuple, bindings_version: str) -> None: - """Record what this build compiled against, for the import-time check. - - cuda/core/__init__.py reads this module before it selects the versioned - subpackage. It refuses an installed cuda-bindings that is older than the - floor or generated from an older header minor than the build's. See - _bindings_floor.check_installed_bindings. Like _version.py, the file is - generated, gitignored, and shipped. - """ - _BUILD_INFO_PATH.write_text( - "# Generated by build_hooks.py at build time. Do not edit or commit.\n" - f"CUDA_MAJOR = {cuda_major}\n" - f"CUDA_VERSION = {cuda_version} # the cuda.h this build compiled against\n" - f"CUDA_BINDINGS_FLOOR = {tuple(floor)!r}\n" - f"CUDA_BINDINGS_BUILD_VERSION = {bindings_version!r} # informational, not read at import\n", - encoding="utf-8", - ) + if bindings_cuda_version is not None: + macros.append(("CUDA_CORE_BINDINGS_CUDA_VERSION", str(bindings_cuda_version))) + return macros # used later by setup() @@ -765,7 +754,7 @@ def module_names(): # _get_cuda_path() and reads cuda.h, which must not run before the # pathfinder import has repaired PEP 517 namespace shadowing. cuda_major, config_key = _check_build_config(toolchain, debug, COMPILE_FOR_COVERAGE) - _check_build_configuration(cuda_path, cuda_major) + bindings_cuda_version = _check_build_configuration(cuda_path, cuda_major) depends = _extension_depends() ext_modules = tuple( @@ -779,8 +768,8 @@ def module_names(): ] + all_include_dirs, # The C++ branches on the CUDA major series only. _cpp/rt/versions.hpp - # re-checks cuda.h against both macros. See _check_build_configuration. - define_macros=_build_define_macros(cuda_major), + # checks the cuda.h the compiler resolved against these macros. + define_macros=_build_define_macros(cuda_major, bindings_cuda_version), language="c++", extra_compile_args=extra_compile_args, extra_link_args=extra_link_args, @@ -789,7 +778,12 @@ def module_names(): ) nthreads = int(os.environ.get("CUDA_PYTHON_PARALLEL_LEVEL", os.cpu_count() // 2)) - compile_time_env = {"CUDA_CORE_BUILD_MAJOR": int(cuda_major)} + # cuda/core/_build_info.pyx records both, next to the CUDA_VERSION of the + # cuda.h the compiler resolved; cuda/core/__init__.py reads them at import. + compile_time_env = { + "CUDA_CORE_BUILD_MAJOR": int(cuda_major), + "CUDA_CORE_BINDINGS_FLOOR": tuple(_floor_for(cuda_major)), + } compiler_directives = {"embedsignature": True, "warn.deprecated.IF": False, "freethreading_compatible": True} _CythonOptions.warning_errors = True if COMPILE_FOR_COVERAGE: diff --git a/cuda_core/cuda/core/__init__.py b/cuda_core/cuda/core/__init__.py index b1221419428..06f7ad79bfe 100644 --- a/cuda_core/cuda/core/__init__.py +++ b/cuda_core/cuda/core/__init__.py @@ -11,8 +11,8 @@ def _import_versioned_module() -> None: The published wheel carries one build per CUDA major series, as the subpackages ``cuda.core.cu12`` and ``cuda.core.cu13``. A conda or local build carries one build at the top level. Each build records the CUDA - header that it compiled against and its cuda-bindings floor in - ``_build_info``, which build_hooks.py generates. The installed cuda-bindings + header that it compiled against and its cuda-bindings floor in the + ``_build_info`` extension module. The installed cuda-bindings must be of the build's major, at least as new as the floor, and generated from a ``cuda.h`` at least as new as the build's. See ``_bindings_floor.check_installed_bindings``. If it is not, import fails @@ -20,6 +20,9 @@ def _import_versioned_module() -> None: or a silently disabled feature. """ import importlib + import os + + from cuda.core import _bindings_floor try: from cuda import bindings @@ -28,14 +31,31 @@ def _import_versioned_module() -> None: raise ImportError("cuda.core requires cuda-bindings. Install cuda-core[cu12] or cuda-core[cu13]") from None raise - def load_build_module(name: str, cuda_major: int): + def load_build_info(cuda_major: int): # Prefer this major's build in the merged wheel, then fall back to a plain build. try: - return importlib.import_module(f".cu{cuda_major}.{name}", __package__) + return importlib.import_module(f".cu{cuda_major}._build_info", __package__) except ModuleNotFoundError as exc: if exc.name != f"{__package__}.cu{cuda_major}": raise - return importlib.import_module(f".{name}", __package__) + return importlib.import_module("._build_info", __package__) + + def missing_build_message(cuda_major: int) -> str: + here = os.path.dirname(__file__) + builds = sorted( + int(name[2:]) + for name in os.listdir(here) + if name.startswith("cu") and name[2:].isdigit() and os.path.isdir(os.path.join(here, name)) + ) + if builds: + return ( + f"This cuda.core installation has builds for CUDA {' and '.join(map(str, builds))}, not for " + f"CUDA {cuda_major}. The installed cuda-bindings is {version_str}." + ) + return ( + "This cuda.core has not been built: cuda/core/_build_info is missing. " + "Build it with pip install, or reinstall the package." + ) version_str = bindings.__version__ # The major decides which build to consult. _bindings_floor validates everything else. @@ -46,16 +66,12 @@ def load_build_module(name: str, cuda_major: int): f"cuda.core requires a cuda-bindings release, but the installed cuda-bindings version is {version_str!r}" ) from None try: - floor = load_build_module("_bindings_floor", cuda_major) - info = load_build_module("_build_info", cuda_major) + info = load_build_info(cuda_major) except ModuleNotFoundError as exc: - raise ImportError( - f"This cuda.core installation has no build for CUDA {cuda_major}. " - f"The installed cuda-bindings is {version_str}." - ) from exc + raise ImportError(missing_build_message(cuda_major)) from exc # By module object: the `cuda` namespace package need not carry a `bindings` attribute. bindings_driver = importlib.import_module("cuda.bindings.driver") - floor.check_installed_bindings( + _bindings_floor.check_installed_bindings( version_str, int(bindings_driver.CUDA_VERSION), info.CUDA_MAJOR, diff --git a/cuda_core/cuda/core/_bindings_floor.py b/cuda_core/cuda/core/_bindings_floor.py index c8c887490ea..bc915916880 100644 --- a/cuda_core/cuda/core/_bindings_floor.py +++ b/cuda_core/cuda/core/_bindings_floor.py @@ -17,8 +17,9 @@ - The build backend, ``build_hooks.py``, checks the installed cuda-bindings against the floor. It checks that the ``cuda.h`` that it compiles against is - the one that cuda-bindings was generated from. It records the floor and the - header in the generated ``_build_info.py``. + the one that cuda-bindings was generated from. It passes the floor to the + ``_build_info`` extension module, which records it next to the header the + compiler used. - ``cuda/core/__init__.py`` checks the installed cuda-bindings against that record with :func:`check_installed_bindings`. - The documentation reads the floors into substitutions in ``docs/source/conf.py``. @@ -164,7 +165,7 @@ def check_installed_bindings( and ``driver.CUDA_VERSION`` of the installed cuda-bindings. The latter is the ``cuda.h`` that it was generated from, for example 13040. ``build_cuda_major``, ``build_cuda_version`` and ``build_floor`` are the - build's record in ``_build_info.py``. + build's record in the ``_build_info`` extension module. Raises ImportError with an actionable message unless the installed cuda-bindings passes all of these checks: diff --git a/cuda_core/cuda/core/_build_info.pyi b/cuda_core/cuda/core/_build_info.pyi new file mode 100644 index 00000000000..9ddb1ec2ffb --- /dev/null +++ b/cuda_core/cuda/core/_build_info.pyi @@ -0,0 +1,13 @@ +# This file was generated by stubgen-pyx v0.2.22 from cuda_core/cuda/core/_build_info.pyx + +"""What this build of cuda.core compiled against. + +``CUDA_MAJOR`` and ``CUDA_BINDINGS_FLOOR`` come from the compile-time environment +that build_hooks.py sets. ``CUDA_VERSION`` is the ``CUDA_VERSION`` macro of the +``cuda.h`` that the compiler resolved when it built this module, so the record +cannot disagree with the binaries. ``cuda/core/__init__.py`` reads this module +before it selects the build. The module imports nothing. +""" +CUDA_MAJOR: int = ... +CUDA_VERSION: int = ... +CUDA_BINDINGS_FLOOR: tuple[int, int, int] = ... diff --git a/cuda_core/cuda/core/_build_info.pyx b/cuda_core/cuda/core/_build_info.pyx new file mode 100644 index 00000000000..e1d3c7d36e7 --- /dev/null +++ b/cuda_core/cuda/core/_build_info.pyx @@ -0,0 +1,19 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# SPDX-License-Identifier: Apache-2.0 + +"""What this build of cuda.core compiled against. + +``CUDA_MAJOR`` and ``CUDA_BINDINGS_FLOOR`` come from the compile-time environment +that build_hooks.py sets. ``CUDA_VERSION`` is the ``CUDA_VERSION`` macro of the +``cuda.h`` that the compiler resolved when it built this module, so the record +cannot disagree with the binaries. ``cuda/core/__init__.py`` reads this module +before it selects the build. The module imports nothing. +""" + +cdef extern from "cuda.h": + enum: CUDA_H_VERSION "CUDA_VERSION" + +CUDA_MAJOR: int = CUDA_CORE_BUILD_MAJOR +CUDA_VERSION: int = CUDA_H_VERSION +CUDA_BINDINGS_FLOOR: tuple[int, int, int] = CUDA_CORE_BINDINGS_FLOOR diff --git a/cuda_core/cuda/core/_cpp/rt/DESIGN.md b/cuda_core/cuda/core/_cpp/rt/DESIGN.md index 36b238a9e8a..07cb17e2bd4 100644 --- a/cuda_core/cuda/core/_cpp/rt/DESIGN.md +++ b/cuda_core/cuda/core/_cpp/rt/DESIGN.md @@ -242,9 +242,16 @@ cuda.core supports one build configuration per CUDA major series. The `cuda.h` it compiles against has the same major.minor as the cuda-bindings it is built with, and that cuda-bindings is at or above the series' floor (`cuda/core/_bindings_floor.py`). `build_hooks.py` enforces both before -compilation and defines `CUDA_CORE_BUILD_MAJOR` and -`CUDA_CORE_MIN_CUDA_VERSION` for the C++ compiler. `versions.hpp`, the first -include of the tree, re-checks `cuda.h` against them with `#error`. +compilation and defines three macros for the C++ compiler: +`CUDA_CORE_BUILD_MAJOR`, `CUDA_CORE_MIN_CUDA_VERSION` (the floor's header) and +`CUDA_CORE_BINDINGS_CUDA_VERSION` (the header that the installed cuda-bindings +was generated from). `versions.hpp`, the first include of the tree, re-checks +the `cuda.h` that the compiler resolved against all three with `#error`, so +the compiler's header can never silently differ from the one `build_hooks.py` +read. The `cuda.core._build_info` extension records `CUDA_VERSION` from that +same `cuda.h`, with the major and the floor from the Cython compile-time +environment; `cuda/core/__init__.py` checks the installed cuda-bindings +against that record at import. The C++ branches on `CUDA_CORE_BUILD_MAJOR` only, and only where the two major series differ. Minor-version fences (`#if CUDA_VERSION >= 130x0`) are not diff --git a/cuda_core/cuda/core/_cpp/rt/versions.hpp b/cuda_core/cuda/core/_cpp/rt/versions.hpp index 5f37c0033ad..88dbd8f4e16 100644 --- a/cuda_core/cuda/core/_cpp/rt/versions.hpp +++ b/cuda_core/cuda/core/_cpp/rt/versions.hpp @@ -11,7 +11,7 @@ // present at build time, and that cuda-bindings is at or above the series' // floor (see cuda/core/_bindings_floor.py and // https://github.com/NVIDIA/cuda-python/issues/2783). build_hooks.py enforces -// both before it compiles and passes the decision down as two macros: +// both before it compiles and passes the decision down as three macros: // // CUDA_CORE_BUILD_MAJOR the CUDA major series of the build, 12 or 13. // The only version the C++ may branch on, as @@ -19,9 +19,15 @@ // for a difference between major series. // CUDA_CORE_MIN_CUDA_VERSION the floor's major.minor as a CUDA_VERSION // value, for example 13040. +// CUDA_CORE_BINDINGS_CUDA_VERSION +// the CUDA_VERSION of the header that the +// installed cuda-bindings was generated from. // -// This file checks cuda.h against both macros again, so that a build that -// bypasses build_hooks.py still cannot compile against an unsupported header. +// This file checks cuda.h against all three macros again, so that a build that +// bypasses build_hooks.py still cannot compile against an unsupported header, +// and so that the cuda.h the compiler resolves is the one build_hooks.py read. +// The driver entry-point table is keyed by that header's macros, so cuda.h must +// have the major.minor of the header cuda-bindings was generated from. // No other file may use a minor-version fence such as // `#if CUDA_VERSION >= 130x0`. Such fences compiled features out of source // builds against an older header, and the run-time checks, which looked at @@ -46,3 +52,9 @@ #error "cuda.h is older than the floor of this cuda.core release for its CUDA major series (see the cuda.core support policy)" #endif #endif + +#ifdef CUDA_CORE_BINDINGS_CUDA_VERSION +#if (CUDA_VERSION / 10) != (CUDA_CORE_BINDINGS_CUDA_VERSION / 10) +#error "the cuda.h the compiler resolved is not of the major.minor that the installed cuda-bindings was generated from (check CUDA_PATH, CUDA_HOME and the compiler's include path)" +#endif +#endif diff --git a/cuda_core/tests/test_bindings_floor.py b/cuda_core/tests/test_bindings_floor.py index deca684591e..4713fdedc6a 100644 --- a/cuda_core/tests/test_bindings_floor.py +++ b/cuda_core/tests/test_bindings_floor.py @@ -42,6 +42,15 @@ HOOK = REPO / "toolshed" / "check_cuda_core_bindings_floor.py" +def _pyproject_extras(): + try: + import tomllib + except ModuleNotFoundError: # Python 3.10 + import tomli as tomllib + with open(CUDA_CORE / "pyproject.toml", "rb") as f: + return tomllib.load(f)["project"]["optional-dependencies"] + + def _load(name, path): spec = importlib.util.spec_from_file_location(name, path) module = importlib.util.module_from_spec(spec) @@ -291,6 +300,15 @@ def _import_error(self, version, cuda_version, tmp_path): assert result.stdout.startswith("IMPORTERROR:"), result.stdout return result.stdout + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_the_build_record_comes_from_the_compiler(self): + # _build_info is an extension module: CUDA_VERSION is the macro of the cuda.h the + # compiler resolved, and the major and floor come from the compile-time environment. + major, cuda_version, floor = self._build() + assert cuda_version // 1000 == major == floor[0] + assert header_minor(cuda_version) >= header_minor(cuda_version_of(floor)) + assert floor == floor_mod.floors_from_extras(_pyproject_extras())[major] + @pytest.mark.agent_authored(model="claude-fable-5-1") def test_below_the_floor_fails_at_import_with_the_fix(self, tmp_path): major, cuda_version, floor = self._build() diff --git a/cuda_core/tests/test_build_hooks.py b/cuda_core/tests/test_build_hooks.py index f454e9de293..a88592ba953 100644 --- a/cuda_core/tests/test_build_hooks.py +++ b/cuda_core/tests/test_build_hooks.py @@ -652,38 +652,15 @@ class TestBuildConfigurationCheck: FLOOR = build_hooks._bindings_floors() - @pytest.fixture(autouse=True) - def _isolate_build_info(self, tmp_path, monkeypatch): - monkeypatch.setattr(build_hooks, "_BUILD_INFO_PATH", tmp_path / "_build_info.py") - @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize("major", [12, 13]) - def test_floor_bindings_and_matching_header_pass_and_are_recorded(self, tmp_path, monkeypatch, major): + def test_floor_bindings_and_matching_header_pass(self, tmp_path, monkeypatch, major): floor = self.FLOOR[major] - version = f"{floor[0]}.{floor[1]}.{floor[2] + 1}.dev3+gabcdef0" - _fake_bindings(monkeypatch, version) - cuda_path = _write_cuda_h(tmp_path, floor[0] * 1000 + floor[1] * 10) - - build_hooks._check_build_configuration(cuda_path, str(major)) - - spec = importlib.util.spec_from_file_location("_build_info_under_test", build_hooks._BUILD_INFO_PATH) - info = importlib.util.module_from_spec(spec) - spec.loader.exec_module(info) - assert major == info.CUDA_MAJOR - assert floor[0] * 1000 + floor[1] * 10 == info.CUDA_VERSION - assert floor == info.CUDA_BINDINGS_FLOOR - assert version == info.CUDA_BINDINGS_BUILD_VERSION - # ci/tools/cuda_core_bindings_floor.py reads this record out of the wheel when BINDINGS_SOURCE=floor. - tool_path = Path(__file__).resolve().parents[2] / "ci" / "tools" / "cuda_core_bindings_floor.py" - if tool_path.is_file(): # absent from an sdist tree - spec = importlib.util.spec_from_file_location("cuda_core_bindings_floor_tool", tool_path) - tool = importlib.util.module_from_spec(spec) - spec.loader.exec_module(tool) - text = build_hooks._BUILD_INFO_PATH.read_text(encoding="utf-8") - assert tool.floor_from_source(text, major) == f"{floor[0]}.{floor[1]}.{floor[2]}" - other = 25 - major - with pytest.raises(SystemExit, match=f"records a CUDA {major} build, not CUDA {other}"): - tool.floor_from_source(text, other) + header = floor[0] * 1000 + floor[1] * 10 + _fake_bindings(monkeypatch, f"{floor[0]}.{floor[1]}.{floor[2] + 1}.dev3+gabcdef0") + cuda_path = _write_cuda_h(tmp_path, header) + # Returns the header cuda-bindings was generated from, for the versions.hpp cross-check. + assert build_hooks._check_build_configuration(cuda_path, str(major)) == header @pytest.mark.agent_authored(model="claude-fable-5-1") def test_bindings_below_the_floor_fail(self, tmp_path, monkeypatch): @@ -692,7 +669,6 @@ def test_bindings_below_the_floor_fail(self, tmp_path, monkeypatch): cuda_path = _write_cuda_h(tmp_path, 13040) with pytest.raises(RuntimeError, match=r"requires cuda-bindings >= 13\.\d+\.\d+ for CUDA 13"): build_hooks._check_build_configuration(cuda_path, "13") - assert not build_hooks._BUILD_INFO_PATH.exists() @pytest.mark.agent_authored(model="claude-fable-5-1") def test_bindings_of_another_major_fail(self, tmp_path, monkeypatch): @@ -738,8 +714,7 @@ def test_development_bindings_pass_on_their_generated_header(self, tmp_path, mon new_header = floor[0] * 1000 + (floor[1] + 1) * 10 _fake_bindings(monkeypatch, f"{floor[0]}.{floor[1]}.{floor[2]}.dev5+gabcdef0", cuda_version=new_header) cuda_path = _write_cuda_h(tmp_path, new_header) - build_hooks._check_build_configuration(cuda_path, "13") - assert f"CUDA_VERSION = {new_header}" in build_hooks._BUILD_INFO_PATH.read_text() + assert build_hooks._check_build_configuration(cuda_path, "13") == new_header @pytest.mark.agent_authored(model="claude-fable-5-1") def test_missing_bindings_is_a_build_error(self, tmp_path, monkeypatch): @@ -862,7 +837,7 @@ def test_unsupported_major_names_the_supported_ones(self, monkeypatch): class TestDefineMacros: - """The C++ learns the build decision through two macros. See _cpp/rt/versions.hpp.""" + """The C++ learns the build decision through three macros. See _cpp/rt/versions.hpp.""" @pytest.mark.agent_authored(model="claude-fable-5-1") @pytest.mark.parametrize("major", ["12", "13"]) @@ -873,6 +848,20 @@ def test_major_and_floor_header_version(self, major): ("CUDA_CORE_MIN_CUDA_VERSION", str(floor[0] * 1000 + floor[1] * 10)), ] + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_the_bindings_header_is_a_third_macro(self): + # versions.hpp compares the compiler's cuda.h with it by major.minor. + assert build_hooks._build_define_macros("13", 13040)[2] == ("CUDA_CORE_BINDINGS_CUDA_VERSION", "13040") + + @pytest.mark.agent_authored(model="claude-fable-5-1") + def test_the_floor_reaches_the_cython_compile_time_environment(self, monkeypatch): + # cuda/core/_build_info.pyx records it next to the header the compiler resolved. + captured = _capture_cythonize_kwargs(monkeypatch, "13") + assert captured["compile_time_env"] == { + "CUDA_CORE_BUILD_MAJOR": 13, + "CUDA_CORE_BINDINGS_FLOOR": build_hooks._bindings_floors()[13], + } + @pytest.mark.agent_authored(model="claude-fable-5-1") def test_extensions_receive_the_macros(self, monkeypatch): captured = {} diff --git a/cuda_python/setup.py b/cuda_python/setup.py index b380d56c349..a92ca092e71 100644 --- a/cuda_python/setup.py +++ b/cuda_python/setup.py @@ -30,7 +30,10 @@ version=version, install_requires=[ f"cuda-bindings{matcher}{version}", - "cuda-core~=1.2.0", + # Unpinned: cuda-core releases on its own cadence and declares its own + # cuda-bindings floors. A pin here would make older cuda-python releases + # unresolvable after a cuda-core release. + "cuda-core", "cuda-pathfinder~=1.1", ], extras_require={ From f53077f05037e43489d7123d7e7c3409b0ead5a1 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Wed, 30 Sep 2026 14:39:01 -0700 Subject: [PATCH 17/18] ci: install yq in the Linux GPU test containers ci/tools/env-vars reads ci/versions.yml with yq, and the ubuntu:24.04 test container has none, so every Linux test job stopped at the environment step. A composite action installs the pinned, checksummed mikefarah/yq release for amd64 or arm64; the Windows GPU runners already provide yq. Co-Authored-By: Claude Fable 5.1 --- .github/actions/setup_yq_linux/action.yml | 41 +++++++++++++++++++++++ .github/workflows/coverage.yml | 6 ++++ .github/workflows/test-wheel-linux.yml | 6 ++++ 3 files changed, 53 insertions(+) create mode 100644 .github/actions/setup_yq_linux/action.yml diff --git a/.github/actions/setup_yq_linux/action.yml b/.github/actions/setup_yq_linux/action.yml new file mode 100644 index 00000000000..aa2c97d29b2 --- /dev/null +++ b/.github/actions/setup_yq_linux/action.yml @@ -0,0 +1,41 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# SPDX-License-Identifier: Apache-2.0 + +name: Set up yq on Linux + +description: > + Install mikefarah/yq where the runner image does not provide it (the Linux GPU + jobs run in a plain ubuntu container). ci/tools/env-vars reads ci/versions.yml + with yq. A no-op when yq is already on PATH. Needs wget. + +inputs: + arch: + description: The yq release architecture, amd64 or arm64. + required: true + +runs: + using: composite + steps: + - name: Install yq + shell: bash --noprofile --norc -xeuo pipefail {0} + env: + YQ_VERSION: v4.52.5 + YQ_ARCH: ${{ inputs.arch }} + run: | + if command -v yq >/dev/null 2>&1; then + yq --version + exit 0 + fi + case "${YQ_ARCH}" in + amd64) YQ_SHA256=75d893a0d5940d1019cb7cdc60001d9e876623852c31cfc6267047bc31149fa9 ;; + arm64) YQ_SHA256=90fa510c50ee8ca75544dbfffed10c88ed59b36834df35916520cddc623d9aaa ;; + *) echo "Error: unsupported yq architecture: ${YQ_ARCH}" >&2; exit 1 ;; + esac + YQ_DIR="${RUNNER_TEMP}/yq" + mkdir -p "${YQ_DIR}" + wget -nv -O "${YQ_DIR}/yq" "https://github.com/mikefarah/yq/releases/download/${YQ_VERSION}/yq_linux_${YQ_ARCH}" + echo "${YQ_SHA256} ${YQ_DIR}/yq" | sha256sum -c - + chmod +x "${YQ_DIR}/yq" + echo "${YQ_DIR}" >> "${GITHUB_PATH}" + "${YQ_DIR}/yq" --version diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index 2f66aa4e156..deafe2729a8 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -83,6 +83,12 @@ jobs: dependencies: "tree rsync libsqlite3-0 g++ jq wget libgl1 libegl1" dependent_exes: "tree rsync libsqlite3-0 g++ jq wget libgl1 libegl1" + - name: Set up yq + # The container image has no yq; ci/tools/env-vars reads ci/versions.yml with it. + uses: ./.github/actions/setup_yq_linux + with: + arch: amd64 + - name: Setup proxy cache uses: nv-gha-runners/setup-proxy-cache@main continue-on-error: true diff --git a/.github/workflows/test-wheel-linux.yml b/.github/workflows/test-wheel-linux.yml index abf88bf0bc7..a0ca11a1440 100644 --- a/.github/workflows/test-wheel-linux.yml +++ b/.github/workflows/test-wheel-linux.yml @@ -143,6 +143,12 @@ jobs: dependencies: "jq wget libgl1 libegl1 g++ util-linux" dependent_exes: "jq wget" + - name: Set up yq + # The container image has no yq; ci/tools/env-vars reads ci/versions.yml with it. + uses: ./.github/actions/setup_yq_linux + with: + arch: ${{ matrix.ARCH }} + - name: Install GPU driver if: ${{ matrix.DRIVER != 'latest' && matrix.DRIVER != 'earliest' }} env: From 5417ea7828b815c697100a8502ce4d639aede9c1 Mon Sep 17 00:00:00 2001 From: Andy Jost Date: Thu, 1 Oct 2026 11:04:31 -0700 Subject: [PATCH 18/18] Address review: restore the cuda-core pin; re-deliver KeyboardInterrupt from the table fill; yq via apt; nits (#2783) - cuda_python/setup.py pins cuda-core~=1.2.0 again. The bump at each cuda-core release, followed by a cuda-python release, is documented in .github/RELEASE-core.md, cuda_python/AGENTS.md and cuda_core/AGENTS.md. - A KeyboardInterrupt raised inside the function-table fill is no longer swallowed. The fill re-arms it with PyErr_SetInterrupt() after its warning, and report_message() does the same when an interrupt fires inside the warnings machinery, so the user sees the interrupt at the next bytecode boundary and the warning survives. Empty exception text falls back to the type name. Test: cuda_core/tests/test_driver_table.py, through the NVRTC table, CPU-only. - The Linux GPU test containers get yq from apt (the Python jq wrapper, which accepts the same command); the composite action is removed. - DeviceArch is a FastEnum again (present at both floors). - support.rst: the driver bullet states what the floor does not change and what an older driver means. Small comment and release-note fixes. Co-Authored-By: Claude Fable 5.1 --- .github/RELEASE-core.md | 11 ++- .github/actions/setup_yq_linux/action.yml | 41 ---------- .github/workflows/coverage.yml | 10 +-- .github/workflows/test-wheel-linux.yml | 11 +-- .../docs/source/release/13.5.0-notes.rst | 2 +- cuda_core/AGENTS.md | 4 + cuda_core/cuda/core/_cpp/rt/DESIGN.md | 9 ++- cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp | 30 ++++++-- cuda_core/cuda/core/_cpp/rt/py_report.cpp | 22 +++++- cuda_core/cuda/core/_memory/_copy_enums.py | 7 +- cuda_core/cuda/core/system/_device.pyi | 2 +- cuda_core/cuda/core/system/_device.pyx | 6 +- cuda_core/cuda/core/system/typing.py | 9 +-- cuda_core/docs/source/support.rst | 6 +- cuda_core/tests/test_driver_table.py | 76 ++++++++++++++++--- cuda_core/tests/test_enum_coverage.py | 5 +- cuda_python/AGENTS.md | 10 +++ cuda_python/setup.py | 7 +- 18 files changed, 161 insertions(+), 107 deletions(-) delete mode 100644 .github/actions/setup_yq_linux/action.yml diff --git a/.github/RELEASE-core.md b/.github/RELEASE-core.md index a93b3ed8d61..0f33dfe44f3 100644 --- a/.github/RELEASE-core.md +++ b/.github/RELEASE-core.md @@ -62,9 +62,14 @@ platforms as appropriate for each release. ## Check (or update if needed) the dependency requirements Review `cuda_core/pyproject.toml` and verify that all dependency -requirements are current. - -Update the cuda_core dependency in `cuda_python/setup.py`. +requirements are current. The `cu12`/`cu13` extras declare the +`cuda-bindings` floors; see "Bumping the cuda-bindings floor" in +`cuda_core/AGENTS.md`. + +Update the `cuda-core` pin in `cuda_python/setup.py` to the new minor +series (`cuda-core~=X.Y.0`), and plan a `cuda-python` release right after +this one. Until that release, `pip install cuda-python` resolves to the +previous `cuda-core`. --- diff --git a/.github/actions/setup_yq_linux/action.yml b/.github/actions/setup_yq_linux/action.yml deleted file mode 100644 index aa2c97d29b2..00000000000 --- a/.github/actions/setup_yq_linux/action.yml +++ /dev/null @@ -1,41 +0,0 @@ -# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# -# SPDX-License-Identifier: Apache-2.0 - -name: Set up yq on Linux - -description: > - Install mikefarah/yq where the runner image does not provide it (the Linux GPU - jobs run in a plain ubuntu container). ci/tools/env-vars reads ci/versions.yml - with yq. A no-op when yq is already on PATH. Needs wget. - -inputs: - arch: - description: The yq release architecture, amd64 or arm64. - required: true - -runs: - using: composite - steps: - - name: Install yq - shell: bash --noprofile --norc -xeuo pipefail {0} - env: - YQ_VERSION: v4.52.5 - YQ_ARCH: ${{ inputs.arch }} - run: | - if command -v yq >/dev/null 2>&1; then - yq --version - exit 0 - fi - case "${YQ_ARCH}" in - amd64) YQ_SHA256=75d893a0d5940d1019cb7cdc60001d9e876623852c31cfc6267047bc31149fa9 ;; - arm64) YQ_SHA256=90fa510c50ee8ca75544dbfffed10c88ed59b36834df35916520cddc623d9aaa ;; - *) echo "Error: unsupported yq architecture: ${YQ_ARCH}" >&2; exit 1 ;; - esac - YQ_DIR="${RUNNER_TEMP}/yq" - mkdir -p "${YQ_DIR}" - wget -nv -O "${YQ_DIR}/yq" "https://github.com/mikefarah/yq/releases/download/${YQ_VERSION}/yq_linux_${YQ_ARCH}" - echo "${YQ_SHA256} ${YQ_DIR}/yq" | sha256sum -c - - chmod +x "${YQ_DIR}/yq" - echo "${YQ_DIR}" >> "${GITHUB_PATH}" - "${YQ_DIR}/yq" --version diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index deafe2729a8..ad23d0e48a4 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -80,14 +80,8 @@ jobs: uses: ./.github/actions/install_unix_deps continue-on-error: false with: - dependencies: "tree rsync libsqlite3-0 g++ jq wget libgl1 libegl1" - dependent_exes: "tree rsync libsqlite3-0 g++ jq wget libgl1 libegl1" - - - name: Set up yq - # The container image has no yq; ci/tools/env-vars reads ci/versions.yml with it. - uses: ./.github/actions/setup_yq_linux - with: - arch: amd64 + dependencies: "tree rsync libsqlite3-0 g++ jq wget libgl1 libegl1 yq" + dependent_exes: "tree rsync libsqlite3-0 g++ jq wget libgl1 libegl1 yq" - name: Setup proxy cache uses: nv-gha-runners/setup-proxy-cache@main diff --git a/.github/workflows/test-wheel-linux.yml b/.github/workflows/test-wheel-linux.yml index a0ca11a1440..f29d3bb41bb 100644 --- a/.github/workflows/test-wheel-linux.yml +++ b/.github/workflows/test-wheel-linux.yml @@ -140,14 +140,9 @@ jobs: # for artifact fetching, graphics libs, g++ required for cffi in # example; util-linux for `nsenter` (custom-DRIVER rows re-exec # install_gpu_driver.sh onto the host through nsenter) - dependencies: "jq wget libgl1 libegl1 g++ util-linux" - dependent_exes: "jq wget" - - - name: Set up yq - # The container image has no yq; ci/tools/env-vars reads ci/versions.yml with it. - uses: ./.github/actions/setup_yq_linux - with: - arch: ${{ matrix.ARCH }} + # yq: ci/tools/env-vars reads ci/versions.yml with it + dependencies: "jq wget libgl1 libegl1 g++ util-linux yq" + dependent_exes: "jq wget yq" - name: Install GPU driver if: ${{ matrix.DRIVER != 'latest' && matrix.DRIVER != 'earliest' }} diff --git a/cuda_bindings/docs/source/release/13.5.0-notes.rst b/cuda_bindings/docs/source/release/13.5.0-notes.rst index 304ad50cf64..2c100bf0f0c 100644 --- a/cuda_bindings/docs/source/release/13.5.0-notes.rst +++ b/cuda_bindings/docs/source/release/13.5.0-notes.rst @@ -301,7 +301,7 @@ Enhancements - ``nvvm.add_module_to_program``, ``nvvm.lazy_add_module_to_program``, ``nvvm.get_compiled_result``, ``nvvm.get_program_log`` (``buffer``) - For all ``cuda-bindings`` APIs, functions that accept a struct wrapper will accept the struct wrapper directly, rather than requiring getting the ``.ptr`` property. - Creating class instances in ``cuda-bindings`` should now be faster in most cases, because it requires 1 heap allocation rather than 2. -- Before a source build compiles anything, it now checks that the CUDA Toolkit's ``cuda.h`` has the major.minor this source tree was generated from. A mismatch stops the build with a message that names the header found and the version needed. Previously a mismatch surfaced as a long list of C++ redefinition errors. +- Before a source build compiles anything, it now checks that the CUDA Toolkit's ``cuda.h`` has the major.minor the source tree was generated from. A mismatch stops the build with a message that names the header found and the version needed. Previously a mismatch surfaced as a long list of C++ redefinition errors. Behavior changes ---------------- diff --git a/cuda_core/AGENTS.md b/cuda_core/AGENTS.md index c4899495c03..4053d48958a 100644 --- a/cuda_core/AGENTS.md +++ b/cuda_core/AGENTS.md @@ -66,6 +66,10 @@ a bump. To bump: 5. Pin `cuda-bindings` to match in the conda-forge `cuda-core` feedstock. The feedstock lives outside this repository. +A `cuda-core` release also bumps the `cuda-core~=X.Y.0` pin in +`cuda_python/setup.py`, followed by a `cuda-python` release (see +`.github/RELEASE-core.md`). + ### CUDA Toolkit minor bumps The build compares the toolkit's `cuda.h` with the header that the installed diff --git a/cuda_core/cuda/core/_cpp/rt/DESIGN.md b/cuda_core/cuda/core/_cpp/rt/DESIGN.md index 07cb17e2bd4..5f8f026dae0 100644 --- a/cuda_core/cuda/core/_cpp/rt/DESIGN.md +++ b/cuda_core/cuda/core/_cpp/rt/DESIGN.md @@ -440,8 +440,13 @@ which a thread holding a C++ lock can deadlock (see "GIL Management"). Python exceptions raised by that code never become C++ exceptions: the C API reports them as return codes, and `report_message` hands them to -`sys.unraisablehook`. Nothing on the report path may allocate or throw, since a -deleter is `noexcept`. +`sys.unraisablehook`. The one exception is `KeyboardInterrupt`: a SIGINT that +fires inside the warnings machinery is not swallowed. `report_message` emits +the warning again and re-arms the interrupt with `PyErr_SetInterrupt()`, so +Python raises it at the next bytecode boundary in the caller's code. The +function-table fill does the same when its Python calls are interrupted +(`take_python_error` in `py_driver_fns.cpp`). Nothing on the report path may +allocate or throw, since a deleter is `noexcept`. So: use `pw_` only in deleters and cleanup paths that hold no C++ lock and have finished updating the layer's own state. Where a lock must stay held, call diff --git a/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp b/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp index 0b3139bddf8..83b136557f2 100644 --- a/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp +++ b/cuda_core/cuda/core/_cpp/rt/py_driver_fns.cpp @@ -107,11 +107,19 @@ const char* library_name(FnTable table) noexcept { // true when the exception says nothing about cuda-bindings or the driver: an // interruption such as KeyboardInterrupt or SystemExit, or exhaustion such as // MemoryError or RecursionError. The fill reports such a failure but does not -// latch it, so the next call tries again. -bool take_python_error(char* buf, std::size_t size) noexcept { +// latch the table as failed, so the next call tries again. +// +// Sets *interrupted when the exception was a KeyboardInterrupt. The fill runs +// in noexcept code reached from deleters and nogil blocks, so it cannot +// propagate the exception, but the caller re-arms the interrupt with +// PyErr_SetInterrupt() once its own Python work (the warning) is done. Python +// then raises KeyboardInterrupt at the next bytecode boundary, so the user +// sees the interrupt, not the CUDAError from the failed call. +bool take_python_error(char* buf, std::size_t size, bool* interrupted) noexcept { const bool transient = PyErr_Occurred() && (!PyErr_ExceptionMatches(PyExc_Exception) || PyErr_ExceptionMatches(PyExc_MemoryError) || PyErr_ExceptionMatches(PyExc_RecursionError)); + *interrupted = PyErr_Occurred() && PyErr_ExceptionMatches(PyExc_KeyboardInterrupt); #if PY_VERSION_HEX >= 0x030C0000 PyObject* exc = PyErr_GetRaisedException(); #else @@ -124,7 +132,10 @@ bool take_python_error(char* buf, std::size_t size) noexcept { #endif PyObject* text = exc ? PyObject_Str(exc) : nullptr; const char* utf8 = text ? PyUnicode_AsUTF8(text) : nullptr; - std::snprintf(buf, size, "%s", utf8 ? utf8 : "unknown error"); + if (utf8 == nullptr || *utf8 == '\0') { + utf8 = exc ? Py_TYPE(exc)->tp_name : "unknown error"; // str(KeyboardInterrupt()) is empty + } + std::snprintf(buf, size, "%s", utf8); Py_XDECREF(text); Py_XDECREF(exc); PyErr_Clear(); @@ -188,20 +199,27 @@ bool ensure_fn_table(FnTable table) noexcept { } PendingExceptionGuard pending; + bool interrupted = false; PyObject* module = PyImport_ImportModule(module_name(table)); if (module == nullptr) { - const bool transient = take_python_error(cause, sizeof(cause)); + const bool transient = take_python_error(cause, sizeof(cause), &interrupted); std::snprintf(message, sizeof(message), "cuda.core cannot import %s from the installed cuda-bindings: %s", module_name(table), cause); record_failure(table, message, !transient); + if (interrupted) { + PyErr_SetInterrupt(); // after the warning: the handler must not consume the interrupt + } return false; } PyObject* pointers = PyObject_CallMethod(module, "_inspect_function_pointers", nullptr); Py_DECREF(module); if (pointers == nullptr) { - const bool transient = take_python_error(cause, sizeof(cause)); + const bool transient = take_python_error(cause, sizeof(cause), &interrupted); std::snprintf(message, sizeof(message), "cuda-bindings could not load the %s: %s", library_name(table), cause); record_failure(table, message, !transient); + if (interrupted) { + PyErr_SetInterrupt(); // after the warning: the handler must not consume the interrupt + } return false; } if (!PyDict_Check(pointers)) { @@ -227,7 +245,7 @@ bool ensure_fn_table(FnTable table) noexcept { void* address = PyLong_AsVoidPtr(item); if (address == nullptr && PyErr_Occurred()) { Py_DECREF(pointers); - take_python_error(cause, sizeof(cause)); + take_python_error(cause, sizeof(cause), &interrupted); std::snprintf(message, sizeof(message), "internal cuda.core error, please report: the entry for %s in %s is not an address: %s", entries[i].key, module_name(table), cause); diff --git a/cuda_core/cuda/core/_cpp/rt/py_report.cpp b/cuda_core/cuda/core/_cpp/rt/py_report.cpp index cde00589650..4beff4dec07 100644 --- a/cuda_core/cuda/core/_cpp/rt/py_report.cpp +++ b/cuda_core/cuda/core/_cpp/rt/py_report.cpp @@ -22,6 +22,12 @@ std::atomic warning_category{nullptr}; // warning was promoted to an error), the failure is written as an unraisable // exception, the CPython convention for exceptions in destructors. Falls back // to stderr when the interpreter cannot be used. +// +// A KeyboardInterrupt is never swallowed here. The warnings machinery runs +// Python code, so a pending SIGINT (one the user typed, or one that a failed +// table fill re-armed; see py_driver_fns.cpp) can fire inside it. The warning +// is then emitted again and the interrupt re-armed with PyErr_SetInterrupt(), +// so Python raises it at the next bytecode boundary in the caller's code. void report_message(const char* message) noexcept { PyObject* category = warning_category.load(std::memory_order_acquire); if (category && Py_IsInitialized() && !py_is_finalizing()) { @@ -34,16 +40,26 @@ void report_message(const char* message) noexcept { PyObject *pending_type, *pending_value, *pending_tb; PyErr_Fetch(&pending_type, &pending_value, &pending_tb); #endif + bool interrupted = false; if (PyErr_WarnEx(category, message, 1) != 0) { - PyObject* subject = PyUnicode_FromString(message); - PyErr_WriteUnraisable(subject); - Py_XDECREF(subject); + interrupted = PyErr_ExceptionMatches(PyExc_KeyboardInterrupt); + if (interrupted) { + PyErr_Clear(); + } + if (!interrupted || PyErr_WarnEx(category, message, 1) != 0) { + PyObject* subject = PyUnicode_FromString(message); + PyErr_WriteUnraisable(subject); + Py_XDECREF(subject); + } } #if PY_VERSION_HEX >= 0x030C0000 PyErr_SetRaisedException(pending); #else PyErr_Restore(pending_type, pending_value, pending_tb); #endif + if (interrupted) { + PyErr_SetInterrupt(); + } return; } } diff --git a/cuda_core/cuda/core/_memory/_copy_enums.py b/cuda_core/cuda/core/_memory/_copy_enums.py index 880d21e2d93..21941f0e21b 100644 --- a/cuda_core/cuda/core/_memory/_copy_enums.py +++ b/cuda_core/cuda/core/_memory/_copy_enums.py @@ -124,11 +124,8 @@ def _to_driver_flags(self) -> int: # CUDA 12.8 added CUmemcpySrcAccessOrder and CUmemcpyFlags. Every -# cuda-bindings that cuda.core accepts has them. Keyed by ``str``: under -# ``python_version = "3.10"`` mypy resolves StrEnum to the unstubbed backports -# shim and so infers the members as plain ``str``. StrEnum members are ``str`` -# instances, so this holds on every version. The values are wrapped in -# ``int()`` because the driver enums are untyped. +# cuda-bindings that cuda.core accepts has them. Keyed by ``str`` (StrEnum +# members are ``str``; mypy on 3.10 cannot see the enum) and valued as ``int``. _src_order = driver.CUmemcpySrcAccessOrder _flags = driver.CUmemcpyFlags _SRC_ACCESS_ORDER_TO_DRIVER: dict[str, int] = { diff --git a/cuda_core/cuda/core/system/_device.pyi b/cuda_core/cuda/core/system/_device.pyi index 1414215300c..5273fffa446 100644 --- a/cuda_core/cuda/core/system/_device.pyi +++ b/cuda_core/cuda/core/system/_device.pyi @@ -29,7 +29,7 @@ _THERMAL_TARGET_MAPPING = {nvml.ThermalTarget.NONE: ThermalTarget.NONE, nvml.The _THERMAL_TARGET_INV_MAPPING = {v: k for k, v in _THERMAL_TARGET_MAPPING.items()} _ADDRESSING_MODE_MAPPING = {nvml.DeviceAddressingModeType.DEVICE_ADDRESSING_MODE_HMM: AddressingMode.HMM, nvml.DeviceAddressingModeType.DEVICE_ADDRESSING_MODE_ATS: AddressingMode.ATS} _AFFINITY_SCOPE_MAPPING = {AffinityScope.NODE: nvml.AffinityScope.NODE, AffinityScope.SOCKET: nvml.AffinityScope.SOCKET} -_BRAND_TYPE_MAPPING = {nvml.BrandType.BRAND_UNKNOWN: 'Unknown', nvml.BrandType.BRAND_QUADRO: 'Quadro', nvml.BrandType.BRAND_TESLA: 'Tesla', nvml.BrandType.BRAND_NVS: 'NVS', nvml.BrandType.BRAND_GRID: 'GRID', nvml.BrandType.BRAND_GEFORCE: 'GeForce', nvml.BrandType.BRAND_TITAN: 'Titan', nvml.BrandType.BRAND_NVIDIA_VAPPS: 'NVIDIA vApps', nvml.BrandType.BRAND_NVIDIA_VPC: 'NVIDIA VPC', nvml.BrandType.BRAND_NVIDIA_VCS: 'NVIDIA VCS', nvml.BrandType.BRAND_NVIDIA_VWS: 'NVIDIA VWS', nvml.BrandType.BRAND_NVIDIA_CLOUD_GAMING: 'NVIDIA Cloud Gaming', nvml.BrandType.BRAND_NVIDIA_VGAMING: 'NVIDIA vGaming', nvml.BrandType.BRAND_QUADRO_RTX: 'Quadro RTX', nvml.BrandType.BRAND_NVIDIA_RTX: 'NVIDIA RTX', nvml.BrandType.BRAND_NVIDIA: 'NVIDIA', nvml.BrandType.BRAND_GEFORCE_RTX: 'GeForce RTX', nvml.BrandType.BRAND_TITAN_RTX: 'Titan RTX'} +_BRAND_TYPE_MAPPING = {nvml.BrandType.BRAND_UNKNOWN: 'Unknown', nvml.BrandType.BRAND_QUADRO: 'Quadro', nvml.BrandType.BRAND_TESLA: 'Tesla', nvml.BrandType.BRAND_NVS: 'NVS', nvml.BrandType.BRAND_GRID: 'GRID', nvml.BrandType.BRAND_GEFORCE: 'GeForce', nvml.BrandType.BRAND_TITAN: 'Titan', nvml.BrandType.BRAND_NVIDIA_VAPPS: 'NVIDIA vApps', nvml.BrandType.BRAND_NVIDIA_VPC: 'NVIDIA VPC', nvml.BrandType.BRAND_NVIDIA_VCS: 'NVIDIA VCS', nvml.BrandType.BRAND_NVIDIA_VWS: 'NVIDIA VWS', nvml.BrandType.BRAND_NVIDIA_CLOUD_GAMING: 'NVIDIA Cloud Gaming', nvml.BrandType.BRAND_NVIDIA_VGAMING: 'NVIDIA vGaming', nvml.BrandType.BRAND_QUADRO_RTX: 'Quadro RTX', nvml.BrandType.BRAND_NVIDIA_RTX: 'NVIDIA RTX', nvml.BrandType.BRAND_NVIDIA: 'NVIDIA', nvml.BrandType.BRAND_GEFORCE_RTX: 'GeForce RTX', nvml.BrandType.BRAND_TITAN_RTX: 'Titan RTX', nvml.BrandType.BRAND_NVIDIA_DLA: 'NVIDIA DLA', nvml.BrandType.BRAND_NVIDIA_VGAMEDEV: 'NVIDIA vGameDev', nvml.BrandType.BRAND_NVIDIA_NPU: 'NVIDIA NPU'} _GPU_P2P_CAPS_INDEX_MAPPING = {GpuP2PCapsIndex.READ: nvml.GpuP2PCapsIndex.P2P_CAPS_INDEX_READ, GpuP2PCapsIndex.WRITE: nvml.GpuP2PCapsIndex.P2P_CAPS_INDEX_WRITE, GpuP2PCapsIndex.NVLINK: nvml.GpuP2PCapsIndex.P2P_CAPS_INDEX_NVLINK, GpuP2PCapsIndex.ATOMICS: nvml.GpuP2PCapsIndex.P2P_CAPS_INDEX_ATOMICS, GpuP2PCapsIndex.PCI: nvml.GpuP2PCapsIndex.P2P_CAPS_INDEX_PCI, GpuP2PCapsIndex.PROP: nvml.GpuP2PCapsIndex.P2P_CAPS_INDEX_PROP, GpuP2PCapsIndex.UNKNOWN: nvml.GpuP2PCapsIndex.P2P_CAPS_INDEX_UNKNOWN} _GPU_P2P_STATUS_MAPPING = {nvml.GpuP2PStatus.P2P_STATUS_OK: GpuP2PStatus.OK, nvml.GpuP2PStatus.P2P_STATUS_CHIPSET_NOT_SUPPORTED: GpuP2PStatus.CHIPSET_NOT_SUPPORTED, nvml.GpuP2PStatus.P2P_STATUS_GPU_NOT_SUPPORTED: GpuP2PStatus.GPU_NOT_SUPPORTED, nvml.GpuP2PStatus.P2P_STATUS_IOH_TOPOLOGY_NOT_SUPPORTED: GpuP2PStatus.IOH_TOPOLOGY_NOT_SUPPORTED, nvml.GpuP2PStatus.P2P_STATUS_DISABLED_BY_REGKEY: GpuP2PStatus.DISABLED_BY_REGKEY, nvml.GpuP2PStatus.P2P_STATUS_NOT_SUPPORTED: GpuP2PStatus.NOT_SUPPORTED, nvml.GpuP2PStatus.P2P_STATUS_UNKNOWN: GpuP2PStatus.UNKNOWN} _GPU_TOPOLOGY_LEVEL_MAPPING = {GpuTopologyLevel.INTERNAL: nvml.GpuTopologyLevel.TOPOLOGY_INTERNAL, GpuTopologyLevel.SINGLE: nvml.GpuTopologyLevel.TOPOLOGY_SINGLE, GpuTopologyLevel.MULTIPLE: nvml.GpuTopologyLevel.TOPOLOGY_MULTIPLE, GpuTopologyLevel.HOSTBRIDGE: nvml.GpuTopologyLevel.TOPOLOGY_HOSTBRIDGE, GpuTopologyLevel.NODE: nvml.GpuTopologyLevel.TOPOLOGY_NODE, GpuTopologyLevel.SYSTEM: nvml.GpuTopologyLevel.TOPOLOGY_SYSTEM} diff --git a/cuda_core/cuda/core/system/_device.pyx b/cuda_core/cuda/core/system/_device.pyx index 2192a2a3f03..b5b68b1d3e2 100644 --- a/cuda_core/cuda/core/system/_device.pyx +++ b/cuda_core/cuda/core/system/_device.pyx @@ -105,14 +105,10 @@ _BRAND_TYPE_MAPPING = { nvml.BrandType.BRAND_NVIDIA: "NVIDIA", nvml.BrandType.BRAND_GEFORCE_RTX: "GeForce RTX", nvml.BrandType.BRAND_TITAN_RTX: "Titan RTX", -} - - -_BRAND_TYPE_MAPPING.update({ nvml.BrandType.BRAND_NVIDIA_DLA: "NVIDIA DLA", nvml.BrandType.BRAND_NVIDIA_VGAMEDEV: "NVIDIA vGameDev", nvml.BrandType.BRAND_NVIDIA_NPU: "NVIDIA NPU", -}) +} _GPU_P2P_CAPS_INDEX_MAPPING = { diff --git a/cuda_core/cuda/core/system/typing.py b/cuda_core/cuda/core/system/typing.py index 208e38876d9..a3888005bff 100644 --- a/cuda_core/cuda/core/system/typing.py +++ b/cuda_core/cuda/core/system/typing.py @@ -2,9 +2,8 @@ # # SPDX-License-Identifier: Apache-2.0 -import enum - from cuda.bindings import nvml as _nvml +from cuda.bindings._internal._fast_enum import FastEnum as _FastEnum from cuda.core._utils.pycompat import StrEnum __all__ = [ @@ -326,9 +325,9 @@ class ThermalTarget(StrEnum): # DeviceArch takes its values from cuda.bindings.nvml at definition time. -# It is an IntEnum rather than a StrEnum because the order of the values is -# meaningful, e.g. Kepler "or later". -class DeviceArch(enum.IntEnum): +# It is a FastEnum (an int) rather than a StrEnum because the order of the +# values is meaningful, e.g. Kepler "or later". +class DeviceArch(_FastEnum): """ Device architecture. """ diff --git a/cuda_core/docs/source/support.rst b/cuda_core/docs/source/support.rst index 36dffdf2c75..3e61cffc0bc 100644 --- a/cuda_core/docs/source/support.rst +++ b/cuda_core/docs/source/support.rst @@ -97,8 +97,10 @@ current release. The build, the import-time check, this page, and CI all read th as the header that ``cuda-bindings`` was generated from. Any other configuration fails the build with a message that names what was found and what is required. ``cuda.core`` does not support a build against a CUDA Toolkit older than the floor's minor. -- **The CUDA driver** is unaffected by the floor. Feature availability is decided by the driver alone: a - feature the installed driver lacks raises when it is used. +- **The CUDA driver.** The floor does not change the driver requirement. A ``cuda-core`` build + works with every driver of its CUDA major. An older driver can lack some of the build's + features. In that case ``cuda-core`` never crashes or returns a wrong result; depending on the + feature, it may raise an error or emulate the feature. A floor moves with each ``cuda-core`` release, to the newest ``cuda-bindings`` of each major at that time. It also moves in any release whose changes need a newer ``cuda-bindings`` API. The diff --git a/cuda_core/tests/test_driver_table.py b/cuda_core/tests/test_driver_table.py index 774c2060e41..725059e3fc5 100644 --- a/cuda_core/tests/test_driver_table.py +++ b/cuda_core/tests/test_driver_table.py @@ -12,8 +12,9 @@ - Every affected call raises :class:`CUDAError` with the reason attached as a note. - The failure latches: there is no second warning and no retry. -The child needs a loadable CUDA driver and a visible device, so this module skips without them. -The module runs with ``--noconftest``. +The driver-table tests need a loadable CUDA driver and a visible device, so they skip without +them. The interrupt test uses the NVRTC table and needs only libnvrtc. The module runs with +``--noconftest``. """ import os @@ -44,10 +45,19 @@ def _gpu_available() -> bool: return False -pytestmark = [ - pytest.mark.skipif(not _gpu_available(), reason="the child needs a CUDA driver and a visible device"), - pytest.mark.thread_unsafe(reason="spawns child interpreters"), -] +def _nvrtc_available() -> bool: + try: + from cuda.bindings import nvrtc + + status, *_ = nvrtc.nvrtcVersion() + return int(status) == 0 + except Exception: + return False + + +pytestmark = pytest.mark.thread_unsafe(reason="spawns child interpreters") +needs_gpu = pytest.mark.skipif(not _gpu_available(), reason="the child needs a CUDA driver and a visible device") +needs_nvrtc = pytest.mark.skipif(not _nvrtc_available(), reason="the child needs libnvrtc") def _table_keys() -> list[str]: @@ -100,10 +110,9 @@ def fake_inspect_function_pointers(): """) -def _run_child(mutation: str, tmp_path: Path) -> str: - code = _CHILD.format(keys=_table_keys(), mutation=mutation) +def _run_code(code: str, tmp_path: Path) -> subprocess.CompletedProcess: env = {k: v for k, v in os.environ.items() if k != "PYTHONPATH"} - result = subprocess.run( # noqa: S603 + return subprocess.run( # noqa: S603 [sys.executable, "-c", code], cwd=tmp_path, env=env, @@ -113,6 +122,10 @@ def _run_child(mutation: str, tmp_path: Path) -> str: timeout=180, check=False, ) + + +def _run_child(mutation: str, tmp_path: Path) -> str: + result = _run_code(_CHILD.format(keys=_table_keys(), mutation=mutation), tmp_path) assert result.returncode == 0, result.stdout + result.stderr return result.stdout @@ -123,6 +136,7 @@ def _build_major() -> int: return _build_info.CUDA_MAJOR +@needs_gpu @pytest.mark.agent_authored(model="claude-fable-5-1") def test_null_baseline_entry_fails_the_fill_once_and_latches(tmp_path): out = _run_child('table["__cuGetErrorName"] = 0', tmp_path) @@ -138,6 +152,7 @@ def test_null_baseline_entry_fails_the_fill_once_and_latches(tmp_path): assert "lacks cuGetErrorName" in warning +@needs_gpu @pytest.mark.agent_authored(model="claude-fable-5-1") def test_missing_table_entry_names_the_mismatch(tmp_path): out = _run_child('table.pop("__cuDevicePrimaryCtxRetain", None)', tmp_path) @@ -149,3 +164,46 @@ def test_missing_table_entry_names_the_mismatch(tmp_path): assert "has no entry for cuDevicePrimaryCtxRetain" in line assert "Install the cuda-bindings this cuda.core requires" in line assert "CUDAWARNINGS 1" in lines + + +_INTERRUPTED_CHILD = textwrap.dedent(""" + import warnings + import cuda.bindings._internal.nvrtc as loader + + real_inspect = loader._inspect_function_pointers + + def interrupting_inspect(): + raise KeyboardInterrupt # Ctrl-C while cuda-bindings loads the library + + loader._inspect_function_pointers = interrupting_inspect + from cuda.core import Program, ProgramOptions + + try: + with warnings.catch_warnings(record=True) as caught: + warnings.simplefilter("always") + # The NVRTC handle constructor fills the NVRTC table; the Program is dropped at + # once, so its deleter also reports through the table. + Program('extern "C" __global__ void k() {}', "c++", options=ProgramOptions(arch="sm_80")) + for _ in range(3): + pass # the re-armed interrupt fires at a bytecode boundary + print("NOT INTERRUPTED", flush=True) + except KeyboardInterrupt: + print("INTERRUPTED", flush=True) + for w in caught: + print("WARNING:", str(w.message), flush=True) + loader._inspect_function_pointers = real_inspect # let the shutdown deleters fill the table +""") + + +@needs_nvrtc +@pytest.mark.agent_authored(model="claude-fable-5-1") +def test_keyboard_interrupt_during_a_fill_reaches_the_user(tmp_path): + """A KeyboardInterrupt raised inside the fill cannot propagate from the noexcept C++, so + it is re-armed: the user's code sees it at the next bytecode boundary, the fill's warning + still names it, and the reporting path does not swallow it as an unraisable exception.""" + result = _run_code(_INTERRUPTED_CHILD, tmp_path) + assert result.returncode == 0, result.stdout + result.stderr + lines = result.stdout.splitlines() + assert "INTERRUPTED" in lines, result.stdout + result.stderr + assert any(line.startswith("WARNING:") and "KeyboardInterrupt" in line for line in lines), result.stdout + assert "Exception ignored" not in result.stderr, result.stderr diff --git a/cuda_core/tests/test_enum_coverage.py b/cuda_core/tests/test_enum_coverage.py index 38badecd29b..b75fc16acae 100644 --- a/cuda_core/tests/test_enum_coverage.py +++ b/cuda_core/tests/test_enum_coverage.py @@ -127,9 +127,6 @@ _MODULES.append(system_typing) -# Every ClocksEventReasons member is mapped: the floor cuda-bindings has them all. -_CLOCKS_EVENT_REASONS_STR_UNMAPPED = set() - _CASES.extend( [ ( @@ -162,7 +159,7 @@ system_typing.ClocksEventReasons, _device._CLOCKS_EVENT_REASONS_MAPPING, set(), - _CLOCKS_EVENT_REASONS_STR_UNMAPPED, + set(), ), ( nvml.EventType, diff --git a/cuda_python/AGENTS.md b/cuda_python/AGENTS.md index 7c4fb9c0b1e..9eef3d4b772 100644 --- a/cuda_python/AGENTS.md +++ b/cuda_python/AGENTS.md @@ -20,5 +20,15 @@ monorepo. component packages rather than here. - Be careful when changing dependency/version logic in `setup.py`; preserve compatibility between metapackage versioning and subpackage constraints. + +## Release coupling + +- `setup.py` pins `cuda-core` to a minor series (`cuda-core~=X.Y.0`). Every + `cuda-core` release must bump that pin, and a `cuda-python` release should + follow right after, so that `pip install cuda-python` resolves to the new + `cuda-core`. The `cuda-core` release checklist (`.github/RELEASE-core.md`) + has this step. +- Users should pin `cuda-python` alone. A separate `cuda-core` pin next to it + can make the install unresolvable after a `cuda-core` release. - If you update docs structure, ensure `docs/build_all_docs.sh` still collects docs from `cuda_python`, `cuda_bindings`, `cuda_core`, and `cuda_pathfinder`. diff --git a/cuda_python/setup.py b/cuda_python/setup.py index a92ca092e71..2c308da1c51 100644 --- a/cuda_python/setup.py +++ b/cuda_python/setup.py @@ -30,10 +30,9 @@ version=version, install_requires=[ f"cuda-bindings{matcher}{version}", - # Unpinned: cuda-core releases on its own cadence and declares its own - # cuda-bindings floors. A pin here would make older cuda-python releases - # unresolvable after a cuda-core release. - "cuda-core", + # Bump this with every cuda-core release and release cuda-python right + # after it; see .github/RELEASE-core.md and cuda_python/AGENTS.md. + "cuda-core~=1.2.0", "cuda-pathfinder~=1.1", ], extras_require={