Skip to content

Commit b0d9639

Browse files
authored
Merge branch 'main' into codex/pathfinder-windows-arch-search-ctk-next
2 parents df91f5f + 88e3df2 commit b0d9639

62 files changed

Lines changed: 2583 additions & 161 deletions

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.
Lines changed: 44 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,44 @@
1+
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
2+
#
3+
# SPDX-License-Identifier: Apache-2.0
4+
5+
name: griffe API check
6+
7+
description: >-
8+
Check a package's public API (as defined by `__all__`) for changes using
9+
griffe.
10+
11+
inputs:
12+
package-name:
13+
description: "Importable package name to check, e.g. cuda.core"
14+
required: true
15+
package-dir:
16+
description: "Directory to search for the package sources, e.g. cuda_core"
17+
required: true
18+
merge-base:
19+
description: >-
20+
Git ref/sha to compare the current code against, typically the PR's
21+
merge-base with its target branch.
22+
required: true
23+
24+
runs:
25+
using: composite
26+
steps:
27+
- name: Install uv
28+
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # v8.2.0
29+
with:
30+
enable-cache: false
31+
32+
- name: Check API
33+
shell: bash --noprofile --norc -euo pipefail {0}
34+
env:
35+
PACKAGE_NAME: ${{ inputs.package-name }}
36+
PACKAGE_DIR: ${{ inputs.package-dir }}
37+
MERGE_BASE: ${{ inputs.merge-base }}
38+
run: |
39+
uvx griffe check "$PACKAGE_NAME" \
40+
--search "$PACKAGE_DIR" \
41+
--find-stubs-packages \
42+
--against "$MERGE_BASE" \
43+
--format github \
44+
2>&1

‎.github/workflows/ci.yml‎

Lines changed: 85 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -103,6 +103,7 @@ jobs:
103103
test_bindings: ${{ steps.compose.outputs.test_bindings }}
104104
test_core: ${{ steps.compose.outputs.test_core }}
105105
test_pathfinder: ${{ steps.compose.outputs.test_pathfinder }}
106+
pr_merge_base: ${{ steps.filter.outputs.merge_base }}
106107
steps:
107108
- name: Checkout repository
108109
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
@@ -156,6 +157,7 @@ jobs:
156157
echo "python_meta=$(has_match '^cuda_python/')"
157158
echo "test_helpers=$(has_match '^cuda_python_test_helpers/')"
158159
echo "shared=$(has_match '^(\.github/|ci/|scripts/|toolshed/|conftest\.py$|pyproject\.toml$|pixi\.(toml|lock)$|pytest\.ini$|ruff\.toml$)')"
160+
echo "merge_base=${base}"
159161
} >> "$GITHUB_OUTPUT"
160162
161163
- name: Compose gating outputs
@@ -228,6 +230,89 @@ jobs:
228230
echo "test_pathfinder=${test_pathfinder}"
229231
} >> "$GITHUB_OUTPUT"
230232
233+
api-check-core-vs-release:
234+
name: API check (cuda_core vs. latest release)
235+
if: >-
236+
${{ !fromJSON(needs.should-skip.outputs.skip) &&
237+
fromJSON(needs.detect-changes.outputs.core) }}
238+
runs-on: ubuntu-latest
239+
needs:
240+
- should-skip
241+
- detect-changes
242+
permissions:
243+
contents: read
244+
steps:
245+
- name: Checkout repository
246+
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
247+
with:
248+
fetch-depth: 1
249+
filter: blob:none
250+
251+
- name: Find latest release tag
252+
id: latest-tag
253+
shell: bash --noprofile --norc -euo pipefail {0}
254+
env:
255+
GH_TOKEN: ${{ github.token }}
256+
run: |
257+
# --paginate fetches all pages; jq outputs one name per line per page;
258+
# head -1 takes the first (newest) match since GitHub returns tags
259+
# newest-first. Fails hard if no cuda-core-v* tag is found.
260+
tag="$(gh api "repos/$GITHUB_REPOSITORY/tags" --paginate \
261+
--jq '.[] | select(.name | startswith("cuda-core-v")) | .name' \
262+
| head -1)"
263+
if [[ -z "${tag}" ]]; then
264+
echo "::error::No cuda-core-v* tag found in the repository." >&2
265+
exit 1
266+
fi
267+
echo "tag=${tag}" >> "$GITHUB_OUTPUT"
268+
269+
- name: Fetch release tag
270+
shell: bash --noprofile --norc -euo pipefail {0}
271+
run: |
272+
git fetch --depth=1 --filter=blob:none origin \
273+
"refs/tags/${{ steps.latest-tag.outputs.tag }}:refs/tags/${{ steps.latest-tag.outputs.tag }}"
274+
275+
- name: Check cuda_core public API
276+
id: griffe
277+
uses: ./.github/actions/griffe-api-check
278+
with:
279+
package-name: cuda.core
280+
package-dir: cuda_core
281+
merge-base: ${{ steps.latest-tag.outputs.tag }}
282+
283+
api-check-core-vs-base:
284+
name: API check (cuda_core vs. merge base)
285+
if: >-
286+
${{ startsWith(github.ref_name, 'pull-request/') &&
287+
!fromJSON(needs.should-skip.outputs.skip) &&
288+
fromJSON(needs.detect-changes.outputs.core) }}
289+
runs-on: ubuntu-latest
290+
needs:
291+
- should-skip
292+
- detect-changes
293+
permissions:
294+
contents: read
295+
steps:
296+
- name: Checkout repository
297+
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
298+
with:
299+
fetch-depth: 1
300+
filter: blob:none
301+
302+
- name: Fetch merge base commit
303+
shell: bash --noprofile --norc -euo pipefail {0}
304+
run: |
305+
git fetch --depth=1 --filter=blob:none origin \
306+
"${{ needs.detect-changes.outputs.pr_merge_base }}"
307+
308+
- name: Check cuda_core public API
309+
id: griffe
310+
uses: ./.github/actions/griffe-api-check
311+
with:
312+
package-name: cuda.core
313+
package-dir: cuda_core
314+
merge-base: ${{ needs.detect-changes.outputs.pr_merge_base }}
315+
231316
# NOTE: Build jobs are intentionally split by platform rather than using a single
232317
# matrix. This allows each test job to depend only on its corresponding build,
233318
# so faster platforms can proceed through build & test without waiting for slower

‎.github/workflows/release.yml‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -158,6 +158,8 @@ jobs:
158158
steps:
159159
- name: Checkout Source
160160
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
161+
with:
162+
ref: ${{ inputs.git-tag }}
161163

162164
- name: Set up Python
163165
uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0

‎CONTRIBUTING.md‎

Lines changed: 92 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -19,6 +19,10 @@ Thank you for your interest in contributing to CUDA Python! Based on the type of
1919

2020
- [Contributing to CUDA Python](#contributing-to-cuda-python)
2121
- [Table of Contents](#table-of-contents)
22+
- [Cloning the repository](#cloning-the-repository)
23+
- [Recommended clone](#recommended-clone)
24+
- [Fixing an existing clone](#fixing-an-existing-clone)
25+
- [Symptoms of a bad clone](#symptoms-of-a-bad-clone)
2226
- [Type stubs for cuda.core](#type-stubs-for-cudacore)
2327
- [Pre-commit](#pre-commit)
2428
- [Signing Your Work](#signing-your-work)
@@ -34,6 +38,94 @@ Thank you for your interest in contributing to CUDA Python! Based on the type of
3438
- [Code coverage](#code-coverage)
3539

3640

41+
## Cloning the repository
42+
43+
Every package in this repository derives its version from git tags using
44+
[`setuptools-scm`](https://setuptools-scm.readthedocs.io/), so **how you clone
45+
determines whether you can build at all, and whether the version you build is
46+
correct.** Each package matches its own tag prefix:
47+
48+
| Package | Tag pattern |
49+
| --- | --- |
50+
| `cuda-bindings`, `cuda-python` | `v*` (e.g. `v13.3.1`) |
51+
| `cuda-core` | `cuda-core-v*` (e.g. `cuda-core-v1.1.0`) |
52+
| `cuda-pathfinder` | `cuda-pathfinder-v*` (e.g. `cuda-pathfinder-v1.6.0`) |
53+
54+
Each package sets `root = ".."` in its `[tool.setuptools_scm]` table, meaning the
55+
version is read from the *repository root* rather than the package directory. A
56+
working build therefore needs all of the following:
57+
58+
1. **A real git clone.** Source zips and GitHub "Download ZIP" archives have no
59+
git metadata and the build fails outright. (Tarballs produced by
60+
`git archive` do work, thanks to the `.git_archival.txt` substitutions
61+
configured in `.gitattributes`.)
62+
2. **The full repository**, not just the package subdirectory, because the
63+
version lookup walks up to the repository root.
64+
3. **Tags, reaching back at least as far as the most recent tag** matching the
65+
package you are building. `git describe` needs to find that tag; the history
66+
between it and your checkout must be present too.
67+
68+
### Recommended clone
69+
70+
The default `git clone` gives you everything you need:
71+
72+
```console
73+
$ git clone https://github.com/NVIDIA/cuda-python.git
74+
```
75+
76+
77+
78+
### Fixing an existing clone
79+
80+
If you already have a shallow clone:
81+
82+
```console
83+
$ git fetch --unshallow --tags
84+
```
85+
86+
If you are working from a personal fork, your fork's tags stop tracking upstream
87+
the moment new releases are cut, which silently yields a stale version. Fetch
88+
tags from upstream directly:
89+
90+
```console
91+
$ git remote add upstream https://github.com/NVIDIA/cuda-python.git
92+
$ git fetch --tags upstream
93+
```
94+
95+
Keep doing this periodically — a fork that was correct when you created it will
96+
drift.
97+
98+
### Symptoms of a bad clone
99+
100+
Only case 3 below reports an error. The first two fail *silently*, producing a
101+
wrong version that surfaces much later as a confusing dependency-resolution or
102+
version-check failure:
103+
104+
1. **No tags reachable.** The build succeeds and produces a version starting at
105+
`0.1.dev`: a `--depth 1` clone yields `0.1.dev1+g0d22cb444`, a full clone made
106+
with `--no-tags` yields `0.1.dev2114+g0d22cb444`. Installing `cuda-python`
107+
built this way then fails, because its `install_requires` pins
108+
`cuda-bindings` to that same bogus version.
109+
2. **Stale tags** (a fork that has not fetched upstream in a while): you get a
110+
plausible-looking but wrong version, e.g. `13.0.4.dev650+g0d22cb44` when the
111+
real latest tag is `v13.3.1`. Nothing warns you. Note there is no leading
112+
`v` — the tag prefix is stripped by `tag_regex`.
113+
3. **No git metadata** (source zip): the build fails with
114+
`LookupError: setuptools-scm was unable to detect version`.
115+
116+
As a last resort — for example when building inside a container that has no git
117+
history — you can bypass the lookup entirely:
118+
119+
```console
120+
$ SETUPTOOLS_SCM_PRETEND_VERSION_FOR_CUDA_CORE=1.1.0 pip install ./cuda_core
121+
```
122+
123+
The environment variable is suffixed with the distribution name, uppercased with
124+
hyphens replaced by underscores: `..._FOR_CUDA_BINDINGS`, `..._FOR_CUDA_CORE`,
125+
`..._FOR_CUDA_PATHFINDER`, `..._FOR_CUDA_PYTHON`. Use this only when you
126+
genuinely cannot provide tags; it is not a substitute for a correct clone.
127+
128+
37129
## Type stubs for cuda.core
38130

39131
`cuda.core` is a PEP 561-compliant package: it ships a `py.typed` marker and

‎cuda_bindings/docs/source/install.rst‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -120,11 +120,14 @@ Requirements
120120

121121
* CUDA Toolkit headers[^1]
122122
* CUDA Runtime static library[^2]
123+
* A git clone of the repository that includes tags[^3]
123124

124125
[^1]: User projects that ``cimport`` CUDA symbols in Cython must also use CUDA Toolkit (CTK) types as provided by the ``cuda.bindings`` major.minor version. This results in CTK headers becoming a transitive dependency of downstream projects through CUDA Python.
125126

126127
[^2]: The CUDA Runtime static library (``libcudart_static.a`` on Linux, ``cudart_static.lib`` on Windows) is part of the CUDA Toolkit. If using conda packages, it is contained in the ``cuda-cudart-static`` package.
127128

129+
[^3]: The version is derived from git tags via ``setuptools-scm``, so the clone must include tags reaching back to at least the latest ``v*`` tag. Clone with ``git clone https://github.com/NVIDIA/cuda-python.git``; do not use ``--depth`` or ``--no-tags``, since a shallow clone builds without error but produces a bogus version such as ``0.1.dev1+g0d22cb444``. See `Cloning the repository <https://github.com/NVIDIA/cuda-python/blob/main/CONTRIBUTING.md>`_ for details and recovery steps.
130+
128131
Source builds require that the provided CUDA headers are of the same major.minor version as the ``cuda.bindings`` you're trying to build. Despite this requirement, note that the minor version compatibility is still maintained. Use the ``CUDA_PATH`` (or ``CUDA_HOME``) environment variable to specify the location of your headers. If both are set, ``CUDA_PATH`` takes precedence. For example, if your headers are located in ``/usr/local/cuda/include``, then you should set ``CUDA_PATH`` with:
129132

130133
.. code-block:: console

‎cuda_core/build_hooks.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -17,6 +17,7 @@
1717
from pathlib import Path
1818

1919
from Cython.Build import cythonize
20+
from Cython.Compiler import Options as _CythonOptions
2021
from setuptools import Extension
2122
from setuptools import build_meta as _build_meta
2223

@@ -212,6 +213,7 @@ def get_sources(mod_name):
212213
nthreads = int(os.environ.get("CUDA_PYTHON_PARALLEL_LEVEL", os.cpu_count() // 2))
213214
compile_time_env = {"CUDA_CORE_BUILD_MAJOR": int(_determine_cuda_major_version())}
214215
compiler_directives = {"embedsignature": True, "warn.deprecated.IF": False, "freethreading_compatible": True}
216+
_CythonOptions.warning_errors = True
215217
if COMPILE_FOR_COVERAGE:
216218
compiler_directives["linetrace"] = True
217219
_extensions = cythonize(

‎cuda_core/cuda/core/__init__.py‎

Lines changed: 45 additions & 39 deletions
Original file line numberDiff line numberDiff line change
@@ -69,45 +69,51 @@ class _PatchedProperty(metaclass=_PatchedPropMeta):
6969

7070

7171
from cuda.core import checkpoint, system, utils
72-
from cuda.core._context import Context, ContextOptions
73-
from cuda.core._device import Device
74-
from cuda.core._device_resources import (
75-
DeviceResources,
76-
SMResource,
77-
SMResourceOptions,
78-
WorkqueueResource,
79-
WorkqueueResourceOptions,
80-
)
81-
from cuda.core._event import Event, EventOptions
82-
from cuda.core._graphics import GraphicsResource
83-
from cuda.core._host import Host
84-
from cuda.core._launch_config import LaunchConfig
85-
from cuda.core._launcher import launch
86-
from cuda.core._linker import Linker, LinkerOptions
87-
from cuda.core._memory import (
88-
Buffer,
89-
DeviceMemoryResource,
90-
DeviceMemoryResourceOptions,
91-
GraphMemoryResource,
92-
LegacyPinnedMemoryResource,
93-
ManagedBuffer,
94-
ManagedMemoryResource,
95-
ManagedMemoryResourceOptions,
96-
MemoryResource,
97-
PinnedMemoryResource,
98-
PinnedMemoryResourceOptions,
99-
VirtualMemoryResource,
100-
VirtualMemoryResourceOptions,
101-
)
102-
from cuda.core._module import Kernel, ObjectCode
103-
from cuda.core._program import Program, ProgramOptions
104-
from cuda.core._stream import (
105-
LEGACY_DEFAULT_STREAM,
106-
PER_THREAD_DEFAULT_STREAM,
107-
Stream,
108-
StreamOptions,
109-
)
110-
from cuda.core._tensor_map import TensorMapDescriptor, TensorMapDescriptorOptions
72+
from cuda.core._context import *
73+
from cuda.core._context import __all__ as _context_all
74+
from cuda.core._device import *
75+
from cuda.core._device import __all__ as _device_all
76+
from cuda.core._device_resources import *
77+
from cuda.core._device_resources import __all__ as _device_resources_all
78+
from cuda.core._event import *
79+
from cuda.core._event import __all__ as _event_all
80+
from cuda.core._graphics import *
81+
from cuda.core._graphics import __all__ as _graphics_all
82+
from cuda.core._host import *
83+
from cuda.core._host import __all__ as _host_all
84+
from cuda.core._launch_config import *
85+
from cuda.core._launch_config import __all__ as _launch_config_all
86+
from cuda.core._launcher import *
87+
from cuda.core._launcher import __all__ as _launcher_all
88+
from cuda.core._linker import *
89+
from cuda.core._linker import __all__ as _linker_all
90+
from cuda.core._memory import *
91+
from cuda.core._memory import __all__ as _memory_all
92+
from cuda.core._module import *
93+
from cuda.core._module import __all__ as _module_all
94+
from cuda.core._program import *
95+
from cuda.core._program import __all__ as _program_all
96+
from cuda.core._stream import *
97+
from cuda.core._stream import __all__ as _stream_all
98+
from cuda.core._tensor_map import *
99+
from cuda.core._tensor_map import __all__ as _tensor_map_all
100+
101+
__all__ = [
102+
*_context_all,
103+
*_device_all,
104+
*_device_resources_all,
105+
*_event_all,
106+
*_graphics_all,
107+
*_host_all,
108+
*_launch_config_all,
109+
*_launcher_all,
110+
*_linker_all,
111+
*_memory_all,
112+
*_module_all,
113+
*_program_all,
114+
*_stream_all,
115+
*_tensor_map_all,
116+
]
111117

112118
# isort: split
113119
# Texture/surface types live under the cuda.core.texture namespace (not the

0 commit comments

Comments
 (0)