From cc51b1334da685102399ac19539c168ab2f45611 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Tue, 22 Sep 2026 12:25:49 -0700 Subject: [PATCH 1/8] test: fix Orin coverage with CUDA 13.4 --- cuda_bindings/tests/nvml/test_pci.py | 11 +++++++---- cuda_core/tests/test_memory.py | 5 +++-- .../cuda_python_test_helpers/arch_check.py | 10 ++++++---- 3 files changed, 16 insertions(+), 10 deletions(-) diff --git a/cuda_bindings/tests/nvml/test_pci.py b/cuda_bindings/tests/nvml/test_pci.py index 877f9d2998a..429742444de 100644 --- a/cuda_bindings/tests/nvml/test_pci.py +++ b/cuda_bindings/tests/nvml/test_pci.py @@ -11,11 +11,14 @@ def test_discover_gpus(all_devices, subtests): for device in all_devices: - with subtests.test(device_index=nvml.device_get_index(device)): - pci_info = nvml.device_get_pci_info_v3(device) + with ( + subtests.test(device_index=nvml.device_get_index(device)), + unsupported_before(device, None), + contextlib.suppress(nvml.OperatingSystemError), + ): # Docs say this should be supported on PASCAL and later - with unsupported_before(device, None), contextlib.suppress(nvml.OperatingSystemError): - nvml.device_discover_gpus(pci_info.ptr) + pci_info = nvml.device_get_pci_info_v3(device) + nvml.device_discover_gpus(pci_info.ptr) def test_bridge_chip_hierarchy_t(): diff --git a/cuda_core/tests/test_memory.py b/cuda_core/tests/test_memory.py index c20e263262b..d81c4b95b74 100644 --- a/cuda_core/tests/test_memory.py +++ b/cuda_core/tests/test_memory.py @@ -1534,8 +1534,9 @@ def allocate_and_close(): allocate_and_close() free = handle_return(driver.cuMemGetInfo())[0] - # Current main leaks aligned_size per iteration; the fixed path stays near baseline. - assert baseline - free < aligned_size + # The broken path leaks aligned_size per iteration. Allow one allocation's + # worth of driver bookkeeping/caching while still detecting repeated leaks. + assert baseline - free < 2 * aligned_size def test_vmm_allocator_rdma_unsupported_exception(): diff --git a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py index 2eb0a61ffca..16655cc4a5c 100644 --- a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py +++ b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py @@ -11,17 +11,19 @@ @cache def hardware_supports_nvml(): - """Try the simplest NVML API to verify basic functionality. + """Verify that NVML supports the device lookup used by cuda.core. Returns False on platforms where NVML is unsupported (e.g. Jetson Orin). """ from cuda.bindings import nvml - from cuda.bindings._internal.utils import FunctionNotFoundError as NvmlSymbolNotFoundError # noqa: F401 nvml.init_v2() try: - nvml.system_get_driver_branch() - except (nvml.NotSupportedError, nvml.UnknownError): + if nvml.device_get_count_v2() == 0: + return False + device = nvml.device_get_handle_by_index_v2(0) + nvml.device_get_handle_by_uuid(nvml.device_get_uuid(device)) + except (nvml.NotFoundError, nvml.NotSupportedError, nvml.UnknownError): return False else: return True From bcbf8232101d773c0f36d2745d5c5b2d46b92e6c Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Wed, 23 Sep 2026 16:59:03 -0700 Subject: [PATCH 2/8] test: preserve partial NVML coverage on Orin --- cuda_core/tests/system/test_system_device.py | 21 ++++++++++++++++- cuda_core/tests/system/test_system_events.py | 4 +++- cuda_core/tests/system/test_system_system.py | 4 ++-- cuda_core/tests/test_device.py | 6 ++--- .../cuda_python_test_helpers/arch_check.py | 23 ++++++++++++++++++- 5 files changed, 50 insertions(+), 8 deletions(-) diff --git a/cuda_core/tests/system/test_system_device.py b/cuda_core/tests/system/test_system_device.py index b52a0cec6d7..05804d1ad5a 100644 --- a/cuda_core/tests/system/test_system_device.py +++ b/cuda_core/tests/system/test_system_device.py @@ -3,7 +3,11 @@ # SPDX-License-Identifier: Apache-2.0 -from cuda_python_test_helpers.arch_check import skip_if_nvml_unsupported, unsupported_before +from cuda_python_test_helpers.arch_check import ( + skip_if_nvml_device_apis_unsupported, + skip_if_nvml_unsupported, + unsupported_before, +) pytestmark = skip_if_nvml_unsupported @@ -86,6 +90,7 @@ def test_device_bar1_memory(subtests): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") +@skip_if_nvml_device_apis_unsupported def test_device_cpu_affinity(subtests): for device in system.Device.get_all_devices(): with subtests.test(device_index=device.index): @@ -97,6 +102,7 @@ def test_device_cpu_affinity(subtests): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") +@skip_if_nvml_device_apis_unsupported def test_affinity(subtests): for device in system.Device.get_all_devices(): for scope in typing.AffinityScope.__members__.values(): @@ -128,6 +134,7 @@ def test_numa_node_id(subtests): assert numa_node_id >= -1 +@skip_if_nvml_device_apis_unsupported def test_device_cuda_compute_capability(): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -166,6 +173,7 @@ def test_device_name(): assert len(name) > 0 +@skip_if_nvml_device_apis_unsupported def test_device_pci_info(subtests): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -272,6 +280,7 @@ def test_unpack_bitmask_single_value(): @pytest.mark.parallel_threads_limit(4) # timeouts are slow @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Events not supported on WSL or Windows") +@skip_if_nvml_device_apis_unsupported def test_register_events(): # This is not the world's greatest test. All of the events are pretty # infrequent and hard to simulate. So all we do here is register an event, @@ -311,6 +320,7 @@ def test_device_brand(): assert isinstance(brand, str) +@skip_if_nvml_device_apis_unsupported def test_device_pci_bus_id(): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -379,6 +389,7 @@ def test_c2c_mode_enabled(subtests): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Persistence mode not supported on WSL or Windows") @pytest.mark.thread_unsafe(reason="device persistence mode is global state") +@skip_if_nvml_device_apis_unsupported def test_persistence_mode_enabled(subtests): for device in system.Device.get_all_devices(): with subtests.test(device_index=device.index): @@ -499,6 +510,7 @@ def test_addressing_mode(subtests): assert addressing_mode is None or addressing_mode in typing.AddressingMode.__members__.values() +@skip_if_nvml_device_apis_unsupported def test_display_mode(): for device in system.Device.get_all_devices(): is_display_connected = device.is_display_connected @@ -562,6 +574,7 @@ def test_get_nearest_gpus(): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") +@skip_if_nvml_device_apis_unsupported def test_get_minor_number(): for device in system.Device.get_all_devices(): minor_number = device.minor_number @@ -684,6 +697,7 @@ def test_clock_event_reasons(subtests): assert all(isinstance(reason, typing.ClocksEventReasons) for reason in reasons) +@skip_if_nvml_device_apis_unsupported def test_fan(subtests): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -735,6 +749,7 @@ def test_fan(subtests): fan_info.set_default_speed() +@skip_if_nvml_device_apis_unsupported def test_cooler(subtests): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -757,6 +772,7 @@ def test_cooler(subtests): @pytest.mark.filterwarnings("ignore::DeprecationWarning") +@skip_if_nvml_device_apis_unsupported def test_temperature(subtests): for device in system.Device.get_all_devices(): device_index = device.index @@ -895,6 +911,7 @@ def test_pstates(subtests): assert isinstance(utilization.dec_threshold, int) +@skip_if_nvml_device_apis_unsupported def test_compute_running_processes(subtests): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -916,6 +933,7 @@ def test_compute_running_processes(subtests): proc.compute_instance_id # noqa: B018 +@skip_if_nvml_device_apis_unsupported def test_nvlink(subtests): for device in system.Device.get_all_devices(): device_index = device.index @@ -1012,6 +1030,7 @@ def test_mig(subtests): assert isinstance(mig_device, system.Device) +@skip_if_nvml_device_apis_unsupported def test_uuid(): for device in system.Device.get_all_devices(): uuid = device.uuid diff --git a/cuda_core/tests/system/test_system_events.py b/cuda_core/tests/system/test_system_events.py index e8b218325cc..2c4323f4419 100644 --- a/cuda_core/tests/system/test_system_events.py +++ b/cuda_core/tests/system/test_system_events.py @@ -3,7 +3,7 @@ # SPDX-License-Identifier: Apache-2.0 -from cuda_python_test_helpers.arch_check import skip_if_nvml_unsupported +from cuda_python_test_helpers.arch_check import skip_if_nvml_device_apis_unsupported, skip_if_nvml_unsupported pytestmark = skip_if_nvml_unsupported @@ -52,6 +52,7 @@ def test_pci_bus_id_from_gpu_id(gpu_id, expected): @pytest.mark.agent_authored(model="claude-opus-4.7") +@skip_if_nvml_device_apis_unsupported def test_system_event_device_resolves_pci_bus_id(): # Round-trip: pack pci_info with the inverse of _pci_bus_id_from_gpu_id, # then resolve Device through SystemEvent.device. @@ -77,6 +78,7 @@ def test_system_event_device_resolves_pci_bus_id(): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="System events not supported on WSL or Windows") +@skip_if_nvml_device_apis_unsupported def test_register_events(): # This is not the world's greatest test. All of the events are pretty # infrequent and hard to simulate. So all we do here is register an event, diff --git a/cuda_core/tests/system/test_system_system.py b/cuda_core/tests/system/test_system_system.py index 9fc600b05da..bc8869349ce 100644 --- a/cuda_core/tests/system/test_system_system.py +++ b/cuda_core/tests/system/test_system_system.py @@ -6,7 +6,7 @@ import os import pytest -from cuda_python_test_helpers.arch_check import skip_if_nvml_unsupported +from cuda_python_test_helpers.arch_check import skip_if_nvml_device_apis_unsupported, skip_if_nvml_unsupported from cuda.bindings import driver from cuda.core import Device as CudaDevice @@ -56,7 +56,7 @@ def test_nvml_version(): assert 0 <= ver_patch[0] <= 99 -@skip_if_nvml_unsupported +@skip_if_nvml_device_apis_unsupported def test_get_process_name(): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() diff --git a/cuda_core/tests/test_device.py b/cuda_core/tests/test_device.py index 6b3aca9dc73..7333630a01b 100644 --- a/cuda_core/tests/test_device.py +++ b/cuda_core/tests/test_device.py @@ -36,10 +36,10 @@ def test_to_system_device(deinit_cuda): device.to_system_device() pytest.skip("NVML support requires cuda.bindings version 12.9.6+ for CUDA 12.x or 13.2.0+ for CUDA 13.x") - from cuda_python_test_helpers.arch_check import hardware_supports_nvml + from cuda_python_test_helpers.arch_check import hardware_supports_nvml_device_apis - if not hardware_supports_nvml(): - pytest.skip("NVML not supported on this platform") + if not hardware_supports_nvml_device_apis(): + pytest.skip("NVML device APIs are incomplete or unavailable on this platform") from cuda.core.system import Device as SystemDevice diff --git a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py index 16655cc4a5c..5474513c9fa 100644 --- a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py +++ b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py @@ -11,12 +11,28 @@ @cache def hardware_supports_nvml(): - """Verify that NVML supports the device lookup used by cuda.core. + """Try the simplest NVML API to verify basic functionality. Returns False on platforms where NVML is unsupported (e.g. Jetson Orin). """ from cuda.bindings import nvml + nvml.init_v2() + try: + nvml.system_get_driver_branch() + except (nvml.NotSupportedError, nvml.UnknownError): + return False + else: + return True + finally: + nvml.shutdown() + + +@cache +def hardware_supports_nvml_device_apis(): + """Verify that NVML supports the device lookup required by cuda.core.""" + from cuda.bindings import nvml + nvml.init_v2() try: if nvml.device_get_count_v2() == 0: @@ -52,6 +68,11 @@ def _should_skip_nvml_tests() -> bool: reason="NVML support requires cuda.bindings version 12.9.6+ for CUDA 12.x or 13.2.0+ for CUDA 13.x, and hardware that supports NVML", ) +skip_if_nvml_device_apis_unsupported = pytest.mark.skipif( + _should_skip_nvml_tests() or not hardware_supports_nvml_device_apis(), + reason="NVML device APIs are incomplete or unavailable on this platform", +) + @contextmanager def unsupported_before(device, expected_device_arch): From f2b782298afbefb99772714d7cb653bc1631fcf0 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Wed, 30 Sep 2026 14:03:22 -0700 Subject: [PATCH 3/8] test: address Orin review feedback --- cuda_core/tests/system/test_system_device.py | 2 ++ cuda_core/tests/system/test_system_events.py | 2 ++ cuda_core/tests/test_memory.py | 38 ++++++++++++-------- 3 files changed, 28 insertions(+), 14 deletions(-) diff --git a/cuda_core/tests/system/test_system_device.py b/cuda_core/tests/system/test_system_device.py index 05804d1ad5a..e9eab5881c6 100644 --- a/cuda_core/tests/system/test_system_device.py +++ b/cuda_core/tests/system/test_system_device.py @@ -9,6 +9,8 @@ unsupported_before, ) +# Keep device-API gating on individual tests: system-level NVML queries in +# this module remain supported on platforms with partial device API support. pytestmark = skip_if_nvml_unsupported import array diff --git a/cuda_core/tests/system/test_system_events.py b/cuda_core/tests/system/test_system_events.py index 2c4323f4419..21b9d42d5ce 100644 --- a/cuda_core/tests/system/test_system_events.py +++ b/cuda_core/tests/system/test_system_events.py @@ -5,6 +5,8 @@ from cuda_python_test_helpers.arch_check import skip_if_nvml_device_apis_unsupported, skip_if_nvml_unsupported +# Keep device-API gating on individual tests so the pure event conversion and +# wrapping tests still run on platforms with partial device API support. pytestmark = skip_if_nvml_unsupported import helpers diff --git a/cuda_core/tests/test_memory.py b/cuda_core/tests/test_memory.py index d81c4b95b74..9e6d333a75e 100644 --- a/cuda_core/tests/test_memory.py +++ b/cuda_core/tests/test_memory.py @@ -2,6 +2,7 @@ # SPDX-License-Identifier: Apache-2.0 import ctypes +import functools import multiprocessing as mp import sys @@ -1506,6 +1507,25 @@ def __init__(self, size): assert ("release", NEW_HANDLE) in calls +def _vmm_allocate_and_close(mr, requested_size, grow): + buf = mr.allocate(requested_size) + if grow: + buf = mr.modify_allocation(buf, 2 * buf.size) + aligned_size = buf.size + buf.close() + return aligned_size + + +@functools.cache +def _warm_up_vmm_allocate_and_close(device_id, grow): + device = Device(device_id) + mr = VirtualMemoryResource( + device, + config=VirtualMemoryResourceOptions(handle_type="win32_kmt" if IS_WINDOWS else "posix_fd"), + ) + return _vmm_allocate_and_close(mr, 8 * 1024 * 1024, grow) + + @pytest.mark.thread_unsafe(reason="cuMemGetInfo measures process-wide free memory") @pytest.mark.parametrize("grow", [False, True], ids=["allocate", "grow"]) def test_vmm_allocate_close_does_not_leak(init_cuda, grow): @@ -1513,30 +1533,20 @@ def test_vmm_allocate_close_does_not_leak(init_cuda, grow): if not device.properties.virtual_memory_management_supported: pytest.skip("Virtual memory management is not supported on this device") + aligned_size = _warm_up_vmm_allocate_and_close(device.device_id, grow) mr = VirtualMemoryResource( device, config=VirtualMemoryResourceOptions(handle_type="win32_kmt" if IS_WINDOWS else "posix_fd"), ) requested_size = 8 * 1024 * 1024 - def allocate_and_close(): - buf = mr.allocate(requested_size) - if grow: - buf = mr.modify_allocation(buf, 2 * buf.size) - aligned_size = buf.size - buf.close() - return aligned_size - - aligned_size = allocate_and_close() # Warm up and learn the aligned allocation size. - baseline = handle_return(driver.cuMemGetInfo())[0] for _ in range(8): - allocate_and_close() + _vmm_allocate_and_close(mr, requested_size, grow) free = handle_return(driver.cuMemGetInfo())[0] - # The broken path leaks aligned_size per iteration. Allow one allocation's - # worth of driver bookkeeping/caching while still detecting repeated leaks. - assert baseline - free < 2 * aligned_size + # A leak would cost at least one allocation per iteration. + assert baseline - free < aligned_size def test_vmm_allocator_rdma_unsupported_exception(): From 91daeaa9d1f26c63a18fb396790798a5941d0169 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Thu, 1 Oct 2026 04:20:02 +0000 Subject: [PATCH 4/8] fix(cuda.core): handle partial NVML support on Orin Preserve UUIDs reported without a prefix and propagate unsupported NVLink queries before reading unpopulated field results. Keep zero-count behavior for devices without link zero, and cover UUID normalization, lookup errors, and NVLink error propagation with regressions. --- cuda_core/cuda/core/system/_device.pyi | 19 +++++-- cuda_core/cuda/core/system/_device.pyx | 32 ++++++++--- cuda_core/docs/source/release/1.2.1-notes.rst | 10 ++++ cuda_core/tests/system/test_system_nvlink.py | 57 +++++++++++++++++++ cuda_core/tests/system/test_system_uuid.py | 41 +++++++++++++ 5 files changed, 145 insertions(+), 14 deletions(-) create mode 100644 cuda_core/tests/system/test_system_nvlink.py create mode 100644 cuda_core/tests/system/test_system_uuid.py diff --git a/cuda_core/cuda/core/system/_device.pyi b/cuda_core/cuda/core/system/_device.pyi index 3e2a6bd018c..9e11781fbaf 100644 --- a/cuda_core/cuda/core/system/_device.pyi +++ b/cuda_core/cuda/core/system/_device.pyi @@ -1131,9 +1131,10 @@ class Device: device, as a 5 part hexadecimal string, that augments the immutable, board serial identifier. - In the upstream NVML C++ API, the UUID includes a ``gpu-`` or ``mig-`` - prefix. If you need a `uuid` without that prefix (for example, to - interact with CUDA), use the `uuid_without_prefix` property. + Returns the UUID exactly as reported by NVML. It usually includes a + ``GPU-``, ``MIG-``, or ``DLA-`` prefix, but some platforms report an + unprefixed UUID. To interact with CUDA, use the `uuid_without_prefix` + property. """ @property def uuid_without_prefix(self) -> str: @@ -1142,9 +1143,10 @@ class Device: device, as a 5 part hexadecimal string, that augments the immutable, board serial identifier. - In the upstream NVML C++ API, the UUID includes a ``gpu-`` or ``mig-`` - prefix. This property returns it without the prefix, to match the UUIDs - used in CUDA. If you need the prefix, use the `uuid` property. + Removes a ``GPU-``, ``MIG-``, or ``DLA-`` prefix when present, to match + the UUIDs used in CUDA. An already unprefixed UUID is returned + unchanged. For the UUID exactly as reported by NVML, use the `uuid` + property. """ @property def pci_bus_id(self) -> str: @@ -1574,6 +1576,11 @@ class Device: For devices with NVLink support. .. version-added:: 1.1.0 + + Raises + ------ + :class:`cuda.core.system.NotSupportedError` + If the device does not support NVLink queries. """ def get_nvlinks(self) -> Iterable[NvlinkInfo]: """ diff --git a/cuda_core/cuda/core/system/_device.pyx b/cuda_core/cuda/core/system/_device.pyx index c3bf23fe025..505dd695e81 100644 --- a/cuda_core/cuda/core/system/_device.pyx +++ b/cuda_core/cuda/core/system/_device.pyx @@ -244,9 +244,10 @@ cdef class Device: device, as a 5 part hexadecimal string, that augments the immutable, board serial identifier. - In the upstream NVML C++ API, the UUID includes a ``gpu-`` or ``mig-`` - prefix. If you need a `uuid` without that prefix (for example, to - interact with CUDA), use the `uuid_without_prefix` property. + Returns the UUID exactly as reported by NVML. It usually includes a + ``GPU-``, ``MIG-``, or ``DLA-`` prefix, but some platforms report an + unprefixed UUID. To interact with CUDA, use the `uuid_without_prefix` + property. """ return nvml.device_get_uuid(self._handle) @@ -257,12 +258,15 @@ cdef class Device: device, as a 5 part hexadecimal string, that augments the immutable, board serial identifier. - In the upstream NVML C++ API, the UUID includes a ``gpu-`` or ``mig-`` - prefix. This property returns it without the prefix, to match the UUIDs - used in CUDA. If you need the prefix, use the `uuid` property. + Removes a ``GPU-``, ``MIG-``, or ``DLA-`` prefix when present, to match + the UUIDs used in CUDA. An already unprefixed UUID is returned + unchanged. For the UUID exactly as reported by NVML, use the `uuid` + property. """ - # NVML UUIDs have a `gpu-` or `mig-` prefix. We remove that here. - return nvml.device_get_uuid(self._handle)[4:] + uuid = self.uuid + if uuid.startswith(("GPU-", "MIG-", "DLA-")): + return uuid[4:] + return uuid @property def pci_bus_id(self) -> str: @@ -914,7 +918,19 @@ cdef class Device: For devices with NVLink support. .. version-added:: 1.1.0 + + Raises + ------ + :class:`cuda.core.system.NotSupportedError` + If the device does not support NVLink queries. """ + # Orin's field query may succeed without populating its output. Check + # NVLink support through the native query before reading that output. + try: + nvml.device_get_nvlink_state(self._handle, 0) + except nvml.InvalidArgumentError: + # A device with no link 0 can still report a valid count of zero. + pass return self.get_field_values([FieldId.DEV_NVLINK_LINK_COUNT])[0].value def get_nvlinks(self) -> Iterable[NvlinkInfo]: diff --git a/cuda_core/docs/source/release/1.2.1-notes.rst b/cuda_core/docs/source/release/1.2.1-notes.rst index bd34d0f50c0..dedb5f9844a 100644 --- a/cuda_core/docs/source/release/1.2.1-notes.rst +++ b/cuda_core/docs/source/release/1.2.1-notes.rst @@ -9,6 +9,16 @@ Fixes and enhancements ---------------------- +- :attr:`system.Device.uuid_without_prefix` preserves UUIDs that NVML already + reports without a prefix, including on Orin. This also allows + :meth:`system.Device.to_cuda_device` to match those devices correctly. + (`#2947 `__) + +- :meth:`system.Device.get_nvlink_count` and :meth:`system.Device.get_nvlinks` + now raise :class:`system.NotSupportedError` when Orin does not support + NVLink queries, instead of reading an unpopulated NVML field result. + (`#2947 `__) + - :meth:`Buffer.from_handle` no longer requires a current CUDA context when the memory resource reports ``is_device_accessible`` as ``False``. Host-only memory records no deallocation stream, so creating and closing such buffers diff --git a/cuda_core/tests/system/test_system_nvlink.py b/cuda_core/tests/system/test_system_nvlink.py new file mode 100644 index 00000000000..50e44d3451f --- /dev/null +++ b/cuda_core/tests/system/test_system_nvlink.py @@ -0,0 +1,57 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import pytest + +from cuda.core import system + + +def _nvlink_count_field(nvml, count): + field = nvml.FieldValue() + field.field_id = int(nvml.FieldId.DEV_NVLINK_LINK_COUNT) + field.nvml_return = int(nvml.Return.SUCCESS) + field.value_type = int(nvml.ValueType.UNSIGNED_INT) + field.value.ui_val[0] = count + return field + + +@pytest.mark.skipif(not system.CUDA_BINDINGS_NVML_IS_COMPATIBLE, reason="Compatible NVML bindings are required") +@pytest.mark.thread_unsafe(reason="Temporarily replaces process-global NVML functions") +@pytest.mark.parametrize("method", ["get_nvlink_count", "get_nvlinks"]) +@pytest.mark.agent_authored(model="gpt-6") +def test_nvlink_queries_propagate_not_supported(monkeypatch, method): + from cuda.bindings import nvml + + def unsupported_state(_handle, _link): + raise nvml.NotSupportedError(nvml.Return.ERROR_NOT_SUPPORTED) + + # A successful field result alone does not establish NVLink support. + field = _nvlink_count_field(nvml, 0) + monkeypatch.setattr(nvml, "device_get_nvlink_state", unsupported_state) + monkeypatch.setattr(nvml, "device_get_field_values", lambda _handle, _fields: field) + device = system.Device.__new__(system.Device) + + with pytest.raises(system.NotSupportedError): + result = getattr(device, method)() + if method == "get_nvlinks": + list(result) + + +@pytest.mark.skipif(not system.CUDA_BINDINGS_NVML_IS_COMPATIBLE, reason="Compatible NVML bindings are required") +@pytest.mark.thread_unsafe(reason="Temporarily replaces process-global NVML functions") +@pytest.mark.parametrize(("state", "count"), [(False, 3), (True, 3), (None, 0)]) +@pytest.mark.agent_authored(model="gpt-6") +def test_nvlink_count_handles_disabled_or_absent_link(monkeypatch, state, count): + from cuda.bindings import nvml + + def link_state(_handle, _link): + if state is None: + raise nvml.InvalidArgumentError(nvml.Return.ERROR_INVALID_ARGUMENT) + return state + + field = _nvlink_count_field(nvml, count) + monkeypatch.setattr(nvml, "device_get_nvlink_state", link_state) + monkeypatch.setattr(nvml, "device_get_field_values", lambda _handle, _fields: field) + device = system.Device.__new__(system.Device) + + assert device.get_nvlink_count() == count diff --git a/cuda_core/tests/system/test_system_uuid.py b/cuda_core/tests/system/test_system_uuid.py new file mode 100644 index 00000000000..c31fed0375c --- /dev/null +++ b/cuda_core/tests/system/test_system_uuid.py @@ -0,0 +1,41 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import pytest + +from cuda.core import system + + +@pytest.mark.skipif(not system.CUDA_BINDINGS_NVML_IS_COMPATIBLE, reason="Compatible NVML bindings are required") +@pytest.mark.thread_unsafe(reason="Temporarily replaces a process-global NVML function") +@pytest.mark.parametrize("prefix", ["GPU-", "MIG-", "DLA-", ""]) +@pytest.mark.agent_authored(model="gpt-6") +def test_device_uuid_preserves_unprefixed_value(monkeypatch, prefix): + from cuda.bindings import nvml + + expected_uuid = "abcdef12-abcd-0123-4567-1234567890ab" + raw_uuid = prefix + expected_uuid + monkeypatch.setattr(nvml, "device_get_uuid", lambda _handle: raw_uuid) + device = system.Device.__new__(system.Device) + + assert device.uuid == raw_uuid + assert device.uuid_without_prefix == expected_uuid + + +@pytest.mark.skipif(not system.CUDA_BINDINGS_NVML_IS_COMPATIBLE, reason="Compatible NVML bindings are required") +@pytest.mark.thread_unsafe(reason="Temporarily replaces a process-global NVML function") +@pytest.mark.parametrize( + ("exception_name", "status_name"), + [("NotFoundError", "ERROR_NOT_FOUND"), ("NotSupportedError", "ERROR_NOT_SUPPORTED")], +) +@pytest.mark.agent_authored(model="gpt-6") +def test_to_system_device_propagates_uuid_lookup_error(init_cuda, monkeypatch, exception_name, status_name): + from cuda.bindings import nvml + + def unavailable_lookup(_uuid): + raise getattr(nvml, exception_name)(getattr(nvml.Return, status_name)) + + monkeypatch.setattr(nvml, "device_get_handle_by_uuid", unavailable_lookup) + + with pytest.raises(getattr(system, exception_name)): + init_cuda.to_system_device() From 2ed2554545dc92930e9e100dbbffe01b109c686f Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Thu, 1 Oct 2026 04:21:57 +0000 Subject: [PATCH 5/8] test(cuda.core): verify unsupported NVML caller behavior Replace the UUID-based device-wide skip with narrow checks of the actual API result. Exercise supported Orin queries, validate precise unsupported errors, preserve CUDA visibility and MIG coverage, and require successful UUID matching when a CUDA counterpart exists. --- cuda_core/tests/system/test_system_device.py | 276 ++++++++++++------ cuda_core/tests/system/test_system_events.py | 62 ++-- cuda_core/tests/system/test_system_system.py | 18 +- cuda_core/tests/test_device.py | 19 +- .../cuda_python_test_helpers/arch_check.py | 24 -- 5 files changed, 236 insertions(+), 163 deletions(-) diff --git a/cuda_core/tests/system/test_system_device.py b/cuda_core/tests/system/test_system_device.py index e9eab5881c6..cf5d0ceb42b 100644 --- a/cuda_core/tests/system/test_system_device.py +++ b/cuda_core/tests/system/test_system_device.py @@ -4,19 +4,17 @@ from cuda_python_test_helpers.arch_check import ( - skip_if_nvml_device_apis_unsupported, skip_if_nvml_unsupported, unsupported_before, ) -# Keep device-API gating on individual tests: system-level NVML queries in -# this module remain supported on platforms with partial device API support. pytestmark = skip_if_nvml_unsupported import array import multiprocessing import os import re +from uuid import UUID import helpers import pytest @@ -37,30 +35,44 @@ def check_gpu_available(): pytest.skip("No GPUs available to run device tests", allow_module_level=True) +def _cuda_visible_system_devices(): + indexed_devices = {device.uuid_without_prefix: device for device in system.Device.get_all_devices()} + for cuda_device in CudaDevice.get_all_devices(): + try: + device = cuda_device.to_system_device() + except nvml.NotFoundError: + # Orin supports index lookup but not UUID lookup. Keep the same + # CUDA-visible devices without blocking unrelated NVML queries. + device = indexed_devices[cuda_device.uuid] + yield device + + def test_device_count(): assert system.Device.get_device_count() == system.get_num_devices() -def test_to_cuda_device(): - from cuda.core import Device as CudaDevice +@pytest.mark.agent_authored(model="gpt-6") +def test_to_cuda_device(init_cuda): + cuda_uuids = {device.uuid for device in CudaDevice.get_all_devices()} for device in system.Device.get_all_devices(): - try: - cuda_device = device.to_cuda_device() - except RuntimeError: - # Not all physical NVML devices may have a matching CUDA device - # when MIG is involved. + if device.uuid_without_prefix not in cuda_uuids: + # A physical MIG device may have no CUDA-visible counterpart. + with pytest.raises(RuntimeError): + device.to_cuda_device() continue + cuda_device = device.to_cuda_device() assert isinstance(cuda_device, CudaDevice) assert cuda_device.uuid == device.uuid_without_prefix - # Technically, this test will only work with PCI devices, but are there - # non-PCI devices we need to support? - # CUDA only returns a 2-byte PCI bus ID domain, whereas NVML returns a # 4-byte domain - assert cuda_device.pci_bus_id == device.pci_info.bus_id[4:] + try: + pci_info = device.pci_info + except nvml.NotSupportedError: + continue + assert cuda_device.pci_bus_id == pci_info.bus_id[4:] def test_device_architecture(): @@ -92,19 +104,25 @@ def test_device_bar1_memory(subtests): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") -@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="gpt-6") def test_device_cpu_affinity(subtests): for device in system.Device.get_all_devices(): with subtests.test(device_index=device.index): - with unsupported_before(device, typing.DeviceArch.KEPLER): + try: affinity = device.get_cpu_affinity(typing.AffinityScope.NODE) + except nvml.NotSupportedError: + continue assert isinstance(affinity, list) - os.sched_setaffinity(0, affinity) - assert os.sched_getaffinity(0) == set(affinity) + original_affinity = os.sched_getaffinity(0) + try: + os.sched_setaffinity(0, affinity) + assert os.sched_getaffinity(0) == set(affinity) + finally: + os.sched_setaffinity(0, original_affinity) @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") -@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="gpt-6") def test_affinity(subtests): for device in system.Device.get_all_devices(): for scope in typing.AffinityScope.__members__.values(): @@ -113,18 +131,24 @@ def test_affinity(subtests): affinity_scope=scope.value, affinity_api="get_cpu_affinity", ): - with unsupported_before(device, typing.DeviceArch.KEPLER): + try: affinity = device.get_cpu_affinity(scope) - assert isinstance(affinity, list) + except nvml.NotSupportedError: + pass + else: + assert isinstance(affinity, list) with subtests.test( device_index=device.index, affinity_scope=scope.value, affinity_api="get_memory_affinity", ): - with unsupported_before(device, typing.DeviceArch.KEPLER): + try: affinity = device.get_memory_affinity(scope) - assert isinstance(affinity, list) + except nvml.NotSupportedError: + pass + else: + assert isinstance(affinity, list) def test_numa_node_id(subtests): @@ -136,11 +160,13 @@ def test_numa_node_id(subtests): assert numa_node_id >= -1 -@skip_if_nvml_device_apis_unsupported -def test_device_cuda_compute_capability(): - for cuda_device in CudaDevice.get_all_devices(): - device = cuda_device.to_system_device() - cuda_compute_capability = device.cuda_compute_capability +@pytest.mark.agent_authored(model="gpt-6") +def test_device_cuda_compute_capability(init_cuda): + for device in _cuda_visible_system_devices(): + try: + cuda_compute_capability = device.cuda_compute_capability + except nvml.NotSupportedError: + continue assert isinstance(cuda_compute_capability, tuple) assert len(cuda_compute_capability) == 2 assert all(isinstance(i, int) for i in cuda_compute_capability) @@ -175,12 +201,14 @@ def test_device_name(): assert len(name) > 0 -@skip_if_nvml_device_apis_unsupported -def test_device_pci_info(subtests): - for cuda_device in CudaDevice.get_all_devices(): - device = cuda_device.to_system_device() +@pytest.mark.agent_authored(model="gpt-6") +def test_device_pci_info(init_cuda, subtests): + for device in _cuda_visible_system_devices(): with subtests.test(device_index=device.index): - pci_info = device.pci_info + try: + pci_info = device.pci_info + except nvml.NotSupportedError: + continue assert isinstance(pci_info, _device.PciInfo) assert isinstance(pci_info.bus_id, str) @@ -247,13 +275,16 @@ def test_device_serial(subtests): assert len(serial) > 0 +@pytest.mark.agent_authored(model="gpt-6") def test_device_uuid_without_prefix(): for device in system.Device.get_all_devices(): uuid = device.uuid_without_prefix assert isinstance(uuid, str) - # Expands to GPU-8hex-4hex-4hex-4hex-12hex, where 8hex means 8 consecutive - # hex characters, e.g.: "GPU-abcdef12-abcd-0123-4567-1234567890ab" + assert str(UUID(uuid)) == uuid.lower() + raw_uuid = device.uuid + expected = raw_uuid[4:] if raw_uuid.startswith(("GPU-", "MIG-", "DLA-")) else raw_uuid + assert uuid == expected @pytest.mark.parametrize( @@ -282,7 +313,7 @@ def test_unpack_bitmask_single_value(): @pytest.mark.parallel_threads_limit(4) # timeouts are slow @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Events not supported on WSL or Windows") -@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="gpt-6") def test_register_events(): # This is not the world's greatest test. All of the events are pretty # infrequent and hard to simulate. So all we do here is register an event, @@ -292,22 +323,34 @@ def test_register_events(): # Also, some hardware doesn't support any event types. for device in system.Device.get_all_devices(): - supported_events = device.get_supported_event_types() + try: + supported_events = device.get_supported_event_types() + except nvml.NotSupportedError: + continue assert isinstance(supported_events, list) assert all(isinstance(ev, typing.EventType) for ev in supported_events) for device in system.Device.get_all_devices(): - events = device.register_events(["xid_critical_error"]) + try: + events = device.register_events(["xid_critical_error"]) + except nvml.NotSupportedError: + continue with pytest.raises(system.TimeoutError): events.wait(timeout_ms=500) for device in system.Device.get_all_devices(): - events = device.register_events([typing.EventType.XID_CRITICAL_ERROR]) + try: + events = device.register_events([typing.EventType.XID_CRITICAL_ERROR]) + except nvml.NotSupportedError: + continue with pytest.raises(system.TimeoutError): events.wait(timeout_ms=500) for device in system.Device.get_all_devices(): - events = device.register_events([]) + try: + events = device.register_events([]) + except nvml.NotSupportedError: + continue with pytest.raises(system.TimeoutError): events.wait(timeout_ms=500) @@ -322,11 +365,13 @@ def test_device_brand(): assert isinstance(brand, str) -@skip_if_nvml_device_apis_unsupported -def test_device_pci_bus_id(): - for cuda_device in CudaDevice.get_all_devices(): - device = cuda_device.to_system_device() - pci_bus_id = device.pci_info.bus_id +@pytest.mark.agent_authored(model="gpt-6") +def test_device_pci_bus_id(init_cuda): + for device in _cuda_visible_system_devices(): + try: + pci_bus_id = device.pci_info.bus_id + except nvml.NotSupportedError: + continue assert isinstance(pci_bus_id, str) new_device = system.Device(pci_bus_id=pci_bus_id) @@ -391,14 +436,19 @@ def test_c2c_mode_enabled(subtests): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Persistence mode not supported on WSL or Windows") @pytest.mark.thread_unsafe(reason="device persistence mode is global state") -@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="gpt-6") def test_persistence_mode_enabled(subtests): for device in system.Device.get_all_devices(): with subtests.test(device_index=device.index): - is_enabled = device.is_persistence_mode_enabled + try: + is_enabled = device.is_persistence_mode_enabled + except nvml.NotSupportedError: + continue assert isinstance(is_enabled, bool) try: device.is_persistence_mode_enabled = False + except nvml.NotSupportedError: + continue except nvml.NoPermissionError as e: pytest.xfail(f"nvml.NoPermissionError: {e}") try: @@ -512,14 +562,22 @@ def test_addressing_mode(subtests): assert addressing_mode is None or addressing_mode in typing.AddressingMode.__members__.values() -@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="gpt-6") def test_display_mode(): for device in system.Device.get_all_devices(): - is_display_connected = device.is_display_connected - assert isinstance(is_display_connected, bool) + try: + is_display_connected = device.is_display_connected + except nvml.NotSupportedError: + pass + else: + assert isinstance(is_display_connected, bool) - is_display_active = device.is_display_active - assert isinstance(is_display_active, bool) + try: + is_display_active = device.is_display_active + except nvml.NotSupportedError: + pass + else: + assert isinstance(is_display_active, bool) def test_repair_status(subtests): @@ -576,10 +634,13 @@ def test_get_nearest_gpus(): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") -@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="gpt-6") def test_get_minor_number(): for device in system.Device.get_all_devices(): - minor_number = device.minor_number + try: + minor_number = device.minor_number + except nvml.NotSupportedError: + continue assert isinstance(minor_number, int) assert minor_number >= 0 @@ -699,16 +760,19 @@ def test_clock_event_reasons(subtests): assert all(isinstance(reason, typing.ClocksEventReasons) for reason in reasons) -@skip_if_nvml_device_apis_unsupported -def test_fan(subtests): - for cuda_device in CudaDevice.get_all_devices(): - device = cuda_device.to_system_device() +@pytest.mark.thread_unsafe(reason="device fan settings are global state") +@pytest.mark.agent_authored(model="gpt-6") +def test_fan(init_cuda, subtests): + for device in _cuda_visible_system_devices(): device_index = device.index num_fans = None # The fan APIs are only supported on discrete devices with fans, # but when they are not available `device.num_fans` returns 0. with subtests.test(device_index=device_index, fan_api="get_num_fans"): - value = device.num_fans + try: + value = device.num_fans + except nvml.NotSupportedError: + continue assert isinstance(value, int) assert value >= 0 num_fans = value @@ -751,18 +815,24 @@ def test_fan(subtests): fan_info.set_default_speed() -@skip_if_nvml_device_apis_unsupported -def test_cooler(subtests): - for cuda_device in CudaDevice.get_all_devices(): - device = cuda_device.to_system_device() +@pytest.mark.agent_authored(model="gpt-6") +def test_cooler(init_cuda, subtests): + for device in _cuda_visible_system_devices(): with subtests.test(device_index=device.index): # The cooler APIs are only supported on discrete devices with fans, # but when they are not available `device.num_fans` returns 0. - if device.num_fans == 0: - pytest.skip("Device has no coolers to test") + try: + num_fans = device.num_fans + except nvml.NotSupportedError: + pass + else: + if num_fans == 0: + pytest.skip("Device has no coolers to test") - with unsupported_before(device, DeviceArch.MAXWELL): + try: cooler_info = device.cooler + except nvml.NotSupportedError: + continue assert isinstance(cooler_info, _device.CoolerInfo) @@ -774,7 +844,7 @@ def test_cooler(subtests): @pytest.mark.filterwarnings("ignore::DeprecationWarning") -@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="gpt-6") def test_temperature(subtests): for device in system.Device.get_all_devices(): device_index = device.index @@ -787,9 +857,13 @@ def test_temperature(subtests): continue with subtests.test(device_index=device_index, temperature_api="get_sensor"): - sensor = temperature.get_sensor() - assert isinstance(sensor, int) - assert sensor >= 0 + try: + sensor = temperature.get_sensor() + except nvml.NotSupportedError: + pass + else: + assert isinstance(sensor, int) + assert sensor >= 0 # By docs, should be supported on KEPLER or newer, but experimentally, # is also unsupported on other hardware. @@ -802,21 +876,28 @@ def test_temperature(subtests): temperature_api="get_threshold", threshold=threshold.value, ): - with unsupported_before(device, None): + try: t = temperature.get_threshold(threshold) + except nvml.NotSupportedError: + continue assert isinstance(t, int) assert t >= 0 with subtests.test(device_index=device_index, temperature_api="margin"): - with unsupported_before(device, None): + try: margin = temperature.margin - assert isinstance(margin, int) - assert margin >= 0 + except nvml.NotSupportedError: + pass + else: + assert isinstance(margin, int) + assert margin >= 0 thermals = None with subtests.test(device_index=device_index, temperature_api="get_thermal_settings"): - with unsupported_before(device, None): + try: value = temperature.get_thermal_settings(typing.ThermalTarget.ALL) + except nvml.NotSupportedError: + continue assert isinstance(value, _device.ThermalSettings) thermals = value if thermals is None: @@ -913,13 +994,14 @@ def test_pstates(subtests): assert isinstance(utilization.dec_threshold, int) -@skip_if_nvml_device_apis_unsupported -def test_compute_running_processes(subtests): - for cuda_device in CudaDevice.get_all_devices(): - device = cuda_device.to_system_device() +@pytest.mark.agent_authored(model="gpt-6") +def test_compute_running_processes(init_cuda, subtests): + for device in _cuda_visible_system_devices(): with subtests.test(device_index=device.index): - with unsupported_before(device, "FERMI"): + try: processes = device.compute_running_processes + except nvml.NotSupportedError: + continue assert isinstance(processes, list) for proc in processes: assert isinstance(proc, _device.ProcessInfo) @@ -935,19 +1017,20 @@ def test_compute_running_processes(subtests): proc.compute_instance_id # noqa: B018 -@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="gpt-6") def test_nvlink(subtests): for device in system.Device.get_all_devices(): device_index = device.index link_count = 0 - with ( - subtests.test(device_index=device_index, nvlink_api="get_nvlink_count"), - unsupported_before(device, None), - ): - value = device.get_nvlink_count() - assert isinstance(value, int) - assert value >= 0 - link_count = value + with subtests.test(device_index=device_index, nvlink_api="get_nvlink_count"): + try: + value = device.get_nvlink_count() + except nvml.NotSupportedError: + pass + else: + assert isinstance(value, int) + assert value >= 0 + link_count = value for link in range(link_count): with subtests.test(device_index=device_index, nvlink_api="get_nvlink", link_index=link): @@ -969,11 +1052,11 @@ def test_nvlink(subtests): assert all(isinstance(i, int) for i in version) nvlink_infos = [] - with ( - subtests.test(device_index=device_index, nvlink_api="get_nvlinks"), - unsupported_before(device, None), - ): - nvlink_infos = list(device.get_nvlinks()) + with subtests.test(device_index=device_index, nvlink_api="get_nvlinks"): + try: + nvlink_infos = list(device.get_nvlinks()) + except nvml.NotSupportedError: + continue for link, nvlink_info in enumerate(nvlink_infos): with subtests.test(device_index=device_index, nvlink_api="get_nvlinks", link_index=link): @@ -1032,10 +1115,11 @@ def test_mig(subtests): assert isinstance(mig_device, system.Device) -@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="gpt-6") def test_uuid(): for device in system.Device.get_all_devices(): uuid = device.uuid assert isinstance(uuid, str) - assert uuid.startswith(("GPU-", "MIG-", "DLA-")) + normalized = uuid[4:] if uuid.startswith(("GPU-", "MIG-", "DLA-")) else uuid + assert str(UUID(normalized)) == normalized.lower() assert uuid == device.uuid diff --git a/cuda_core/tests/system/test_system_events.py b/cuda_core/tests/system/test_system_events.py index 21b9d42d5ce..de9758ad853 100644 --- a/cuda_core/tests/system/test_system_events.py +++ b/cuda_core/tests/system/test_system_events.py @@ -3,16 +3,13 @@ # SPDX-License-Identifier: Apache-2.0 -from cuda_python_test_helpers.arch_check import skip_if_nvml_device_apis_unsupported, skip_if_nvml_unsupported +from cuda_python_test_helpers.arch_check import skip_if_nvml_unsupported -# Keep device-API gating on individual tests so the pure event conversion and -# wrapping tests still run on platforms with partial device API support. pytestmark = skip_if_nvml_unsupported import helpers import pytest -from cuda.core import Device as CudaDevice from cuda.core import system from cuda.core.system import typing @@ -53,21 +50,32 @@ def test_pci_bus_id_from_gpu_id(gpu_id, expected): assert _pci_bus_id_from_gpu_id(gpu_id) == expected -@pytest.mark.agent_authored(model="claude-opus-4.7") -@skip_if_nvml_device_apis_unsupported -def test_system_event_device_resolves_pci_bus_id(): - # Round-trip: pack pci_info with the inverse of _pci_bus_id_from_gpu_id, - # then resolve Device through SystemEvent.device. - cuda_devices = list(CudaDevice.get_all_devices()) - if not cuda_devices: - pytest.skip("No CUDA devices available") - - for cuda_device in cuda_devices: - device = cuda_device.to_system_device() - pci = device.pci_info - if pci.domain > 0xFFFF: - pytest.skip(f"PCI domain {pci.domain:#x} does not fit in a packed gpu_id") - gpu_id = (pci.domain << 16) | (pci.bus << 8) | (pci.device & 0xFF) +@pytest.mark.agent_authored(model="gpt-6") +def test_system_event_device_resolves_pci_bus_id(init_cuda): + devices = list(system.Device.get_all_devices()) + if not devices: + pytest.skip("No NVML devices available") + + for device in devices: + try: + original_pci = device.pci_info + except system.NotSupportedError: + # Orin supports PCI lookup but needs CUDA to supply the PCI bus ID. + cuda_pci_bus_id = device.to_cuda_device().pci_bus_id + domain_string, bus_string, device_function = cuda_pci_bus_id.split(":") + device_string, function_string = device_function.split(".") + domain = int(domain_string, 16) + bus = int(bus_string, 16) + pci_device = int(device_string, 16) + assert int(function_string, 16) == 0 + expected_pci_bus_id = f"{domain:08X}:{bus:02X}:{pci_device:02X}.0" + else: + domain, bus, pci_device = original_pci.domain, original_pci.bus, original_pci.device + expected_pci_bus_id = original_pci.bus_id + + if domain > 0xFFFF: + pytest.skip(f"PCI domain {domain:#x} does not fit in a packed gpu_id") + gpu_id = (domain << 16) | (bus << 8) | pci_device event_data = nvml.SystemEventData_v1(1) event_data.event_type = nvml.SystemEventType.GPU_DRIVER_BIND @@ -75,12 +83,21 @@ def test_system_event_device_resolves_pci_bus_id(): event = SystemEvent(event_data) resolved_device = event.device - assert resolved_device.pci_bus_id == device.pci_bus_id + assert resolved_device.uuid == device.uuid assert resolved_device.index == device.index + # PCI lookup can work even when the PCI information query is unsupported. + try: + pci = resolved_device.pci_info + except system.NotSupportedError: + continue + assert (pci.domain, pci.bus, pci.device) == (domain, bus, pci_device) + assert pci.bus_id == expected_pci_bus_id + assert resolved_device.pci_bus_id == expected_pci_bus_id + @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="System events not supported on WSL or Windows") -@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="gpt-6") def test_register_events(): # This is not the world's greatest test. All of the events are pretty # infrequent and hard to simulate. So all we do here is register an event, @@ -91,6 +108,9 @@ def test_register_events(): try: events = system.register_events([typing.SystemEventType.UNBIND]) + except system.NotSupportedError: + # The documented outcome when none of the requested events are supported. + return except system.UnknownError: pytest.skip("system events may only be registered once per process") diff --git a/cuda_core/tests/system/test_system_system.py b/cuda_core/tests/system/test_system_system.py index bc8869349ce..d47fd1cfc35 100644 --- a/cuda_core/tests/system/test_system_system.py +++ b/cuda_core/tests/system/test_system_system.py @@ -6,10 +6,9 @@ import os import pytest -from cuda_python_test_helpers.arch_check import skip_if_nvml_device_apis_unsupported, skip_if_nvml_unsupported +from cuda_python_test_helpers.arch_check import skip_if_nvml_unsupported from cuda.bindings import driver -from cuda.core import Device as CudaDevice from cuda.core import system from cuda.core._utils.cuda_utils import handle_return @@ -56,17 +55,10 @@ def test_nvml_version(): assert 0 <= ver_patch[0] <= 99 -@skip_if_nvml_device_apis_unsupported -def test_get_process_name(): - for cuda_device in CudaDevice.get_all_devices(): - device = cuda_device.to_system_device() - _ = device.compute_running_processes - - try: - process_name = system.get_process_name(os.getpid()) - except system.NotFoundError: - pytest.skip("Process not found") - +@skip_if_nvml_unsupported +@pytest.mark.agent_authored(model="gpt-6") +def test_get_process_name(init_cuda): + process_name = system.get_process_name(os.getpid()) assert isinstance(process_name, str) assert "python" in process_name diff --git a/cuda_core/tests/test_device.py b/cuda_core/tests/test_device.py index 7333630a01b..6f48fd11820 100644 --- a/cuda_core/tests/test_device.py +++ b/cuda_core/tests/test_device.py @@ -26,24 +26,25 @@ def test_device_init_disabled(): cuda.core._device.DeviceProperties() # Ensure back door is locked. -def test_to_system_device(deinit_cuda): +@pytest.mark.agent_authored(model="gpt-6") +def test_to_system_device(init_cuda): from cuda.core.system import _system - device = Device() + device = init_cuda if not _system.CUDA_BINDINGS_NVML_IS_COMPATIBLE: with pytest.raises(RuntimeError): device.to_system_device() - pytest.skip("NVML support requires cuda.bindings version 12.9.6+ for CUDA 12.x or 13.2.0+ for CUDA 13.x") - - from cuda_python_test_helpers.arch_check import hardware_supports_nvml_device_apis - - if not hardware_supports_nvml_device_apis(): - pytest.skip("NVML device APIs are incomplete or unavailable on this platform") + return + from cuda.core import system from cuda.core.system import Device as SystemDevice - system_device = device.to_system_device() + try: + system_device = device.to_system_device() + except system.NotFoundError: + # Orin enumerates NVML devices but does not support lookup by UUID. + return assert isinstance(system_device, SystemDevice) assert system_device.uuid_without_prefix == device.uuid diff --git a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py index 5474513c9fa..0b687c3758d 100644 --- a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py +++ b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py @@ -28,25 +28,6 @@ def hardware_supports_nvml(): nvml.shutdown() -@cache -def hardware_supports_nvml_device_apis(): - """Verify that NVML supports the device lookup required by cuda.core.""" - from cuda.bindings import nvml - - nvml.init_v2() - try: - if nvml.device_get_count_v2() == 0: - return False - device = nvml.device_get_handle_by_index_v2(0) - nvml.device_get_handle_by_uuid(nvml.device_get_uuid(device)) - except (nvml.NotFoundError, nvml.NotSupportedError, nvml.UnknownError): - return False - else: - return True - finally: - nvml.shutdown() - - def _should_skip_nvml_tests() -> bool: """Return True if NVML tests should be skipped on this system. @@ -68,11 +49,6 @@ def _should_skip_nvml_tests() -> bool: reason="NVML support requires cuda.bindings version 12.9.6+ for CUDA 12.x or 13.2.0+ for CUDA 13.x, and hardware that supports NVML", ) -skip_if_nvml_device_apis_unsupported = pytest.mark.skipif( - _should_skip_nvml_tests() or not hardware_supports_nvml_device_apis(), - reason="NVML device APIs are incomplete or unavailable on this platform", -) - @contextmanager def unsupported_before(device, expected_device_arch): From 7554cd70d37e6a9ccfea22ae7b243433e6f3d237 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Thu, 1 Oct 2026 16:42:29 +0000 Subject: [PATCH 6/8] test(cuda.core): restore skips for unsupported NVML APIs Restore the conservative device capability gate and the original device and event test bodies instead of accepting unsupported calls as successful tests. Keep positive UUID mapping checks, report unsupported PCI validation as a separate skipped subtest, restore the process-name NotFound skip, and preserve affinity cleanup and fan serialization. --- cuda_core/tests/system/test_system_device.py | 250 ++++++------------ cuda_core/tests/system/test_system_events.py | 62 ++--- cuda_core/tests/system/test_system_system.py | 5 +- cuda_core/tests/system/test_system_uuid.py | 7 +- cuda_core/tests/test_device.py | 14 +- .../cuda_python_test_helpers/arch_check.py | 24 ++ 6 files changed, 147 insertions(+), 215 deletions(-) diff --git a/cuda_core/tests/system/test_system_device.py b/cuda_core/tests/system/test_system_device.py index cf5d0ceb42b..6018e982dfb 100644 --- a/cuda_core/tests/system/test_system_device.py +++ b/cuda_core/tests/system/test_system_device.py @@ -4,10 +4,13 @@ from cuda_python_test_helpers.arch_check import ( + skip_if_nvml_device_apis_unsupported, skip_if_nvml_unsupported, unsupported_before, ) +# Keep the broader device-API gate on individual tests so supported queries +# can still run on platforms with partial NVML support. pytestmark = skip_if_nvml_unsupported import array @@ -35,44 +38,29 @@ def check_gpu_available(): pytest.skip("No GPUs available to run device tests", allow_module_level=True) -def _cuda_visible_system_devices(): - indexed_devices = {device.uuid_without_prefix: device for device in system.Device.get_all_devices()} - for cuda_device in CudaDevice.get_all_devices(): - try: - device = cuda_device.to_system_device() - except nvml.NotFoundError: - # Orin supports index lookup but not UUID lookup. Keep the same - # CUDA-visible devices without blocking unrelated NVML queries. - device = indexed_devices[cuda_device.uuid] - yield device - - def test_device_count(): assert system.Device.get_device_count() == system.get_num_devices() @pytest.mark.agent_authored(model="gpt-6") -def test_to_cuda_device(init_cuda): +def test_to_cuda_device(init_cuda, subtests): cuda_uuids = {device.uuid for device in CudaDevice.get_all_devices()} - for device in system.Device.get_all_devices(): if device.uuid_without_prefix not in cuda_uuids: # A physical MIG device may have no CUDA-visible counterpart. with pytest.raises(RuntimeError): device.to_cuda_device() continue - cuda_device = device.to_cuda_device() - assert isinstance(cuda_device, CudaDevice) - assert cuda_device.uuid == device.uuid_without_prefix + with subtests.test(device_index=device.index, operation="uuid_mapping"): + assert isinstance(cuda_device, CudaDevice) + assert cuda_device.uuid == device.uuid_without_prefix - # CUDA only returns a 2-byte PCI bus ID domain, whereas NVML returns a - # 4-byte domain - try: - pci_info = device.pci_info - except nvml.NotSupportedError: - continue - assert cuda_device.pci_bus_id == pci_info.bus_id[4:] + with subtests.test(device_index=device.index, operation="pci_mapping"): + with unsupported_before(device, None): + pci_info = device.pci_info + # CUDA returns a 2-byte PCI bus ID domain; NVML returns 4 bytes. + assert cuda_device.pci_bus_id == pci_info.bus_id[4:] def test_device_architecture(): @@ -104,14 +92,13 @@ def test_device_bar1_memory(subtests): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") +@skip_if_nvml_device_apis_unsupported @pytest.mark.agent_authored(model="gpt-6") def test_device_cpu_affinity(subtests): for device in system.Device.get_all_devices(): with subtests.test(device_index=device.index): - try: + with unsupported_before(device, typing.DeviceArch.KEPLER): affinity = device.get_cpu_affinity(typing.AffinityScope.NODE) - except nvml.NotSupportedError: - continue assert isinstance(affinity, list) original_affinity = os.sched_getaffinity(0) try: @@ -122,7 +109,7 @@ def test_device_cpu_affinity(subtests): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") -@pytest.mark.agent_authored(model="gpt-6") +@skip_if_nvml_device_apis_unsupported def test_affinity(subtests): for device in system.Device.get_all_devices(): for scope in typing.AffinityScope.__members__.values(): @@ -131,24 +118,18 @@ def test_affinity(subtests): affinity_scope=scope.value, affinity_api="get_cpu_affinity", ): - try: + with unsupported_before(device, typing.DeviceArch.KEPLER): affinity = device.get_cpu_affinity(scope) - except nvml.NotSupportedError: - pass - else: - assert isinstance(affinity, list) + assert isinstance(affinity, list) with subtests.test( device_index=device.index, affinity_scope=scope.value, affinity_api="get_memory_affinity", ): - try: + with unsupported_before(device, typing.DeviceArch.KEPLER): affinity = device.get_memory_affinity(scope) - except nvml.NotSupportedError: - pass - else: - assert isinstance(affinity, list) + assert isinstance(affinity, list) def test_numa_node_id(subtests): @@ -160,13 +141,11 @@ def test_numa_node_id(subtests): assert numa_node_id >= -1 -@pytest.mark.agent_authored(model="gpt-6") -def test_device_cuda_compute_capability(init_cuda): - for device in _cuda_visible_system_devices(): - try: - cuda_compute_capability = device.cuda_compute_capability - except nvml.NotSupportedError: - continue +@skip_if_nvml_device_apis_unsupported +def test_device_cuda_compute_capability(): + for cuda_device in CudaDevice.get_all_devices(): + device = cuda_device.to_system_device() + cuda_compute_capability = device.cuda_compute_capability assert isinstance(cuda_compute_capability, tuple) assert len(cuda_compute_capability) == 2 assert all(isinstance(i, int) for i in cuda_compute_capability) @@ -201,14 +180,12 @@ def test_device_name(): assert len(name) > 0 -@pytest.mark.agent_authored(model="gpt-6") -def test_device_pci_info(init_cuda, subtests): - for device in _cuda_visible_system_devices(): +@skip_if_nvml_device_apis_unsupported +def test_device_pci_info(subtests): + for cuda_device in CudaDevice.get_all_devices(): + device = cuda_device.to_system_device() with subtests.test(device_index=device.index): - try: - pci_info = device.pci_info - except nvml.NotSupportedError: - continue + pci_info = device.pci_info assert isinstance(pci_info, _device.PciInfo) assert isinstance(pci_info.bus_id, str) @@ -313,7 +290,7 @@ def test_unpack_bitmask_single_value(): @pytest.mark.parallel_threads_limit(4) # timeouts are slow @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Events not supported on WSL or Windows") -@pytest.mark.agent_authored(model="gpt-6") +@skip_if_nvml_device_apis_unsupported def test_register_events(): # This is not the world's greatest test. All of the events are pretty # infrequent and hard to simulate. So all we do here is register an event, @@ -323,34 +300,22 @@ def test_register_events(): # Also, some hardware doesn't support any event types. for device in system.Device.get_all_devices(): - try: - supported_events = device.get_supported_event_types() - except nvml.NotSupportedError: - continue + supported_events = device.get_supported_event_types() assert isinstance(supported_events, list) assert all(isinstance(ev, typing.EventType) for ev in supported_events) for device in system.Device.get_all_devices(): - try: - events = device.register_events(["xid_critical_error"]) - except nvml.NotSupportedError: - continue + events = device.register_events(["xid_critical_error"]) with pytest.raises(system.TimeoutError): events.wait(timeout_ms=500) for device in system.Device.get_all_devices(): - try: - events = device.register_events([typing.EventType.XID_CRITICAL_ERROR]) - except nvml.NotSupportedError: - continue + events = device.register_events([typing.EventType.XID_CRITICAL_ERROR]) with pytest.raises(system.TimeoutError): events.wait(timeout_ms=500) for device in system.Device.get_all_devices(): - try: - events = device.register_events([]) - except nvml.NotSupportedError: - continue + events = device.register_events([]) with pytest.raises(system.TimeoutError): events.wait(timeout_ms=500) @@ -365,13 +330,11 @@ def test_device_brand(): assert isinstance(brand, str) -@pytest.mark.agent_authored(model="gpt-6") -def test_device_pci_bus_id(init_cuda): - for device in _cuda_visible_system_devices(): - try: - pci_bus_id = device.pci_info.bus_id - except nvml.NotSupportedError: - continue +@skip_if_nvml_device_apis_unsupported +def test_device_pci_bus_id(): + for cuda_device in CudaDevice.get_all_devices(): + device = cuda_device.to_system_device() + pci_bus_id = device.pci_info.bus_id assert isinstance(pci_bus_id, str) new_device = system.Device(pci_bus_id=pci_bus_id) @@ -436,19 +399,14 @@ def test_c2c_mode_enabled(subtests): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Persistence mode not supported on WSL or Windows") @pytest.mark.thread_unsafe(reason="device persistence mode is global state") -@pytest.mark.agent_authored(model="gpt-6") +@skip_if_nvml_device_apis_unsupported def test_persistence_mode_enabled(subtests): for device in system.Device.get_all_devices(): with subtests.test(device_index=device.index): - try: - is_enabled = device.is_persistence_mode_enabled - except nvml.NotSupportedError: - continue + is_enabled = device.is_persistence_mode_enabled assert isinstance(is_enabled, bool) try: device.is_persistence_mode_enabled = False - except nvml.NotSupportedError: - continue except nvml.NoPermissionError as e: pytest.xfail(f"nvml.NoPermissionError: {e}") try: @@ -562,22 +520,14 @@ def test_addressing_mode(subtests): assert addressing_mode is None or addressing_mode in typing.AddressingMode.__members__.values() -@pytest.mark.agent_authored(model="gpt-6") +@skip_if_nvml_device_apis_unsupported def test_display_mode(): for device in system.Device.get_all_devices(): - try: - is_display_connected = device.is_display_connected - except nvml.NotSupportedError: - pass - else: - assert isinstance(is_display_connected, bool) + is_display_connected = device.is_display_connected + assert isinstance(is_display_connected, bool) - try: - is_display_active = device.is_display_active - except nvml.NotSupportedError: - pass - else: - assert isinstance(is_display_active, bool) + is_display_active = device.is_display_active + assert isinstance(is_display_active, bool) def test_repair_status(subtests): @@ -634,13 +584,10 @@ def test_get_nearest_gpus(): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") -@pytest.mark.agent_authored(model="gpt-6") +@skip_if_nvml_device_apis_unsupported def test_get_minor_number(): for device in system.Device.get_all_devices(): - try: - minor_number = device.minor_number - except nvml.NotSupportedError: - continue + minor_number = device.minor_number assert isinstance(minor_number, int) assert minor_number >= 0 @@ -761,18 +708,16 @@ def test_clock_event_reasons(subtests): @pytest.mark.thread_unsafe(reason="device fan settings are global state") -@pytest.mark.agent_authored(model="gpt-6") -def test_fan(init_cuda, subtests): - for device in _cuda_visible_system_devices(): +@skip_if_nvml_device_apis_unsupported +def test_fan(subtests): + for cuda_device in CudaDevice.get_all_devices(): + device = cuda_device.to_system_device() device_index = device.index num_fans = None # The fan APIs are only supported on discrete devices with fans, # but when they are not available `device.num_fans` returns 0. with subtests.test(device_index=device_index, fan_api="get_num_fans"): - try: - value = device.num_fans - except nvml.NotSupportedError: - continue + value = device.num_fans assert isinstance(value, int) assert value >= 0 num_fans = value @@ -815,24 +760,18 @@ def test_fan(init_cuda, subtests): fan_info.set_default_speed() -@pytest.mark.agent_authored(model="gpt-6") -def test_cooler(init_cuda, subtests): - for device in _cuda_visible_system_devices(): +@skip_if_nvml_device_apis_unsupported +def test_cooler(subtests): + for cuda_device in CudaDevice.get_all_devices(): + device = cuda_device.to_system_device() with subtests.test(device_index=device.index): # The cooler APIs are only supported on discrete devices with fans, # but when they are not available `device.num_fans` returns 0. - try: - num_fans = device.num_fans - except nvml.NotSupportedError: - pass - else: - if num_fans == 0: - pytest.skip("Device has no coolers to test") + if device.num_fans == 0: + pytest.skip("Device has no coolers to test") - try: + with unsupported_before(device, DeviceArch.MAXWELL): cooler_info = device.cooler - except nvml.NotSupportedError: - continue assert isinstance(cooler_info, _device.CoolerInfo) @@ -844,7 +783,7 @@ def test_cooler(init_cuda, subtests): @pytest.mark.filterwarnings("ignore::DeprecationWarning") -@pytest.mark.agent_authored(model="gpt-6") +@skip_if_nvml_device_apis_unsupported def test_temperature(subtests): for device in system.Device.get_all_devices(): device_index = device.index @@ -857,13 +796,9 @@ def test_temperature(subtests): continue with subtests.test(device_index=device_index, temperature_api="get_sensor"): - try: - sensor = temperature.get_sensor() - except nvml.NotSupportedError: - pass - else: - assert isinstance(sensor, int) - assert sensor >= 0 + sensor = temperature.get_sensor() + assert isinstance(sensor, int) + assert sensor >= 0 # By docs, should be supported on KEPLER or newer, but experimentally, # is also unsupported on other hardware. @@ -876,28 +811,21 @@ def test_temperature(subtests): temperature_api="get_threshold", threshold=threshold.value, ): - try: + with unsupported_before(device, None): t = temperature.get_threshold(threshold) - except nvml.NotSupportedError: - continue assert isinstance(t, int) assert t >= 0 with subtests.test(device_index=device_index, temperature_api="margin"): - try: + with unsupported_before(device, None): margin = temperature.margin - except nvml.NotSupportedError: - pass - else: - assert isinstance(margin, int) - assert margin >= 0 + assert isinstance(margin, int) + assert margin >= 0 thermals = None with subtests.test(device_index=device_index, temperature_api="get_thermal_settings"): - try: + with unsupported_before(device, None): value = temperature.get_thermal_settings(typing.ThermalTarget.ALL) - except nvml.NotSupportedError: - continue assert isinstance(value, _device.ThermalSettings) thermals = value if thermals is None: @@ -994,14 +922,13 @@ def test_pstates(subtests): assert isinstance(utilization.dec_threshold, int) -@pytest.mark.agent_authored(model="gpt-6") -def test_compute_running_processes(init_cuda, subtests): - for device in _cuda_visible_system_devices(): +@skip_if_nvml_device_apis_unsupported +def test_compute_running_processes(subtests): + for cuda_device in CudaDevice.get_all_devices(): + device = cuda_device.to_system_device() with subtests.test(device_index=device.index): - try: + with unsupported_before(device, "FERMI"): processes = device.compute_running_processes - except nvml.NotSupportedError: - continue assert isinstance(processes, list) for proc in processes: assert isinstance(proc, _device.ProcessInfo) @@ -1017,20 +944,19 @@ def test_compute_running_processes(init_cuda, subtests): proc.compute_instance_id # noqa: B018 -@pytest.mark.agent_authored(model="gpt-6") +@skip_if_nvml_device_apis_unsupported def test_nvlink(subtests): for device in system.Device.get_all_devices(): device_index = device.index link_count = 0 - with subtests.test(device_index=device_index, nvlink_api="get_nvlink_count"): - try: - value = device.get_nvlink_count() - except nvml.NotSupportedError: - pass - else: - assert isinstance(value, int) - assert value >= 0 - link_count = value + with ( + subtests.test(device_index=device_index, nvlink_api="get_nvlink_count"), + unsupported_before(device, None), + ): + value = device.get_nvlink_count() + assert isinstance(value, int) + assert value >= 0 + link_count = value for link in range(link_count): with subtests.test(device_index=device_index, nvlink_api="get_nvlink", link_index=link): @@ -1052,11 +978,11 @@ def test_nvlink(subtests): assert all(isinstance(i, int) for i in version) nvlink_infos = [] - with subtests.test(device_index=device_index, nvlink_api="get_nvlinks"): - try: - nvlink_infos = list(device.get_nvlinks()) - except nvml.NotSupportedError: - continue + with ( + subtests.test(device_index=device_index, nvlink_api="get_nvlinks"), + unsupported_before(device, None), + ): + nvlink_infos = list(device.get_nvlinks()) for link, nvlink_info in enumerate(nvlink_infos): with subtests.test(device_index=device_index, nvlink_api="get_nvlinks", link_index=link): diff --git a/cuda_core/tests/system/test_system_events.py b/cuda_core/tests/system/test_system_events.py index de9758ad853..034a2764dc9 100644 --- a/cuda_core/tests/system/test_system_events.py +++ b/cuda_core/tests/system/test_system_events.py @@ -3,13 +3,16 @@ # SPDX-License-Identifier: Apache-2.0 -from cuda_python_test_helpers.arch_check import skip_if_nvml_unsupported +from cuda_python_test_helpers.arch_check import skip_if_nvml_device_apis_unsupported, skip_if_nvml_unsupported +# Keep device-API gating on individual tests so the pure event conversion and +# wrapping tests still run on platforms with partial device API support. pytestmark = skip_if_nvml_unsupported import helpers import pytest +from cuda.core import Device as CudaDevice from cuda.core import system from cuda.core.system import typing @@ -50,32 +53,21 @@ def test_pci_bus_id_from_gpu_id(gpu_id, expected): assert _pci_bus_id_from_gpu_id(gpu_id) == expected -@pytest.mark.agent_authored(model="gpt-6") -def test_system_event_device_resolves_pci_bus_id(init_cuda): - devices = list(system.Device.get_all_devices()) - if not devices: - pytest.skip("No NVML devices available") - - for device in devices: - try: - original_pci = device.pci_info - except system.NotSupportedError: - # Orin supports PCI lookup but needs CUDA to supply the PCI bus ID. - cuda_pci_bus_id = device.to_cuda_device().pci_bus_id - domain_string, bus_string, device_function = cuda_pci_bus_id.split(":") - device_string, function_string = device_function.split(".") - domain = int(domain_string, 16) - bus = int(bus_string, 16) - pci_device = int(device_string, 16) - assert int(function_string, 16) == 0 - expected_pci_bus_id = f"{domain:08X}:{bus:02X}:{pci_device:02X}.0" - else: - domain, bus, pci_device = original_pci.domain, original_pci.bus, original_pci.device - expected_pci_bus_id = original_pci.bus_id - - if domain > 0xFFFF: - pytest.skip(f"PCI domain {domain:#x} does not fit in a packed gpu_id") - gpu_id = (domain << 16) | (bus << 8) | pci_device +@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="claude-opus-4.7") +def test_system_event_device_resolves_pci_bus_id(): + # Round-trip: pack pci_info with the inverse of _pci_bus_id_from_gpu_id, + # then resolve Device through SystemEvent.device. + cuda_devices = list(CudaDevice.get_all_devices()) + if not cuda_devices: + pytest.skip("No CUDA devices available") + + for cuda_device in cuda_devices: + device = cuda_device.to_system_device() + pci = device.pci_info + if pci.domain > 0xFFFF: + pytest.skip(f"PCI domain {pci.domain:#x} does not fit in a packed gpu_id") + gpu_id = (pci.domain << 16) | (pci.bus << 8) | (pci.device & 0xFF) event_data = nvml.SystemEventData_v1(1) event_data.event_type = nvml.SystemEventType.GPU_DRIVER_BIND @@ -83,21 +75,12 @@ def test_system_event_device_resolves_pci_bus_id(init_cuda): event = SystemEvent(event_data) resolved_device = event.device - assert resolved_device.uuid == device.uuid + assert resolved_device.pci_bus_id == device.pci_bus_id assert resolved_device.index == device.index - # PCI lookup can work even when the PCI information query is unsupported. - try: - pci = resolved_device.pci_info - except system.NotSupportedError: - continue - assert (pci.domain, pci.bus, pci.device) == (domain, bus, pci_device) - assert pci.bus_id == expected_pci_bus_id - assert resolved_device.pci_bus_id == expected_pci_bus_id - @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="System events not supported on WSL or Windows") -@pytest.mark.agent_authored(model="gpt-6") +@skip_if_nvml_device_apis_unsupported def test_register_events(): # This is not the world's greatest test. All of the events are pretty # infrequent and hard to simulate. So all we do here is register an event, @@ -108,9 +91,6 @@ def test_register_events(): try: events = system.register_events([typing.SystemEventType.UNBIND]) - except system.NotSupportedError: - # The documented outcome when none of the requested events are supported. - return except system.UnknownError: pytest.skip("system events may only be registered once per process") diff --git a/cuda_core/tests/system/test_system_system.py b/cuda_core/tests/system/test_system_system.py index d47fd1cfc35..fef0e49fc1a 100644 --- a/cuda_core/tests/system/test_system_system.py +++ b/cuda_core/tests/system/test_system_system.py @@ -58,7 +58,10 @@ def test_nvml_version(): @skip_if_nvml_unsupported @pytest.mark.agent_authored(model="gpt-6") def test_get_process_name(init_cuda): - process_name = system.get_process_name(os.getpid()) + try: + process_name = system.get_process_name(os.getpid()) + except system.NotFoundError: + pytest.skip("Process not found") assert isinstance(process_name, str) assert "python" in process_name diff --git a/cuda_core/tests/system/test_system_uuid.py b/cuda_core/tests/system/test_system_uuid.py index c31fed0375c..704934285b0 100644 --- a/cuda_core/tests/system/test_system_uuid.py +++ b/cuda_core/tests/system/test_system_uuid.py @@ -5,14 +5,15 @@ from cuda.core import system +if system.CUDA_BINDINGS_NVML_IS_COMPATIBLE: + from cuda.bindings import nvml + @pytest.mark.skipif(not system.CUDA_BINDINGS_NVML_IS_COMPATIBLE, reason="Compatible NVML bindings are required") @pytest.mark.thread_unsafe(reason="Temporarily replaces a process-global NVML function") @pytest.mark.parametrize("prefix", ["GPU-", "MIG-", "DLA-", ""]) @pytest.mark.agent_authored(model="gpt-6") def test_device_uuid_preserves_unprefixed_value(monkeypatch, prefix): - from cuda.bindings import nvml - expected_uuid = "abcdef12-abcd-0123-4567-1234567890ab" raw_uuid = prefix + expected_uuid monkeypatch.setattr(nvml, "device_get_uuid", lambda _handle: raw_uuid) @@ -30,8 +31,6 @@ def test_device_uuid_preserves_unprefixed_value(monkeypatch, prefix): ) @pytest.mark.agent_authored(model="gpt-6") def test_to_system_device_propagates_uuid_lookup_error(init_cuda, monkeypatch, exception_name, status_name): - from cuda.bindings import nvml - def unavailable_lookup(_uuid): raise getattr(nvml, exception_name)(getattr(nvml.Return, status_name)) diff --git a/cuda_core/tests/test_device.py b/cuda_core/tests/test_device.py index 6f48fd11820..3c6ce6b9153 100644 --- a/cuda_core/tests/test_device.py +++ b/cuda_core/tests/test_device.py @@ -35,16 +35,16 @@ def test_to_system_device(init_cuda): if not _system.CUDA_BINDINGS_NVML_IS_COMPATIBLE: with pytest.raises(RuntimeError): device.to_system_device() - return + pytest.skip("NVML support requires cuda.bindings version 12.9.6+ for CUDA 12.x or 13.2.0+ for CUDA 13.x") + + from cuda_python_test_helpers.arch_check import hardware_supports_nvml_device_apis + + if not hardware_supports_nvml_device_apis(): + pytest.skip("NVML device APIs are incomplete or unavailable on this platform") - from cuda.core import system from cuda.core.system import Device as SystemDevice - try: - system_device = device.to_system_device() - except system.NotFoundError: - # Orin enumerates NVML devices but does not support lookup by UUID. - return + system_device = device.to_system_device() assert isinstance(system_device, SystemDevice) assert system_device.uuid_without_prefix == device.uuid diff --git a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py index 0b687c3758d..5474513c9fa 100644 --- a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py +++ b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py @@ -28,6 +28,25 @@ def hardware_supports_nvml(): nvml.shutdown() +@cache +def hardware_supports_nvml_device_apis(): + """Verify that NVML supports the device lookup required by cuda.core.""" + from cuda.bindings import nvml + + nvml.init_v2() + try: + if nvml.device_get_count_v2() == 0: + return False + device = nvml.device_get_handle_by_index_v2(0) + nvml.device_get_handle_by_uuid(nvml.device_get_uuid(device)) + except (nvml.NotFoundError, nvml.NotSupportedError, nvml.UnknownError): + return False + else: + return True + finally: + nvml.shutdown() + + def _should_skip_nvml_tests() -> bool: """Return True if NVML tests should be skipped on this system. @@ -49,6 +68,11 @@ def _should_skip_nvml_tests() -> bool: reason="NVML support requires cuda.bindings version 12.9.6+ for CUDA 12.x or 13.2.0+ for CUDA 13.x, and hardware that supports NVML", ) +skip_if_nvml_device_apis_unsupported = pytest.mark.skipif( + _should_skip_nvml_tests() or not hardware_supports_nvml_device_apis(), + reason="NVML device APIs are incomplete or unavailable on this platform", +) + @contextmanager def unsupported_before(device, expected_device_arch): From 4ed9dd0350d0693a81ec79e9a1af94e9e99dfc04 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Thu, 1 Oct 2026 16:42:58 +0000 Subject: [PATCH 7/8] fix(cuda.core): remove NVLink state preflight Read the NVLink count from its own field API without requiring support for the link-zero state API. Replace the preflight regressions with valid zero-count and typed per-field error assertions, regenerate the public stub, and retain only the verified UUID release note with the requested wording. --- cuda_core/cuda/core/system/_device.pyi | 5 --- cuda_core/cuda/core/system/_device.pyx | 12 ----- cuda_core/docs/source/release/1.2.1-notes.rst | 7 +-- cuda_core/tests/system/test_system_nvlink.py | 44 +++++++++---------- 4 files changed, 21 insertions(+), 47 deletions(-) diff --git a/cuda_core/cuda/core/system/_device.pyi b/cuda_core/cuda/core/system/_device.pyi index 9e11781fbaf..aed15a32d17 100644 --- a/cuda_core/cuda/core/system/_device.pyi +++ b/cuda_core/cuda/core/system/_device.pyi @@ -1576,11 +1576,6 @@ class Device: For devices with NVLink support. .. version-added:: 1.1.0 - - Raises - ------ - :class:`cuda.core.system.NotSupportedError` - If the device does not support NVLink queries. """ def get_nvlinks(self) -> Iterable[NvlinkInfo]: """ diff --git a/cuda_core/cuda/core/system/_device.pyx b/cuda_core/cuda/core/system/_device.pyx index 505dd695e81..295bca675d1 100644 --- a/cuda_core/cuda/core/system/_device.pyx +++ b/cuda_core/cuda/core/system/_device.pyx @@ -918,19 +918,7 @@ cdef class Device: For devices with NVLink support. .. version-added:: 1.1.0 - - Raises - ------ - :class:`cuda.core.system.NotSupportedError` - If the device does not support NVLink queries. """ - # Orin's field query may succeed without populating its output. Check - # NVLink support through the native query before reading that output. - try: - nvml.device_get_nvlink_state(self._handle, 0) - except nvml.InvalidArgumentError: - # A device with no link 0 can still report a valid count of zero. - pass return self.get_field_values([FieldId.DEV_NVLINK_LINK_COUNT])[0].value def get_nvlinks(self) -> Iterable[NvlinkInfo]: diff --git a/cuda_core/docs/source/release/1.2.1-notes.rst b/cuda_core/docs/source/release/1.2.1-notes.rst index dedb5f9844a..5920b9489e4 100644 --- a/cuda_core/docs/source/release/1.2.1-notes.rst +++ b/cuda_core/docs/source/release/1.2.1-notes.rst @@ -10,15 +10,10 @@ Fixes and enhancements ---------------------- - :attr:`system.Device.uuid_without_prefix` preserves UUIDs that NVML already - reports without a prefix, including on Orin. This also allows + reports without a prefix, such as on Orin. This also allows :meth:`system.Device.to_cuda_device` to match those devices correctly. (`#2947 `__) -- :meth:`system.Device.get_nvlink_count` and :meth:`system.Device.get_nvlinks` - now raise :class:`system.NotSupportedError` when Orin does not support - NVLink queries, instead of reading an unpopulated NVML field result. - (`#2947 `__) - - :meth:`Buffer.from_handle` no longer requires a current CUDA context when the memory resource reports ``is_device_accessible`` as ``False``. Host-only memory records no deallocation stream, so creating and closing such buffers diff --git a/cuda_core/tests/system/test_system_nvlink.py b/cuda_core/tests/system/test_system_nvlink.py index 50e44d3451f..669c53f3262 100644 --- a/cuda_core/tests/system/test_system_nvlink.py +++ b/cuda_core/tests/system/test_system_nvlink.py @@ -5,11 +5,14 @@ from cuda.core import system +if system.CUDA_BINDINGS_NVML_IS_COMPATIBLE: + from cuda.bindings import nvml + -def _nvlink_count_field(nvml, count): +def _nvlink_count_field(count, status): field = nvml.FieldValue() field.field_id = int(nvml.FieldId.DEV_NVLINK_LINK_COUNT) - field.nvml_return = int(nvml.Return.SUCCESS) + field.nvml_return = int(status) field.value_type = int(nvml.ValueType.UNSIGNED_INT) field.value.ui_val[0] = count return field @@ -17,41 +20,34 @@ def _nvlink_count_field(nvml, count): @pytest.mark.skipif(not system.CUDA_BINDINGS_NVML_IS_COMPATIBLE, reason="Compatible NVML bindings are required") @pytest.mark.thread_unsafe(reason="Temporarily replaces process-global NVML functions") -@pytest.mark.parametrize("method", ["get_nvlink_count", "get_nvlinks"]) @pytest.mark.agent_authored(model="gpt-6") -def test_nvlink_queries_propagate_not_supported(monkeypatch, method): - from cuda.bindings import nvml - +def test_nvlink_zero_count_does_not_require_link_state(monkeypatch): def unsupported_state(_handle, _link): raise nvml.NotSupportedError(nvml.Return.ERROR_NOT_SUPPORTED) - # A successful field result alone does not establish NVLink support. - field = _nvlink_count_field(nvml, 0) + field = _nvlink_count_field(0, nvml.Return.SUCCESS) monkeypatch.setattr(nvml, "device_get_nvlink_state", unsupported_state) monkeypatch.setattr(nvml, "device_get_field_values", lambda _handle, _fields: field) device = system.Device.__new__(system.Device) - with pytest.raises(system.NotSupportedError): - result = getattr(device, method)() - if method == "get_nvlinks": - list(result) + assert device.get_nvlink_count() == 0 + assert list(device.get_nvlinks()) == [] @pytest.mark.skipif(not system.CUDA_BINDINGS_NVML_IS_COMPATIBLE, reason="Compatible NVML bindings are required") @pytest.mark.thread_unsafe(reason="Temporarily replaces process-global NVML functions") -@pytest.mark.parametrize(("state", "count"), [(False, 3), (True, 3), (None, 0)]) +@pytest.mark.parametrize("method", ["get_nvlink_count", "get_nvlinks"]) +@pytest.mark.parametrize( + ("status_name", "exception_name"), + [("ERROR_NOT_SUPPORTED", "NotSupportedError"), ("ERROR_UNKNOWN", "UnknownError")], +) @pytest.mark.agent_authored(model="gpt-6") -def test_nvlink_count_handles_disabled_or_absent_link(monkeypatch, state, count): - from cuda.bindings import nvml - - def link_state(_handle, _link): - if state is None: - raise nvml.InvalidArgumentError(nvml.Return.ERROR_INVALID_ARGUMENT) - return state - - field = _nvlink_count_field(nvml, count) - monkeypatch.setattr(nvml, "device_get_nvlink_state", link_state) +def test_nvlink_queries_propagate_field_error(monkeypatch, method, status_name, exception_name): + field = _nvlink_count_field(0, getattr(nvml.Return, status_name)) monkeypatch.setattr(nvml, "device_get_field_values", lambda _handle, _fields: field) device = system.Device.__new__(system.Device) - assert device.get_nvlink_count() == count + with pytest.raises(getattr(system, exception_name)): + result = getattr(device, method)() + if method == "get_nvlinks": + list(result) From 497203c78d15b025e880ddc7e215eef5ef6a8587 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Thu, 1 Oct 2026 16:43:04 +0000 Subject: [PATCH 8/8] test(cuda.core): share VMM leak test allocation size Use one module constant for the cached warm-up and uncached measured allocations. Preserve eight measured iterations and the strict one-allocation leak threshold. --- cuda_core/tests/test_memory.py | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/cuda_core/tests/test_memory.py b/cuda_core/tests/test_memory.py index 9e6d333a75e..f11c3e802bd 100644 --- a/cuda_core/tests/test_memory.py +++ b/cuda_core/tests/test_memory.py @@ -1507,6 +1507,9 @@ def __init__(self, size): assert ("release", NEW_HANDLE) in calls +VMM_LEAK_TEST_REQUESTED_SIZE = 8 * 1024 * 1024 + + def _vmm_allocate_and_close(mr, requested_size, grow): buf = mr.allocate(requested_size) if grow: @@ -1523,7 +1526,7 @@ def _warm_up_vmm_allocate_and_close(device_id, grow): device, config=VirtualMemoryResourceOptions(handle_type="win32_kmt" if IS_WINDOWS else "posix_fd"), ) - return _vmm_allocate_and_close(mr, 8 * 1024 * 1024, grow) + return _vmm_allocate_and_close(mr, VMM_LEAK_TEST_REQUESTED_SIZE, grow) @pytest.mark.thread_unsafe(reason="cuMemGetInfo measures process-wide free memory") @@ -1538,11 +1541,9 @@ def test_vmm_allocate_close_does_not_leak(init_cuda, grow): device, config=VirtualMemoryResourceOptions(handle_type="win32_kmt" if IS_WINDOWS else "posix_fd"), ) - requested_size = 8 * 1024 * 1024 - baseline = handle_return(driver.cuMemGetInfo())[0] for _ in range(8): - _vmm_allocate_and_close(mr, requested_size, grow) + _vmm_allocate_and_close(mr, VMM_LEAK_TEST_REQUESTED_SIZE, grow) free = handle_return(driver.cuMemGetInfo())[0] # A leak would cost at least one allocation per iteration.