diff --git a/.github/workflows/backport.yml b/.github/workflows/backport.yml index 1331315924b..cb64af94faf 100644 --- a/.github/workflows/backport.yml +++ b/.github/workflows/backport.yml @@ -1,4 +1,4 @@ -# SPDX-FileCopyrightText: Copyright (c) 2024-2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-FileCopyrightText: Copyright (c) 2024-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # # SPDX-License-Identifier: Apache-2.0 diff --git a/cuda_bindings/tests/nvml/test_pci.py b/cuda_bindings/tests/nvml/test_pci.py index 877f9d2998a..429742444de 100644 --- a/cuda_bindings/tests/nvml/test_pci.py +++ b/cuda_bindings/tests/nvml/test_pci.py @@ -11,11 +11,14 @@ def test_discover_gpus(all_devices, subtests): for device in all_devices: - with subtests.test(device_index=nvml.device_get_index(device)): - pci_info = nvml.device_get_pci_info_v3(device) + with ( + subtests.test(device_index=nvml.device_get_index(device)), + unsupported_before(device, None), + contextlib.suppress(nvml.OperatingSystemError), + ): # Docs say this should be supported on PASCAL and later - with unsupported_before(device, None), contextlib.suppress(nvml.OperatingSystemError): - nvml.device_discover_gpus(pci_info.ptr) + pci_info = nvml.device_get_pci_info_v3(device) + nvml.device_discover_gpus(pci_info.ptr) def test_bridge_chip_hierarchy_t(): diff --git a/cuda_core/cuda/core/system/_device.pyi b/cuda_core/cuda/core/system/_device.pyi index 5273fffa446..f8c87a8b486 100644 --- a/cuda_core/cuda/core/system/_device.pyi +++ b/cuda_core/cuda/core/system/_device.pyi @@ -1130,9 +1130,10 @@ class Device: device, as a 5 part hexadecimal string, that augments the immutable, board serial identifier. - In the upstream NVML C++ API, the UUID includes a ``gpu-`` or ``mig-`` - prefix. If you need a `uuid` without that prefix (for example, to - interact with CUDA), use the `uuid_without_prefix` property. + Returns the UUID exactly as reported by NVML. It usually includes a + ``GPU-``, ``MIG-``, or ``DLA-`` prefix, but some platforms report an + unprefixed UUID. To interact with CUDA, use the `uuid_without_prefix` + property. """ @property def uuid_without_prefix(self) -> str: @@ -1141,9 +1142,10 @@ class Device: device, as a 5 part hexadecimal string, that augments the immutable, board serial identifier. - In the upstream NVML C++ API, the UUID includes a ``gpu-`` or ``mig-`` - prefix. This property returns it without the prefix, to match the UUIDs - used in CUDA. If you need the prefix, use the `uuid` property. + Removes a ``GPU-``, ``MIG-``, or ``DLA-`` prefix when present, to match + the UUIDs used in CUDA. An already unprefixed UUID is returned + unchanged. For the UUID exactly as reported by NVML, use the `uuid` + property. """ @property def pci_bus_id(self) -> str: diff --git a/cuda_core/cuda/core/system/_device.pyx b/cuda_core/cuda/core/system/_device.pyx index b5b68b1d3e2..bcea5ff79dd 100644 --- a/cuda_core/cuda/core/system/_device.pyx +++ b/cuda_core/cuda/core/system/_device.pyx @@ -239,9 +239,10 @@ cdef class Device: device, as a 5 part hexadecimal string, that augments the immutable, board serial identifier. - In the upstream NVML C++ API, the UUID includes a ``gpu-`` or ``mig-`` - prefix. If you need a `uuid` without that prefix (for example, to - interact with CUDA), use the `uuid_without_prefix` property. + Returns the UUID exactly as reported by NVML. It usually includes a + ``GPU-``, ``MIG-``, or ``DLA-`` prefix, but some platforms report an + unprefixed UUID. To interact with CUDA, use the `uuid_without_prefix` + property. """ return nvml.device_get_uuid(self._handle) @@ -252,12 +253,15 @@ cdef class Device: device, as a 5 part hexadecimal string, that augments the immutable, board serial identifier. - In the upstream NVML C++ API, the UUID includes a ``gpu-`` or ``mig-`` - prefix. This property returns it without the prefix, to match the UUIDs - used in CUDA. If you need the prefix, use the `uuid` property. + Removes a ``GPU-``, ``MIG-``, or ``DLA-`` prefix when present, to match + the UUIDs used in CUDA. An already unprefixed UUID is returned + unchanged. For the UUID exactly as reported by NVML, use the `uuid` + property. """ - # NVML UUIDs have a `gpu-` or `mig-` prefix. We remove that here. - return nvml.device_get_uuid(self._handle)[4:] + uuid = self.uuid + if uuid.startswith(("GPU-", "MIG-", "DLA-")): + return uuid[4:] + return uuid @property def pci_bus_id(self) -> str: diff --git a/cuda_core/docs/source/release/1.3.0-notes.rst b/cuda_core/docs/source/release/1.3.0-notes.rst index 7baead8d15f..81321e71006 100644 --- a/cuda_core/docs/source/release/1.3.0-notes.rst +++ b/cuda_core/docs/source/release/1.3.0-notes.rst @@ -81,6 +81,11 @@ New features Fixes and enhancements ---------------------- +- :attr:`system.Device.uuid_without_prefix` preserves UUIDs that NVML already + reports without a prefix, such as on Orin. This also allows + :meth:`system.Device.to_cuda_device` to match those devices correctly. + (`#2947 `__) + - A :class:`Program` that is collected as part of a reference cycle no longer raises ``AttributeError`` from its destructor, and the source file it wrote for a ``debug`` or ``lineinfo`` build is removed. The cyclic collector diff --git a/cuda_core/tests/system/test_system_device.py b/cuda_core/tests/system/test_system_device.py index 841424e7caa..278719b3a1b 100644 --- a/cuda_core/tests/system/test_system_device.py +++ b/cuda_core/tests/system/test_system_device.py @@ -3,14 +3,21 @@ # SPDX-License-Identifier: Apache-2.0 -from cuda_python_test_helpers.arch_check import skip_if_nvml_unsupported, unsupported_before +from cuda_python_test_helpers.arch_check import ( + skip_if_nvml_device_apis_unsupported, + skip_if_nvml_unsupported, + unsupported_before, +) +# Keep the broader device-API gate on individual tests so supported queries +# can still run on platforms with partial NVML support. pytestmark = skip_if_nvml_unsupported import array import multiprocessing import os import re +from uuid import UUID import helpers import pytest @@ -32,26 +39,25 @@ def test_device_count(): assert system.Device.get_device_count() == system.get_num_devices() -def test_to_cuda_device(): - from cuda.core import Device as CudaDevice - +@pytest.mark.agent_authored(model="gpt-6") +def test_to_cuda_device(init_cuda, subtests): + cuda_uuids = {device.uuid for device in CudaDevice.get_all_devices()} for device in system.Device.get_all_devices(): - try: - cuda_device = device.to_cuda_device() - except RuntimeError: - # Not all physical NVML devices may have a matching CUDA device - # when MIG is involved. + if device.uuid_without_prefix not in cuda_uuids: + # A physical MIG device may have no CUDA-visible counterpart. + with pytest.raises(RuntimeError): + device.to_cuda_device() continue + cuda_device = device.to_cuda_device() + with subtests.test(device_index=device.index, operation="uuid_mapping"): + assert isinstance(cuda_device, CudaDevice) + assert cuda_device.uuid == device.uuid_without_prefix - assert isinstance(cuda_device, CudaDevice) - assert cuda_device.uuid == device.uuid_without_prefix - - # Technically, this test will only work with PCI devices, but are there - # non-PCI devices we need to support? - - # CUDA only returns a 2-byte PCI bus ID domain, whereas NVML returns a - # 4-byte domain - assert cuda_device.pci_bus_id == device.pci_info.bus_id[4:] + with subtests.test(device_index=device.index, operation="pci_mapping"): + with unsupported_before(device, None): + pci_info = device.pci_info + # CUDA returns a 2-byte PCI bus ID domain; NVML returns 4 bytes. + assert cuda_device.pci_bus_id == pci_info.bus_id[4:] def test_device_architecture(): @@ -83,17 +89,24 @@ def test_device_bar1_memory(subtests): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") +@skip_if_nvml_device_apis_unsupported +@pytest.mark.agent_authored(model="gpt-6") def test_device_cpu_affinity(subtests): for device in system.Device.get_all_devices(): with subtests.test(device_index=device.index): with unsupported_before(device, typing.DeviceArch.KEPLER): affinity = device.get_cpu_affinity(typing.AffinityScope.NODE) assert isinstance(affinity, list) - os.sched_setaffinity(0, affinity) - assert os.sched_getaffinity(0) == set(affinity) + original_affinity = os.sched_getaffinity(0) + try: + os.sched_setaffinity(0, affinity) + assert os.sched_getaffinity(0) == set(affinity) + finally: + os.sched_setaffinity(0, original_affinity) @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") +@skip_if_nvml_device_apis_unsupported def test_affinity(subtests): for device in system.Device.get_all_devices(): for scope in typing.AffinityScope.__members__.values(): @@ -125,6 +138,7 @@ def test_numa_node_id(subtests): assert numa_node_id >= -1 +@skip_if_nvml_device_apis_unsupported def test_device_cuda_compute_capability(): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -163,6 +177,7 @@ def test_device_name(): assert len(name) > 0 +@skip_if_nvml_device_apis_unsupported def test_device_pci_info(subtests): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -234,13 +249,16 @@ def test_device_serial(subtests): assert len(serial) > 0 +@pytest.mark.agent_authored(model="gpt-6") def test_device_uuid_without_prefix(): for device in system.Device.get_all_devices(): uuid = device.uuid_without_prefix assert isinstance(uuid, str) - # Expands to GPU-8hex-4hex-4hex-4hex-12hex, where 8hex means 8 consecutive - # hex characters, e.g.: "GPU-abcdef12-abcd-0123-4567-1234567890ab" + assert str(UUID(uuid)) == uuid.lower() + raw_uuid = device.uuid + expected = raw_uuid[4:] if raw_uuid.startswith(("GPU-", "MIG-", "DLA-")) else raw_uuid + assert uuid == expected @pytest.mark.parametrize( @@ -269,6 +287,7 @@ def test_unpack_bitmask_single_value(): @pytest.mark.parallel_threads_limit(4) # timeouts are slow @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Events not supported on WSL or Windows") +@skip_if_nvml_device_apis_unsupported def test_register_events(): # This is not the world's greatest test. All of the events are pretty # infrequent and hard to simulate. So all we do here is register an event, @@ -308,6 +327,7 @@ def test_device_brand(): assert isinstance(brand, str) +@skip_if_nvml_device_apis_unsupported def test_device_pci_bus_id(): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -376,6 +396,7 @@ def test_c2c_mode_enabled(subtests): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Persistence mode not supported on WSL or Windows") @pytest.mark.thread_unsafe(reason="device persistence mode is global state") +@skip_if_nvml_device_apis_unsupported def test_persistence_mode_enabled(subtests): for device in system.Device.get_all_devices(): with subtests.test(device_index=device.index): @@ -496,6 +517,7 @@ def test_addressing_mode(subtests): assert addressing_mode is None or addressing_mode in typing.AddressingMode.__members__.values() +@skip_if_nvml_device_apis_unsupported def test_display_mode(): for device in system.Device.get_all_devices(): is_display_connected = device.is_display_connected @@ -559,6 +581,7 @@ def test_get_nearest_gpus(): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="Device attributes not supported on WSL or Windows") +@skip_if_nvml_device_apis_unsupported def test_get_minor_number(): for device in system.Device.get_all_devices(): minor_number = device.minor_number @@ -681,6 +704,8 @@ def test_clock_event_reasons(subtests): assert all(isinstance(reason, typing.ClocksEventReasons) for reason in reasons) +@pytest.mark.thread_unsafe(reason="device fan settings are global state") +@skip_if_nvml_device_apis_unsupported def test_fan(subtests): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -732,6 +757,7 @@ def test_fan(subtests): fan_info.set_default_speed() +@skip_if_nvml_device_apis_unsupported def test_cooler(subtests): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -754,6 +780,7 @@ def test_cooler(subtests): @pytest.mark.filterwarnings("ignore::DeprecationWarning") +@skip_if_nvml_device_apis_unsupported def test_temperature(subtests): for device in system.Device.get_all_devices(): device_index = device.index @@ -892,6 +919,7 @@ def test_pstates(subtests): assert isinstance(utilization.dec_threshold, int) +@skip_if_nvml_device_apis_unsupported def test_compute_running_processes(subtests): for cuda_device in CudaDevice.get_all_devices(): device = cuda_device.to_system_device() @@ -913,6 +941,7 @@ def test_compute_running_processes(subtests): proc.compute_instance_id # noqa: B018 +@skip_if_nvml_device_apis_unsupported def test_nvlink(subtests): for device in system.Device.get_all_devices(): device_index = device.index @@ -1009,9 +1038,11 @@ def test_mig(subtests): assert isinstance(mig_device, system.Device) +@pytest.mark.agent_authored(model="gpt-6") def test_uuid(): for device in system.Device.get_all_devices(): uuid = device.uuid assert isinstance(uuid, str) - assert uuid.startswith(("GPU-", "MIG-", "DLA-")) + normalized = uuid[4:] if uuid.startswith(("GPU-", "MIG-", "DLA-")) else uuid + assert str(UUID(normalized)) == normalized.lower() assert uuid == device.uuid diff --git a/cuda_core/tests/system/test_system_events.py b/cuda_core/tests/system/test_system_events.py index 9f91e03675b..92443365923 100644 --- a/cuda_core/tests/system/test_system_events.py +++ b/cuda_core/tests/system/test_system_events.py @@ -3,8 +3,10 @@ # SPDX-License-Identifier: Apache-2.0 -from cuda_python_test_helpers.arch_check import skip_if_nvml_unsupported +from cuda_python_test_helpers.arch_check import skip_if_nvml_device_apis_unsupported, skip_if_nvml_unsupported +# Keep device-API gating on individual tests so the pure event conversion and +# wrapping tests still run on platforms with partial device API support. pytestmark = skip_if_nvml_unsupported import helpers @@ -49,6 +51,7 @@ def test_pci_bus_id_from_gpu_id(gpu_id, expected): assert _pci_bus_id_from_gpu_id(gpu_id) == expected +@skip_if_nvml_device_apis_unsupported @pytest.mark.agent_authored(model="claude-opus-4.7") def test_system_event_device_resolves_pci_bus_id(): # Round-trip: pack pci_info with the inverse of _pci_bus_id_from_gpu_id, @@ -75,6 +78,7 @@ def test_system_event_device_resolves_pci_bus_id(): @pytest.mark.skipif(helpers.IS_WSL or helpers.IS_WINDOWS, reason="System events not supported on WSL or Windows") +@skip_if_nvml_device_apis_unsupported def test_register_events(): # This is not the world's greatest test. All of the events are pretty # infrequent and hard to simulate. So all we do here is register an event, diff --git a/cuda_core/tests/system/test_system_nvlink.py b/cuda_core/tests/system/test_system_nvlink.py new file mode 100644 index 00000000000..7b7feeed47f --- /dev/null +++ b/cuda_core/tests/system/test_system_nvlink.py @@ -0,0 +1,49 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import pytest + +from cuda.bindings import nvml +from cuda.core import system + + +def _nvlink_count_field(count, status): + field = nvml.FieldValue() + field.field_id = int(nvml.FieldId.DEV_NVLINK_LINK_COUNT) + field.nvml_return = int(status) + field.value_type = int(nvml.ValueType.UNSIGNED_INT) + field.value.ui_val[0] = count + return field + + +@pytest.mark.thread_unsafe(reason="Temporarily replaces process-global NVML functions") +@pytest.mark.agent_authored(model="gpt-6") +def test_nvlink_zero_count_does_not_require_link_state(monkeypatch): + def unsupported_state(_handle, _link): + raise nvml.NotSupportedError(nvml.Return.ERROR_NOT_SUPPORTED) + + field = _nvlink_count_field(0, nvml.Return.SUCCESS) + monkeypatch.setattr(nvml, "device_get_nvlink_state", unsupported_state) + monkeypatch.setattr(nvml, "device_get_field_values", lambda _handle, _fields: field) + device = system.Device.__new__(system.Device) + + assert device.get_nvlink_count() == 0 + assert list(device.get_nvlinks()) == [] + + +@pytest.mark.thread_unsafe(reason="Temporarily replaces process-global NVML functions") +@pytest.mark.parametrize("method", ["get_nvlink_count", "get_nvlinks"]) +@pytest.mark.parametrize( + ("status_name", "exception_name"), + [("ERROR_NOT_SUPPORTED", "NotSupportedError"), ("ERROR_UNKNOWN", "UnknownError")], +) +@pytest.mark.agent_authored(model="gpt-6") +def test_nvlink_queries_propagate_field_error(monkeypatch, method, status_name, exception_name): + field = _nvlink_count_field(0, getattr(nvml.Return, status_name)) + monkeypatch.setattr(nvml, "device_get_field_values", lambda _handle, _fields: field) + device = system.Device.__new__(system.Device) + + with pytest.raises(getattr(system, exception_name)): + result = getattr(device, method)() + if method == "get_nvlinks": + list(result) diff --git a/cuda_core/tests/system/test_system_system.py b/cuda_core/tests/system/test_system_system.py index fca171a28b0..45eec34c0e1 100644 --- a/cuda_core/tests/system/test_system_system.py +++ b/cuda_core/tests/system/test_system_system.py @@ -9,7 +9,6 @@ from cuda_python_test_helpers.arch_check import skip_if_nvml_unsupported from cuda.bindings import driver -from cuda.core import Device as CudaDevice from cuda.core import system from cuda.core._utils.cuda_utils import handle_return @@ -50,16 +49,12 @@ def test_nvml_version(): @skip_if_nvml_unsupported -def test_get_process_name(): - for cuda_device in CudaDevice.get_all_devices(): - device = cuda_device.to_system_device() - _ = device.compute_running_processes - +@pytest.mark.agent_authored(model="gpt-6") +def test_get_process_name(init_cuda): try: process_name = system.get_process_name(os.getpid()) except system.NotFoundError: pytest.skip("Process not found") - assert isinstance(process_name, str) assert "python" in process_name diff --git a/cuda_core/tests/system/test_system_uuid.py b/cuda_core/tests/system/test_system_uuid.py new file mode 100644 index 00000000000..733ad13f1c0 --- /dev/null +++ b/cuda_core/tests/system/test_system_uuid.py @@ -0,0 +1,36 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import pytest + +from cuda.bindings import nvml +from cuda.core import system + + +@pytest.mark.thread_unsafe(reason="Temporarily replaces a process-global NVML function") +@pytest.mark.parametrize("prefix", ["GPU-", "MIG-", "DLA-", ""]) +@pytest.mark.agent_authored(model="gpt-6") +def test_device_uuid_preserves_unprefixed_value(monkeypatch, prefix): + expected_uuid = "abcdef12-abcd-0123-4567-1234567890ab" + raw_uuid = prefix + expected_uuid + monkeypatch.setattr(nvml, "device_get_uuid", lambda _handle: raw_uuid) + device = system.Device.__new__(system.Device) + + assert device.uuid == raw_uuid + assert device.uuid_without_prefix == expected_uuid + + +@pytest.mark.thread_unsafe(reason="Temporarily replaces a process-global NVML function") +@pytest.mark.parametrize( + ("exception_name", "status_name"), + [("NotFoundError", "ERROR_NOT_FOUND"), ("NotSupportedError", "ERROR_NOT_SUPPORTED")], +) +@pytest.mark.agent_authored(model="gpt-6") +def test_to_system_device_propagates_uuid_lookup_error(init_cuda, monkeypatch, exception_name, status_name): + def unavailable_lookup(_uuid): + raise getattr(nvml, exception_name)(getattr(nvml.Return, status_name)) + + monkeypatch.setattr(nvml, "device_get_handle_by_uuid", unavailable_lookup) + + with pytest.raises(getattr(system, exception_name)): + init_cuda.to_system_device() diff --git a/cuda_core/tests/test_device.py b/cuda_core/tests/test_device.py index de72888f757..d12b86e0190 100644 --- a/cuda_core/tests/test_device.py +++ b/cuda_core/tests/test_device.py @@ -26,14 +26,14 @@ def test_device_init_disabled(): cuda.core._device.DeviceProperties() # Ensure back door is locked. -def test_to_system_device(deinit_cuda): +@pytest.mark.agent_authored(model="gpt-6") +def test_to_system_device(init_cuda): + device = init_cuda - device = Device() - - from cuda_python_test_helpers.arch_check import hardware_supports_nvml + from cuda_python_test_helpers.arch_check import hardware_supports_nvml_device_apis - if not hardware_supports_nvml(): - pytest.skip("NVML not supported on this platform") + if not hardware_supports_nvml_device_apis(): + pytest.skip("NVML device APIs are incomplete or unavailable on this platform") from cuda.core.system import Device as SystemDevice diff --git a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py index 1db4d210693..19a231c807b 100644 --- a/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py +++ b/cuda_python_test_helpers/cuda_python_test_helpers/arch_check.py @@ -16,7 +16,6 @@ def hardware_supports_nvml(): Returns False on platforms where NVML is unsupported (e.g. Jetson Orin). """ from cuda.bindings import nvml - from cuda.bindings._internal.utils import FunctionNotFoundError as NvmlSymbolNotFoundError # noqa: F401 nvml.init_v2() try: @@ -29,6 +28,25 @@ def hardware_supports_nvml(): nvml.shutdown() +@cache +def hardware_supports_nvml_device_apis(): + """Verify that NVML supports the device lookup required by cuda.core.""" + from cuda.bindings import nvml + + nvml.init_v2() + try: + if nvml.device_get_count_v2() == 0: + return False + device = nvml.device_get_handle_by_index_v2(0) + nvml.device_get_handle_by_uuid(nvml.device_get_uuid(device)) + except (nvml.NotFoundError, nvml.NotSupportedError, nvml.UnknownError): + return False + else: + return True + finally: + nvml.shutdown() + + def _should_skip_nvml_tests() -> bool: """Return True if the NVML tests should skip on this system. @@ -42,6 +60,11 @@ def _should_skip_nvml_tests() -> bool: reason="this hardware does not support NVML", ) +skip_if_nvml_device_apis_unsupported = pytest.mark.skipif( + _should_skip_nvml_tests() or not hardware_supports_nvml_device_apis(), + reason="NVML device APIs are incomplete or unavailable on this platform", +) + @contextmanager def unsupported_before(device, expected_device_arch):