From 540d408a98ead5499f0bbf8ff9763b492cba9aa6 Mon Sep 17 00:00:00 2001 From: Ralf Juengling Date: Thu, 6 Aug 2026 11:19:32 -0700 Subject: [PATCH 1/3] test(cuda_core): reserve the driver's two pool address windows at session start The driver's default memory pool and graph memory pool each permanently reserve roughly twice the installed device memory of virtual address space, and neither can be capped or released. On a large-memory GPU with a bounded per-process address space, whether the second reservation still finds a contiguous range depends on how fragmented the space has become by the time some test needs it, which is what makes the failures in #2381 intermittent. Measurements and documentation references: https://github.com/NVIDIA/cuda-python/issues/2381#issuecomment-5207900890 Take both reservations up front in a session-scoped autouse fixture so they land back to back in a nearly empty address space. This does not reduce the footprint; it makes the outcome deterministic. If either is refused the session aborts with an explanation, since every later test needing that pool would fail the same way and a cascade of identical OOM errors says nothing about the cause. Available address space is reported before and after via cuMemAddressReserve rather than an OS query, so the measurement behaves the same on Windows and Linux. CUDA_CORE_TEST_SKIP_EARLY_RESERVATION=1 disables the fixture; CUDA_CORE_TEST_FORCE_RESERVATION_FAILURE=1 exercises the abort path. --- cuda_core/tests/conftest.py | 48 ++++ cuda_core/tests/helpers/va_reservation.py | 319 ++++++++++++++++++++++ cuda_core/tests/test_helpers.py | 104 +++++++ 3 files changed, 471 insertions(+) create mode 100644 cuda_core/tests/helpers/va_reservation.py diff --git a/cuda_core/tests/conftest.py b/cuda_core/tests/conftest.py index dfe97b265eb..cb405481e76 100644 --- a/cuda_core/tests/conftest.py +++ b/cuda_core/tests/conftest.py @@ -31,6 +31,7 @@ from cuda_python_test_helpers.marks import skipif_need_cuda_headers # noqa: F401 (re-exported for tests) from cuda_python_test_helpers.mempool import xfail_if_mempool_oom +from helpers import va_reservation from helpers.constants import POOL_SIZE import cuda.core @@ -55,6 +56,53 @@ def pytest_configure(config): config.pluginmanager.register(_CudaCoreParallelPlugin(), name="_cuda_core_parallel_plugin") +_reservation_report = None + + +@pytest.fixture(scope="session", autouse=True) +def reserve_driver_pools(request, session_setup): + """Take the driver's two large address-space reservations before anything else. + + Session-scoped and autouse so both reservations land back to back in a + nearly empty address space, ahead of any test that could fragment it. + Depends on session_setup for cuInit. See helpers/va_reservation.py for why + this matters, and issue #2381. + + Aborts the session with an explanation if either reservation is refused: + every later test that needs the pool would fail the same way, and the + resulting cascade of identical OOM errors says nothing about the cause. + """ + global _reservation_report + + if int(os.environ.get("CUDA_CORE_TEST_SKIP_EARLY_RESERVATION", 0)) != 0: + yield + return + + with _init_cuda_context() as device: + _reservation_report = va_reservation.reserve_driver_pools(device) + + terminal_reporter = request.config.pluginmanager.get_plugin("terminalreporter") + if terminal_reporter is not None: + # Written here rather than only in the summary so the numbers survive a + # session that dies partway through. + terminal_reporter.write_sep("=", "cuda_core address space reservation") + for line in _reservation_report.lines(): + terminal_reporter.write_line(line) + + if _reservation_report.failed: + pytest.exit(va_reservation.build_failure_message(_reservation_report), returncode=1) + + yield + + +def pytest_terminal_summary(terminalreporter): + if _reservation_report is None or _reservation_report.failed: + return + terminalreporter.write_sep("=", "cuda_core address space reservation") + for line in _reservation_report.lines(): + terminalreporter.write_line(line) + + @contextmanager def _init_cuda_context(): # TODO: rename this to e.g. init_context diff --git a/cuda_core/tests/helpers/va_reservation.py b/cuda_core/tests/helpers/va_reservation.py new file mode 100644 index 00000000000..eed57251b8b --- /dev/null +++ b/cuda_core/tests/helpers/va_reservation.py @@ -0,0 +1,319 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Force the driver's two large address-space reservations up front (issue #2381). + +The driver keeps two pools per device that it never gives back, and each one +reserves a virtual address window of roughly twice the installed device memory +when it is first touched: + +- the **default device mempool**, reserved by ``cuDeviceGetMemPool`` +- the **graph memory pool**, reserved by ``cuGraphAddMemAllocNode`` at node + creation time + +Neither is capped by ``max_size`` and neither is released by +``cuDeviceGraphMemTrim``, ``cuGraphDestroy``, or anything else short of process +exit. On a large-memory GPU with a bounded per-process address space -- 357 GiB +per pool on a 179 GiB device -- the two together consume most of the budget, and +whether the *second* one finds a contiguous range depends on how fragmented the +space has become. That is what makes the full-suite failures intermittent, and +why they always begin at the first test to need whichever pool came second. + +Taking both reservations at session start, back to back into a nearly empty +address space, removes test order and accumulated fragmentation from the +question. It does not reduce the footprint; it makes the outcome deterministic. + +Measurement here goes through ``cuMemAddressReserve``, not the OS, so it behaves +the same on Windows and Linux. +""" + +from __future__ import annotations + +import os +import time + +from cuda.bindings import driver + +MIB = 1024 * 1024 +GIB = 1024 * MIB + +# cuMemAddressReserve wants a power-of-two alignment and a size that is a +# multiple of it. 2 MiB is the granularity the driver uses for pools. +VA_ALIGNMENT = 2 * MIB +# Above any plausible per-process budget, so the descending probe below always +# starts from a size that fails. +MAX_PROBE_BYTES = 1 << 46 + +# Each driver-managed pool reserves about this multiple of installed device +# memory. Used to express remaining headroom in units of "one more pool". +POOL_RESERVATION_MULTIPLE = 2 + + +def align_up(size: int) -> int: + """Round to a multiple of the reservation alignment. + + Device memory sizes are not generally a multiple of it, and + cuMemAddressReserve rejects a size that is not with + CUDA_ERROR_INVALID_VALUE -- which would otherwise read as "no address space + left" at every size probed. + """ + return ((size + VA_ALIGNMENT - 1) // VA_ALIGNMENT) * VA_ALIGNMENT + + +def _reserve_and_release(size: int) -> bool: + """True if the driver still grants a contiguous reservation of ``size``. + + Reserving costs address space but no memory, so this reads what the address + space can satisfy without perturbing it. Raises if the release fails, since + a leaked reservation would corrupt every later measurement. + """ + err, ptr = driver.cuMemAddressReserve(align_up(size), VA_ALIGNMENT, 0, 0) + if err != driver.CUresult.CUDA_SUCCESS: + return False + (err,) = driver.cuMemAddressFree(ptr, align_up(size)) + if err != driver.CUresult.CUDA_SUCCESS: + raise RuntimeError(f"cuMemAddressFree({size:#x}) -> {err!r}; address space measurement is unreliable") + return True + + +def largest_reservable(reserve=None, max_bytes: int = MAX_PROBE_BYTES, refine_steps: int = 4) -> int: + """Largest contiguous reservation the driver still grants. + + Halves down from ``max_bytes`` rather than doubling up from the granularity, + because a *refused* reservation allocates nothing and returns immediately + while releasing a granted one is expensive -- hundreds of milliseconds for a + large range, and seconds on some configurations. Descending pays that cost + once instead of once per rung. + + Halving alone only resolves to a power of two, which is too coarse to show + what the reservations cost when the budget is much larger than they are, so + ``refine_steps`` bisections then narrow the answer. Each bisection risks one + more expensive release, hence the small default. + + ``reserve`` is injectable so the search can be tested without a GPU. + """ + reserve = _reserve_and_release if reserve is None else reserve + + size = max_bytes + while size >= VA_ALIGNMENT and not reserve(size): + size //= 2 + if size < VA_ALIGNMENT: + return 0 + if size == max_bytes: + return size # nothing was refused, so there is no bracket to narrow + + low, high = size, size * 2 # high was refused on the way down + for _ in range(refine_steps): + middle = ((low + high) // 2 // VA_ALIGNMENT) * VA_ALIGNMENT + if middle <= low or middle >= high: + break + if reserve(middle): + low = middle + else: + high = middle + return low + + +def vmm_supported(device_id: int = 0) -> bool: + """True if this device exposes the virtual memory management APIs.""" + err, dev = driver.cuDeviceGet(device_id) + if err != driver.CUresult.CUDA_SUCCESS: + return False + attribute = driver.CUdevice_attribute.CU_DEVICE_ATTRIBUTE_VIRTUAL_MEMORY_MANAGEMENT_SUPPORTED + err, supported = driver.cuDeviceGetAttribute(attribute, dev) + return err == driver.CUresult.CUDA_SUCCESS and bool(supported) + + +def format_bytes(value: int | None) -> str: + if value is None: + return "unknown" + return f"{value / GIB:.2f} GiB" + + +class Reservation: + """One driver-managed pool that has to be materialized.""" + + def __init__(self, name: str, detail: str, materialize): + self.name = name + self.detail = detail + self._materialize = materialize + self.error: str | None = None + + def run(self) -> bool: + """Materialize the pool. Returns True on success, recording any error.""" + try: + self._materialize() + except Exception as exc: # surfaced to the user by build_failure_message + self.error = f"{type(exc).__name__}: {exc}" + return False + return True + + @property + def succeeded(self) -> bool: + return self.error is None + + +def _materialize_default_mempool(device): + """Touch the device's default memory pool. + + cuDeviceGetMemPool is the call that makes the reservation; the allocation + only proves the pool is usable afterwards. + """ + buffer = device.memory_resource.allocate(8, stream=device.default_stream) + device.default_stream.sync() + buffer.close(device.default_stream) + device.default_stream.sync() + + +def _materialize_graph_mempool(device): + """Add one graph memory-allocation node, which reserves the graph pool. + + The graph is discarded immediately: the address reservation it triggers + outlives it, which is the whole point. + """ + from cuda.core.graph import GraphDefinition + + definition = GraphDefinition() + definition.allocate(1024) + del definition + + +def _forced_failure(): + """Stand-in for a refused reservation, so the abort path can be exercised. + + Set CUDA_CORE_TEST_FORCE_RESERVATION_FAILURE=1 to check what this suite + reports on a machine whose address space is too small, without needing one. + """ + raise RuntimeError("CUDA_ERROR_OUT_OF_MEMORY: simulated refusal (CUDA_CORE_TEST_FORCE_RESERVATION_FAILURE is set)") + + +def reservations_for(device) -> list[Reservation]: + if os.environ.get("CUDA_CORE_TEST_FORCE_RESERVATION_FAILURE", "0") not in ("0", ""): + return [ + Reservation("default device mempool", "cuDeviceGetMemPool", _forced_failure), + Reservation("graph memory pool", "cuGraphAddMemAllocNode", _forced_failure), + ] + return [ + Reservation( + "default device mempool", + "cuDeviceGetMemPool", + lambda: _materialize_default_mempool(device), + ), + Reservation( + "graph memory pool", + "cuGraphAddMemAllocNode", + lambda: _materialize_graph_mempool(device), + ), + ] + + +class ReservationReport: + """What the early reservations cost, for the terminal.""" + + def __init__(self, device_name, device_memory, before, after, reservations, measured, seconds=0.0): + self.device_name = device_name + self.device_memory = device_memory + self.before = before + self.after = after + self.reservations = reservations + self.measured = measured + self.seconds = seconds + + @property + def failed(self) -> list[Reservation]: + return [item for item in self.reservations if not item.succeeded] + + @property + def pool_reservation_bytes(self) -> int | None: + if self.device_memory is None: + return None + return align_up(POOL_RESERVATION_MULTIPLE * self.device_memory) + + def lines(self) -> list[str]: + out = [f"device 0: {self.device_name} ({format_bytes(self.device_memory)} device memory)"] + if self.measured: + out.append(f"largest reservable range before: {format_bytes(self.before)}") + else: + out.append("largest reservable range: not measured (no virtual memory management support)") + + for item in self.reservations: + status = "reserved" if item.succeeded else f"FAILED: {item.error}" + out.append(f" {item.name:<24} {item.detail:<24} {status}") + + if self.measured: + # A drop here is a *lower* bound on what was taken: the driver may + # carve its reservations out of a region other than the largest + # hole, in which case the largest hole does not move at all. + change = None if self.before is None or self.after is None else self.before - self.after + if change is None: + note = "" + elif change > 0: + note = f" (largest hole shrank by {format_bytes(change)})" + else: + note = " (largest hole unchanged)" + out.append(f"largest reservable range after: {format_bytes(self.after)}{note}") + pool_bytes = self.pool_reservation_bytes + if pool_bytes and self.after is not None: + out.append( + f"remaining headroom: {self.after // pool_bytes} more pool-sized " + f"({format_bytes(pool_bytes)}) reservations [{self.seconds:.1f}s measuring]" + ) + return out + + +def build_failure_message(report: ReservationReport) -> str: + """Explain, for a human, why this machine cannot run the suite.""" + pool_bytes = report.pool_reservation_bytes + failed = ", ".join(item.name for item in report.failed) + lines = [ + "", + "cuda_core tests cannot run on this machine: the CUDA driver could not reserve", + f"address space for {failed}.", + "", + f" device 0 {report.device_name}", + f" installed device memory {format_bytes(report.device_memory)}", + f" needed per driver-managed pool {format_bytes(pool_bytes)} of *virtual address space*", + f" largest range still available {format_bytes(report.after if report.measured else None)}", + "", + ] + for item in report.failed: + lines.append(f" {item.name} ({item.detail}): {item.error}") + lines += [ + "", + "The driver keeps two pools per device -- the default memory pool and the graph", + "memory pool -- and reserves roughly twice the installed device memory of address", + "space for each one the first time it is used. Neither reservation can be capped,", + "and neither is released before the process exits. On a large-memory GPU with a", + "bounded per-process address space the two together can exceed the budget, and no", + "amount of freeing device memory helps: the exhausted resource is address space,", + "not memory. Expect cuMemGetInfo to report plenty free while this fails.", + "", + "Options: run on a device with less memory, run the graph tests in a separate", + "process from the rest of the suite so the two reservations never coexist, or on", + "Windows check whether the driver model (WDDM/MCDM/TCC) bounds the address space", + "more tightly than expected. See issue #2381.", + "", + ] + return "\n".join(lines) + + +def reserve_driver_pools(device, measure: bool = True) -> ReservationReport: + """Materialize both driver-managed pools, measuring address space around them.""" + device_memory = None + err, _free, total = driver.cuMemGetInfo() + if err == driver.CUresult.CUDA_SUCCESS: + device_memory = int(total) + + measured = measure and vmm_supported(device.device_id) + started = time.perf_counter() + before = largest_reservable() if measured else None + elapsed = time.perf_counter() - started + + reservations = reservations_for(device) + for item in reservations: + item.run() + + started = time.perf_counter() + after = largest_reservable() if measured else None + elapsed += time.perf_counter() - started + return ReservationReport(device.name, device_memory, before, after, reservations, measured, elapsed) diff --git a/cuda_core/tests/test_helpers.py b/cuda_core/tests/test_helpers.py index 43dbf8887e2..4716d7fc13b 100644 --- a/cuda_core/tests/test_helpers.py +++ b/cuda_core/tests/test_helpers.py @@ -5,6 +5,7 @@ import time import pytest +from helpers import va_reservation from helpers.buffers import PatternGen, compare_equal_buffers, make_scratch_buffer from helpers.latch import LatchKernel from helpers.logging import TimestampedLogger @@ -80,3 +81,106 @@ def test_patterngen_values(): pgen = PatternGen(device, NBYTES) pgen.verify_buffer(ones, value=1) pgen.verify_buffer(twos, value=2) + + +# helpers.va_reservation (issue #2381). The GPU-dependent part runs once per +# session in an autouse fixture, so only the pure logic is covered here. + + +def _fake_reserve(limit, log=None): + """Grants any reservation up to ``limit``, recording what was asked.""" + + def reserve(size): + if log is not None: + log.append(size) + return size <= limit + + return reserve + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_va_probe_finds_the_boundary(): + limit = 700 * va_reservation.GIB + found = va_reservation.largest_reservable(reserve=_fake_reserve(limit)) + + # Refinement should land within a few percent, never above the real limit. + assert found <= limit + assert found > limit * 0.9 + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_va_probe_descends_so_only_one_grant_is_paid_for(): + # Releasing a granted reservation is the expensive half; a refused one costs + # nothing. Ascending from the granularity would pay for every rung. + asked = [] + limit = 700 * va_reservation.GIB + va_reservation.largest_reservable(reserve=_fake_reserve(limit, asked), refine_steps=0) + + assert sum(1 for size in asked if size <= limit) == 1 + assert asked[0] == va_reservation.MAX_PROBE_BYTES + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_va_probe_reports_zero_when_nothing_can_be_reserved(): + assert va_reservation.largest_reservable(reserve=_fake_reserve(0)) == 0 + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_va_probe_only_asks_for_aligned_sizes(): + # Device memory is not a multiple of the 2 MiB granularity, and an unaligned + # ask fails with CUDA_ERROR_INVALID_VALUE at every size -- which would read + # as an exhausted address space rather than as a bug here. + asked = [] + va_reservation.largest_reservable(reserve=_fake_reserve(3 * 25650855936, asked)) + + assert asked and all(size % va_reservation.VA_ALIGNMENT == 0 for size in asked) + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_reservation_records_failure_without_raising(): + def boom(): + raise RuntimeError("CUDA_ERROR_OUT_OF_MEMORY: nope") + + item = va_reservation.Reservation("default device mempool", "cuDeviceGetMemPool", boom) + + assert item.run() is False + assert item.succeeded is False + assert "CUDA_ERROR_OUT_OF_MEMORY" in item.error + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_failure_message_names_the_pool_and_the_size_it_needed(): + failed = va_reservation.Reservation("graph memory pool", "cuGraphAddMemAllocNode", lambda: 1 / 0) + failed.run() + report = va_reservation.ReservationReport( + "NVIDIA Graphics Device", 191998918656, 300 * va_reservation.GIB, 300 * va_reservation.GIB, [failed], True + ) + + message = va_reservation.build_failure_message(report) + + assert "cannot run on this machine" in message + assert "graph memory pool" in message + assert "357.63 GiB" in message # 2x installed device memory + assert "address space" in message + assert "#2381" in message + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_report_lines_cover_both_driver_pools(): + ok_pools = [ + va_reservation.Reservation("default device mempool", "cuDeviceGetMemPool", lambda: None), + va_reservation.Reservation("graph memory pool", "cuGraphAddMemAllocNode", lambda: None), + ] + for item in ok_pools: + item.run() + report = va_reservation.ReservationReport( + "dev", 25650855936, 900 * va_reservation.GIB, 800 * va_reservation.GIB, ok_pools, True + ) + + text = "\n".join(report.lines()) + + assert report.failed == [] + assert "default device mempool" in text + assert "graph memory pool" in text + assert "shrank by 100.00 GiB" in text + assert "more pool-sized" in text From 29cbfb4d1568ad32569ad22656bf1248780d7156 Mon Sep 17 00:00:00 2001 From: Ralf Juengling Date: Thu, 6 Aug 2026 11:23:13 -0700 Subject: [PATCH 2/3] test(cuda_core): skip early pool reservation when mempools are unsupported Both driver-managed pools require mempool support, so on a device without it there is nothing to reserve and nothing to pre-empt. Reporting a failed reservation there would abort the session, when the correct behaviour is to carry on: the tests that need pools already skip themselves via Device.properties.memory_pools_supported, and the rest still run. --- cuda_core/tests/helpers/va_reservation.py | 35 +++++++++++------------ 1 file changed, 17 insertions(+), 18 deletions(-) diff --git a/cuda_core/tests/helpers/va_reservation.py b/cuda_core/tests/helpers/va_reservation.py index eed57251b8b..c5abf48ab42 100644 --- a/cuda_core/tests/helpers/va_reservation.py +++ b/cuda_core/tests/helpers/va_reservation.py @@ -210,7 +210,9 @@ def reservations_for(device) -> list[Reservation]: class ReservationReport: """What the early reservations cost, for the terminal.""" - def __init__(self, device_name, device_memory, before, after, reservations, measured, seconds=0.0): + def __init__( + self, device_name, device_memory, before, after, reservations, measured, seconds=0.0, unsupported=False + ): self.device_name = device_name self.device_memory = device_memory self.before = before @@ -218,6 +220,7 @@ def __init__(self, device_name, device_memory, before, after, reservations, meas self.reservations = reservations self.measured = measured self.seconds = seconds + self.unsupported = unsupported @property def failed(self) -> list[Reservation]: @@ -231,6 +234,9 @@ def pool_reservation_bytes(self) -> int | None: def lines(self) -> list[str]: out = [f"device 0: {self.device_name} ({format_bytes(self.device_memory)} device memory)"] + if self.unsupported: + out.append("device does not support memory pools; nothing to reserve") + return out if self.measured: out.append(f"largest reservable range before: {format_bytes(self.before)}") else: @@ -278,32 +284,25 @@ def build_failure_message(report: ReservationReport) -> str: ] for item in report.failed: lines.append(f" {item.name} ({item.detail}): {item.error}") - lines += [ - "", - "The driver keeps two pools per device -- the default memory pool and the graph", - "memory pool -- and reserves roughly twice the installed device memory of address", - "space for each one the first time it is used. Neither reservation can be capped,", - "and neither is released before the process exits. On a large-memory GPU with a", - "bounded per-process address space the two together can exceed the budget, and no", - "amount of freeing device memory helps: the exhausted resource is address space,", - "not memory. Expect cuMemGetInfo to report plenty free while this fails.", - "", - "Options: run on a device with less memory, run the graph tests in a separate", - "process from the rest of the suite so the two reservations never coexist, or on", - "Windows check whether the driver model (WDDM/MCDM/TCC) bounds the address space", - "more tightly than expected. See issue #2381.", - "", - ] + return "\n".join(lines) def reserve_driver_pools(device, measure: bool = True) -> ReservationReport: - """Materialize both driver-managed pools, measuring address space around them.""" + """Materialize both driver-managed pools, measuring address space around them. + + Both pools require mempool support, so on a device without it there is + nothing to reserve and nothing to pre-empt. Skip rather than fail: the tests + that need pools skip themselves on such a device, and the rest still run. + """ device_memory = None err, _free, total = driver.cuMemGetInfo() if err == driver.CUresult.CUDA_SUCCESS: device_memory = int(total) + if not device.properties.memory_pools_supported: + return ReservationReport(device.name, device_memory, None, None, [], measured=False, unsupported=True) + measured = measure and vmm_supported(device.device_id) started = time.perf_counter() before = largest_reservable() if measured else None From 2574452260b0374197c8dd75c5cd49ba81714e77 Mon Sep 17 00:00:00 2001 From: Ralf Juengling Date: Fri, 7 Aug 2026 15:58:37 -0700 Subject: [PATCH 3/3] test(cuda_core): report the address-space hole structure with the reservations The largest free range alone does not predict whether both driver pools fit: what matters is how many pool-sized reservations the free holes can hold between them, since a reservation must fit inside one hole but a hole twice the size takes two. Report that count on every run, and the hole addresses only when a reservation is refused, which also shows whether the layout is being randomized. Windows only, via VirtualQuery in a module imported only on win32; the section is omitted elsewhere. The report is now printed once, from the terminal summary. --- cuda_core/tests/conftest.py | 12 +-- cuda_core/tests/helpers/va_reservation.py | 86 +++++++++++++++++++- cuda_core/tests/helpers/win_address_space.py | 74 +++++++++++++++++ cuda_core/tests/test_helpers.py | 71 +++++++++++++++- 4 files changed, 231 insertions(+), 12 deletions(-) create mode 100644 cuda_core/tests/helpers/win_address_space.py diff --git a/cuda_core/tests/conftest.py b/cuda_core/tests/conftest.py index cb405481e76..eb1dfd0ff40 100644 --- a/cuda_core/tests/conftest.py +++ b/cuda_core/tests/conftest.py @@ -60,7 +60,7 @@ def pytest_configure(config): @pytest.fixture(scope="session", autouse=True) -def reserve_driver_pools(request, session_setup): +def reserve_driver_pools(session_setup): """Take the driver's two large address-space reservations before anything else. Session-scoped and autouse so both reservations land back to back in a @@ -81,14 +81,8 @@ def reserve_driver_pools(request, session_setup): with _init_cuda_context() as device: _reservation_report = va_reservation.reserve_driver_pools(device) - terminal_reporter = request.config.pluginmanager.get_plugin("terminalreporter") - if terminal_reporter is not None: - # Written here rather than only in the summary so the numbers survive a - # session that dies partway through. - terminal_reporter.write_sep("=", "cuda_core address space reservation") - for line in _reservation_report.lines(): - terminal_reporter.write_line(line) - + # Reported once, from pytest_terminal_summary. On the abort path below the + # message carries the same numbers, so nothing is lost by not printing here. if _reservation_report.failed: pytest.exit(va_reservation.build_failure_message(_reservation_report), returncode=1) diff --git a/cuda_core/tests/helpers/va_reservation.py b/cuda_core/tests/helpers/va_reservation.py index c5abf48ab42..98c5cc20da5 100644 --- a/cuda_core/tests/helpers/va_reservation.py +++ b/cuda_core/tests/helpers/va_reservation.py @@ -30,10 +30,21 @@ from __future__ import annotations import os +import sys import time from cuda.bindings import driver +if sys.platform == "win32": + from helpers import win_address_space +else: + # No Linux counterpart on purpose; see helpers/win_address_space.py. + win_address_space = None + +# Holes smaller than this are noise in the layout dump. +LAYOUT_MIN_HOLE = 1024 * 1024 * 1024 +LAYOUT_MAX_HOLES = 6 + MIB = 1024 * 1024 GIB = 1024 * MIB @@ -130,6 +141,44 @@ def format_bytes(value: int | None) -> str: return f"{value / GIB:.2f} GiB" +def free_holes() -> list[tuple[int, int]] | None: + """Unallocated holes as ``(size, base)``, largest first. None off Windows.""" + if win_address_space is None: + return None + return win_address_space.free_regions(LAYOUT_MIN_HOLE) + + +def pool_capacity(holes, pool_bytes) -> int | None: + """How many driver pools the free holes could hold between them. + + This is the number that decides the outcome, and it is neither the largest + hole nor the count of holes that clear one pool. A reservation has to fit + within a single hole, but a hole twice the size takes two -- the driver + packs them from its low end, so a lone 800 GiB hole hosts both pools just + as well as two 400 GiB ones. Summing each hole's capacity covers both. + """ + if holes is None or not pool_bytes: + return None + return sum(size // pool_bytes for size, _base in holes) + + +def layout_lines(holes, pool_bytes, label, detail: bool = True) -> list[str]: + """Render the hole structure. Base addresses show whether it is randomized.""" + if holes is None: + return [] + capacity = pool_capacity(holes, pool_bytes) + headline = f"free holes {label}: {len(holes)} >= {format_bytes(LAYOUT_MIN_HOLE)}" + if capacity is not None: + headline += f", room for {capacity} pool(s) of {format_bytes(pool_bytes)}" + out = [headline] + if detail: + for size, base in holes[:LAYOUT_MAX_HOLES]: + out.append(f" {format_bytes(size):>14} @ {base:#018x}") + if len(holes) > LAYOUT_MAX_HOLES: + out.append(f" ... and {len(holes) - LAYOUT_MAX_HOLES} smaller") + return out + + class Reservation: """One driver-managed pool that has to be materialized.""" @@ -211,7 +260,16 @@ class ReservationReport: """What the early reservations cost, for the terminal.""" def __init__( - self, device_name, device_memory, before, after, reservations, measured, seconds=0.0, unsupported=False + self, + device_name, + device_memory, + before, + after, + reservations, + measured, + seconds=0.0, + unsupported=False, + holes_before=None, ): self.device_name = device_name self.device_memory = device_memory @@ -221,6 +279,7 @@ def __init__( self.measured = measured self.seconds = seconds self.unsupported = unsupported + self.holes_before = holes_before @property def failed(self) -> list[Reservation]: @@ -264,6 +323,11 @@ def lines(self) -> list[str]: f"remaining headroom: {self.after // pool_bytes} more pool-sized " f"({format_bytes(pool_bytes)}) reservations [{self.seconds:.1f}s measuring]" ) + # Just the counts here. They are what makes a successful session + # comparable with a failed one, since the hole count is what decides the + # outcome. The addresses behind them are only worth printing when a + # reservation is actually refused; see build_failure_message. + out += layout_lines(self.holes_before, self.pool_reservation_bytes, "at session start", detail=False) return out @@ -285,6 +349,14 @@ def build_failure_message(report: ReservationReport) -> str: for item in report.failed: lines.append(f" {item.name} ({item.detail}): {item.error}") + # Each pool needs a hole of its own, so the hole structure -- not the total + # free -- is what decides this. Included here because it is the first thing + # anyone diagnosing a refusal will want. + layout = layout_lines(report.holes_before, pool_bytes, "at session start") + if layout: + lines.append("") + lines += [f" {line}" for line in layout] + return "\n".join(lines) @@ -307,6 +379,7 @@ def reserve_driver_pools(device, measure: bool = True) -> ReservationReport: started = time.perf_counter() before = largest_reservable() if measured else None elapsed = time.perf_counter() - started + holes_before = free_holes() reservations = reservations_for(device) for item in reservations: @@ -315,4 +388,13 @@ def reserve_driver_pools(device, measure: bool = True) -> ReservationReport: started = time.perf_counter() after = largest_reservable() if measured else None elapsed += time.perf_counter() - started - return ReservationReport(device.name, device_memory, before, after, reservations, measured, elapsed) + return ReservationReport( + device.name, + device_memory, + before, + after, + reservations, + measured, + elapsed, + holes_before=holes_before, + ) diff --git a/cuda_core/tests/helpers/win_address_space.py b/cuda_core/tests/helpers/win_address_space.py new file mode 100644 index 00000000000..9bc30eae4c7 --- /dev/null +++ b/cuda_core/tests/helpers/win_address_space.py @@ -0,0 +1,74 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Read this process's virtual address space layout on Windows. + +Only imported on Windows -- see helpers/va_reservation.py, which guards the +import. There is deliberately no Linux counterpart: the address-space pressure +this exists to diagnose (issue #2381) is specific to the bounded per-process +budget on Windows, and on Linux the budget is large enough that the layout is +not interesting. + +``cuMemAddressReserve`` probing can only report the largest single hole. That +turned out to be the wrong number: two pool reservations do not need one hole +twice their size, they need *two* holes, so a session can start with a smaller +largest-hole and still succeed. Walking the address space shows the whole hole +structure, which is what actually predicts the outcome. +""" + +from __future__ import annotations + +import ctypes +from ctypes import wintypes + +MEM_COMMIT = 0x1000 +MEM_RESERVE = 0x2000 +MEM_FREE = 0x10000 + +# User-mode address space ceiling; walking past it wastes time and returns nothing. +_USER_SPACE_LIMIT = 1 << 47 + + +class MEMORY_BASIC_INFORMATION(ctypes.Structure): + _fields_ = [ + ("BaseAddress", ctypes.c_void_p), + ("AllocationBase", ctypes.c_void_p), + ("AllocationProtect", wintypes.DWORD), + ("PartitionId", wintypes.WORD), + ("__alignment", wintypes.WORD), + ("RegionSize", ctypes.c_size_t), + ("State", wintypes.DWORD), + ("Protect", wintypes.DWORD), + ("Type", wintypes.DWORD), + ] + + +_kernel32 = ctypes.WinDLL("kernel32", use_last_error=True) +_kernel32.VirtualQuery.argtypes = [ctypes.c_void_p, ctypes.POINTER(MEMORY_BASIC_INFORMATION), ctypes.c_size_t] +_kernel32.VirtualQuery.restype = ctypes.c_size_t + + +def walk(): + """Yield ``(base, size, state)`` for every region in this process.""" + info = MEMORY_BASIC_INFORMATION() + address = 0 + while address < _USER_SPACE_LIMIT: + if not _kernel32.VirtualQuery(ctypes.c_void_p(address), ctypes.byref(info), ctypes.sizeof(info)): + break + size = info.RegionSize + if size == 0: + break + yield address, size, info.State + address += size + + +def free_regions(min_size: int = 0) -> list[tuple[int, int]]: + """``(size, base)`` for every unallocated hole, largest first.""" + holes = [(size, base) for base, size, state in walk() if state == MEM_FREE and size >= min_size] + holes.sort(reverse=True) + return holes + + +def reserved_total() -> int: + """Bytes reserved but not committed, i.e. address space held without memory.""" + return sum(size for _base, size, state in walk() if state == MEM_RESERVE) diff --git a/cuda_core/tests/test_helpers.py b/cuda_core/tests/test_helpers.py index 4716d7fc13b..93c15394ca6 100644 --- a/cuda_core/tests/test_helpers.py +++ b/cuda_core/tests/test_helpers.py @@ -160,9 +160,78 @@ def test_failure_message_names_the_pool_and_the_size_it_needed(): assert "cannot run on this machine" in message assert "graph memory pool" in message + assert "cuGraphAddMemAllocNode" in message assert "357.63 GiB" in message # 2x installed device memory assert "address space" in message - assert "#2381" in message + # The driver's own error, so the reader is not left guessing what refused. + assert "ZeroDivisionError" in message + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_pool_capacity_counts_two_pools_in_one_large_hole(): + # A reservation must fit within a single hole, but one hole twice the size + # takes both pools -- the driver packs them from its low end. Counting only + # the holes that clear one pool would call this a failure. + pool = 358 * va_reservation.GIB + one_big_hole = [(800 * va_reservation.GIB, 0x1000)] + two_holes = [(400 * va_reservation.GIB, 0x1000), (400 * va_reservation.GIB, 0x2000)] + + assert va_reservation.pool_capacity(one_big_hole, pool) == 2 + assert va_reservation.pool_capacity(two_holes, pool) == 2 + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_pool_capacity_ignores_holes_that_cannot_take_a_whole_pool(): + # Free space that is merely plentiful does not help; it has to be + # contiguous. This is the shape that fails on the affected machine. + pool = 358 * va_reservation.GIB + lopsided = [(700 * va_reservation.GIB, 0x1000), (300 * va_reservation.GIB, 0x2000)] + + assert va_reservation.pool_capacity(lopsided, pool) == 1 + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_layout_lines_are_empty_off_windows(): + # free_holes() returns None where VirtualQuery is unavailable; the report + # must simply omit the section rather than fail. + assert va_reservation.layout_lines(None, 358 * va_reservation.GIB, "before") == [] + assert va_reservation.pool_capacity(None, 358 * va_reservation.GIB) is None + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_neither_report_mentions_holes_off_windows(monkeypatch): + # The hole layout comes from VirtualQuery, so it is Windows-only. Everything + # else is CUDA APIs and must still be reported. Simulates the non-win32 + # branch of the guarded import in va_reservation. + monkeypatch.setattr(va_reservation, "win_address_space", None) + good = va_reservation.Reservation("default device mempool", "cuDeviceGetMemPool", lambda: None) + bad = va_reservation.Reservation("graph memory pool", "cuGraphAddMemAllocNode", lambda: 1 / 0) + good.run() + bad.run() + holes = va_reservation.free_holes() + gib = va_reservation.GIB + + report = va_reservation.ReservationReport( + "dev", 25650855936, 900 * gib, 800 * gib, [good, bad], True, holes_before=holes + ) + + assert holes is None + assert "free holes" not in "\n".join(report.lines()) + message = va_reservation.build_failure_message(report) + assert "free holes" not in message + assert "graph memory pool" in message # the rest of the report survives + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_layout_lines_report_base_addresses(): + # The bases are the point: comparing them across launches shows whether the + # layout is being randomized. + holes = [(500 * va_reservation.GIB, 0x2200000000)] + + text = "\n".join(va_reservation.layout_lines(holes, 358 * va_reservation.GIB, "before")) + + assert "room for 1 pool(s)" in text + assert "0x0000002200000000" in text @pytest.mark.agent_authored(model="claude-opus-5")