From 8567243eb1a37e50e7d55a5942f6070c3d3b71a8 Mon Sep 17 00:00:00 2001 From: Ralf Juengling Date: Thu, 6 Aug 2026 11:19:32 -0700 Subject: [PATCH 1/2] test(cuda_core): reserve the driver's two pool address windows at session start The driver's default memory pool and graph memory pool each permanently reserve roughly twice the installed device memory of virtual address space, and neither can be capped or released. On a large-memory GPU with a bounded per-process address space, whether the second reservation still finds a contiguous range depends on how fragmented the space has become by the time some test needs it, which is what makes the failures in #2381 intermittent. Measurements and documentation references: https://github.com/NVIDIA/cuda-python/issues/2381#issuecomment-5207900890 Take both reservations up front in a session-scoped autouse fixture so they land back to back in a nearly empty address space. This does not reduce the footprint; it makes the outcome deterministic. If either is refused the session aborts with an explanation, since every later test needing that pool would fail the same way and a cascade of identical OOM errors says nothing about the cause. Available address space is reported before and after via cuMemAddressReserve rather than an OS query, so the measurement behaves the same on Windows and Linux. CUDA_CORE_TEST_SKIP_EARLY_RESERVATION=1 disables the fixture; CUDA_CORE_TEST_FORCE_RESERVATION_FAILURE=1 exercises the abort path. --- cuda_core/tests/conftest.py | 48 ++++ cuda_core/tests/helpers/va_reservation.py | 319 ++++++++++++++++++++++ cuda_core/tests/test_helpers.py | 104 +++++++ 3 files changed, 471 insertions(+) create mode 100644 cuda_core/tests/helpers/va_reservation.py diff --git a/cuda_core/tests/conftest.py b/cuda_core/tests/conftest.py index dfe97b265eb..cb405481e76 100644 --- a/cuda_core/tests/conftest.py +++ b/cuda_core/tests/conftest.py @@ -31,6 +31,7 @@ from cuda_python_test_helpers.marks import skipif_need_cuda_headers # noqa: F401 (re-exported for tests) from cuda_python_test_helpers.mempool import xfail_if_mempool_oom +from helpers import va_reservation from helpers.constants import POOL_SIZE import cuda.core @@ -55,6 +56,53 @@ def pytest_configure(config): config.pluginmanager.register(_CudaCoreParallelPlugin(), name="_cuda_core_parallel_plugin") +_reservation_report = None + + +@pytest.fixture(scope="session", autouse=True) +def reserve_driver_pools(request, session_setup): + """Take the driver's two large address-space reservations before anything else. + + Session-scoped and autouse so both reservations land back to back in a + nearly empty address space, ahead of any test that could fragment it. + Depends on session_setup for cuInit. See helpers/va_reservation.py for why + this matters, and issue #2381. + + Aborts the session with an explanation if either reservation is refused: + every later test that needs the pool would fail the same way, and the + resulting cascade of identical OOM errors says nothing about the cause. + """ + global _reservation_report + + if int(os.environ.get("CUDA_CORE_TEST_SKIP_EARLY_RESERVATION", 0)) != 0: + yield + return + + with _init_cuda_context() as device: + _reservation_report = va_reservation.reserve_driver_pools(device) + + terminal_reporter = request.config.pluginmanager.get_plugin("terminalreporter") + if terminal_reporter is not None: + # Written here rather than only in the summary so the numbers survive a + # session that dies partway through. + terminal_reporter.write_sep("=", "cuda_core address space reservation") + for line in _reservation_report.lines(): + terminal_reporter.write_line(line) + + if _reservation_report.failed: + pytest.exit(va_reservation.build_failure_message(_reservation_report), returncode=1) + + yield + + +def pytest_terminal_summary(terminalreporter): + if _reservation_report is None or _reservation_report.failed: + return + terminalreporter.write_sep("=", "cuda_core address space reservation") + for line in _reservation_report.lines(): + terminalreporter.write_line(line) + + @contextmanager def _init_cuda_context(): # TODO: rename this to e.g. init_context diff --git a/cuda_core/tests/helpers/va_reservation.py b/cuda_core/tests/helpers/va_reservation.py new file mode 100644 index 00000000000..eed57251b8b --- /dev/null +++ b/cuda_core/tests/helpers/va_reservation.py @@ -0,0 +1,319 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Force the driver's two large address-space reservations up front (issue #2381). + +The driver keeps two pools per device that it never gives back, and each one +reserves a virtual address window of roughly twice the installed device memory +when it is first touched: + +- the **default device mempool**, reserved by ``cuDeviceGetMemPool`` +- the **graph memory pool**, reserved by ``cuGraphAddMemAllocNode`` at node + creation time + +Neither is capped by ``max_size`` and neither is released by +``cuDeviceGraphMemTrim``, ``cuGraphDestroy``, or anything else short of process +exit. On a large-memory GPU with a bounded per-process address space -- 357 GiB +per pool on a 179 GiB device -- the two together consume most of the budget, and +whether the *second* one finds a contiguous range depends on how fragmented the +space has become. That is what makes the full-suite failures intermittent, and +why they always begin at the first test to need whichever pool came second. + +Taking both reservations at session start, back to back into a nearly empty +address space, removes test order and accumulated fragmentation from the +question. It does not reduce the footprint; it makes the outcome deterministic. + +Measurement here goes through ``cuMemAddressReserve``, not the OS, so it behaves +the same on Windows and Linux. +""" + +from __future__ import annotations + +import os +import time + +from cuda.bindings import driver + +MIB = 1024 * 1024 +GIB = 1024 * MIB + +# cuMemAddressReserve wants a power-of-two alignment and a size that is a +# multiple of it. 2 MiB is the granularity the driver uses for pools. +VA_ALIGNMENT = 2 * MIB +# Above any plausible per-process budget, so the descending probe below always +# starts from a size that fails. +MAX_PROBE_BYTES = 1 << 46 + +# Each driver-managed pool reserves about this multiple of installed device +# memory. Used to express remaining headroom in units of "one more pool". +POOL_RESERVATION_MULTIPLE = 2 + + +def align_up(size: int) -> int: + """Round to a multiple of the reservation alignment. + + Device memory sizes are not generally a multiple of it, and + cuMemAddressReserve rejects a size that is not with + CUDA_ERROR_INVALID_VALUE -- which would otherwise read as "no address space + left" at every size probed. + """ + return ((size + VA_ALIGNMENT - 1) // VA_ALIGNMENT) * VA_ALIGNMENT + + +def _reserve_and_release(size: int) -> bool: + """True if the driver still grants a contiguous reservation of ``size``. + + Reserving costs address space but no memory, so this reads what the address + space can satisfy without perturbing it. Raises if the release fails, since + a leaked reservation would corrupt every later measurement. + """ + err, ptr = driver.cuMemAddressReserve(align_up(size), VA_ALIGNMENT, 0, 0) + if err != driver.CUresult.CUDA_SUCCESS: + return False + (err,) = driver.cuMemAddressFree(ptr, align_up(size)) + if err != driver.CUresult.CUDA_SUCCESS: + raise RuntimeError(f"cuMemAddressFree({size:#x}) -> {err!r}; address space measurement is unreliable") + return True + + +def largest_reservable(reserve=None, max_bytes: int = MAX_PROBE_BYTES, refine_steps: int = 4) -> int: + """Largest contiguous reservation the driver still grants. + + Halves down from ``max_bytes`` rather than doubling up from the granularity, + because a *refused* reservation allocates nothing and returns immediately + while releasing a granted one is expensive -- hundreds of milliseconds for a + large range, and seconds on some configurations. Descending pays that cost + once instead of once per rung. + + Halving alone only resolves to a power of two, which is too coarse to show + what the reservations cost when the budget is much larger than they are, so + ``refine_steps`` bisections then narrow the answer. Each bisection risks one + more expensive release, hence the small default. + + ``reserve`` is injectable so the search can be tested without a GPU. + """ + reserve = _reserve_and_release if reserve is None else reserve + + size = max_bytes + while size >= VA_ALIGNMENT and not reserve(size): + size //= 2 + if size < VA_ALIGNMENT: + return 0 + if size == max_bytes: + return size # nothing was refused, so there is no bracket to narrow + + low, high = size, size * 2 # high was refused on the way down + for _ in range(refine_steps): + middle = ((low + high) // 2 // VA_ALIGNMENT) * VA_ALIGNMENT + if middle <= low or middle >= high: + break + if reserve(middle): + low = middle + else: + high = middle + return low + + +def vmm_supported(device_id: int = 0) -> bool: + """True if this device exposes the virtual memory management APIs.""" + err, dev = driver.cuDeviceGet(device_id) + if err != driver.CUresult.CUDA_SUCCESS: + return False + attribute = driver.CUdevice_attribute.CU_DEVICE_ATTRIBUTE_VIRTUAL_MEMORY_MANAGEMENT_SUPPORTED + err, supported = driver.cuDeviceGetAttribute(attribute, dev) + return err == driver.CUresult.CUDA_SUCCESS and bool(supported) + + +def format_bytes(value: int | None) -> str: + if value is None: + return "unknown" + return f"{value / GIB:.2f} GiB" + + +class Reservation: + """One driver-managed pool that has to be materialized.""" + + def __init__(self, name: str, detail: str, materialize): + self.name = name + self.detail = detail + self._materialize = materialize + self.error: str | None = None + + def run(self) -> bool: + """Materialize the pool. Returns True on success, recording any error.""" + try: + self._materialize() + except Exception as exc: # surfaced to the user by build_failure_message + self.error = f"{type(exc).__name__}: {exc}" + return False + return True + + @property + def succeeded(self) -> bool: + return self.error is None + + +def _materialize_default_mempool(device): + """Touch the device's default memory pool. + + cuDeviceGetMemPool is the call that makes the reservation; the allocation + only proves the pool is usable afterwards. + """ + buffer = device.memory_resource.allocate(8, stream=device.default_stream) + device.default_stream.sync() + buffer.close(device.default_stream) + device.default_stream.sync() + + +def _materialize_graph_mempool(device): + """Add one graph memory-allocation node, which reserves the graph pool. + + The graph is discarded immediately: the address reservation it triggers + outlives it, which is the whole point. + """ + from cuda.core.graph import GraphDefinition + + definition = GraphDefinition() + definition.allocate(1024) + del definition + + +def _forced_failure(): + """Stand-in for a refused reservation, so the abort path can be exercised. + + Set CUDA_CORE_TEST_FORCE_RESERVATION_FAILURE=1 to check what this suite + reports on a machine whose address space is too small, without needing one. + """ + raise RuntimeError("CUDA_ERROR_OUT_OF_MEMORY: simulated refusal (CUDA_CORE_TEST_FORCE_RESERVATION_FAILURE is set)") + + +def reservations_for(device) -> list[Reservation]: + if os.environ.get("CUDA_CORE_TEST_FORCE_RESERVATION_FAILURE", "0") not in ("0", ""): + return [ + Reservation("default device mempool", "cuDeviceGetMemPool", _forced_failure), + Reservation("graph memory pool", "cuGraphAddMemAllocNode", _forced_failure), + ] + return [ + Reservation( + "default device mempool", + "cuDeviceGetMemPool", + lambda: _materialize_default_mempool(device), + ), + Reservation( + "graph memory pool", + "cuGraphAddMemAllocNode", + lambda: _materialize_graph_mempool(device), + ), + ] + + +class ReservationReport: + """What the early reservations cost, for the terminal.""" + + def __init__(self, device_name, device_memory, before, after, reservations, measured, seconds=0.0): + self.device_name = device_name + self.device_memory = device_memory + self.before = before + self.after = after + self.reservations = reservations + self.measured = measured + self.seconds = seconds + + @property + def failed(self) -> list[Reservation]: + return [item for item in self.reservations if not item.succeeded] + + @property + def pool_reservation_bytes(self) -> int | None: + if self.device_memory is None: + return None + return align_up(POOL_RESERVATION_MULTIPLE * self.device_memory) + + def lines(self) -> list[str]: + out = [f"device 0: {self.device_name} ({format_bytes(self.device_memory)} device memory)"] + if self.measured: + out.append(f"largest reservable range before: {format_bytes(self.before)}") + else: + out.append("largest reservable range: not measured (no virtual memory management support)") + + for item in self.reservations: + status = "reserved" if item.succeeded else f"FAILED: {item.error}" + out.append(f" {item.name:<24} {item.detail:<24} {status}") + + if self.measured: + # A drop here is a *lower* bound on what was taken: the driver may + # carve its reservations out of a region other than the largest + # hole, in which case the largest hole does not move at all. + change = None if self.before is None or self.after is None else self.before - self.after + if change is None: + note = "" + elif change > 0: + note = f" (largest hole shrank by {format_bytes(change)})" + else: + note = " (largest hole unchanged)" + out.append(f"largest reservable range after: {format_bytes(self.after)}{note}") + pool_bytes = self.pool_reservation_bytes + if pool_bytes and self.after is not None: + out.append( + f"remaining headroom: {self.after // pool_bytes} more pool-sized " + f"({format_bytes(pool_bytes)}) reservations [{self.seconds:.1f}s measuring]" + ) + return out + + +def build_failure_message(report: ReservationReport) -> str: + """Explain, for a human, why this machine cannot run the suite.""" + pool_bytes = report.pool_reservation_bytes + failed = ", ".join(item.name for item in report.failed) + lines = [ + "", + "cuda_core tests cannot run on this machine: the CUDA driver could not reserve", + f"address space for {failed}.", + "", + f" device 0 {report.device_name}", + f" installed device memory {format_bytes(report.device_memory)}", + f" needed per driver-managed pool {format_bytes(pool_bytes)} of *virtual address space*", + f" largest range still available {format_bytes(report.after if report.measured else None)}", + "", + ] + for item in report.failed: + lines.append(f" {item.name} ({item.detail}): {item.error}") + lines += [ + "", + "The driver keeps two pools per device -- the default memory pool and the graph", + "memory pool -- and reserves roughly twice the installed device memory of address", + "space for each one the first time it is used. Neither reservation can be capped,", + "and neither is released before the process exits. On a large-memory GPU with a", + "bounded per-process address space the two together can exceed the budget, and no", + "amount of freeing device memory helps: the exhausted resource is address space,", + "not memory. Expect cuMemGetInfo to report plenty free while this fails.", + "", + "Options: run on a device with less memory, run the graph tests in a separate", + "process from the rest of the suite so the two reservations never coexist, or on", + "Windows check whether the driver model (WDDM/MCDM/TCC) bounds the address space", + "more tightly than expected. See issue #2381.", + "", + ] + return "\n".join(lines) + + +def reserve_driver_pools(device, measure: bool = True) -> ReservationReport: + """Materialize both driver-managed pools, measuring address space around them.""" + device_memory = None + err, _free, total = driver.cuMemGetInfo() + if err == driver.CUresult.CUDA_SUCCESS: + device_memory = int(total) + + measured = measure and vmm_supported(device.device_id) + started = time.perf_counter() + before = largest_reservable() if measured else None + elapsed = time.perf_counter() - started + + reservations = reservations_for(device) + for item in reservations: + item.run() + + started = time.perf_counter() + after = largest_reservable() if measured else None + elapsed += time.perf_counter() - started + return ReservationReport(device.name, device_memory, before, after, reservations, measured, elapsed) diff --git a/cuda_core/tests/test_helpers.py b/cuda_core/tests/test_helpers.py index 43dbf8887e2..4716d7fc13b 100644 --- a/cuda_core/tests/test_helpers.py +++ b/cuda_core/tests/test_helpers.py @@ -5,6 +5,7 @@ import time import pytest +from helpers import va_reservation from helpers.buffers import PatternGen, compare_equal_buffers, make_scratch_buffer from helpers.latch import LatchKernel from helpers.logging import TimestampedLogger @@ -80,3 +81,106 @@ def test_patterngen_values(): pgen = PatternGen(device, NBYTES) pgen.verify_buffer(ones, value=1) pgen.verify_buffer(twos, value=2) + + +# helpers.va_reservation (issue #2381). The GPU-dependent part runs once per +# session in an autouse fixture, so only the pure logic is covered here. + + +def _fake_reserve(limit, log=None): + """Grants any reservation up to ``limit``, recording what was asked.""" + + def reserve(size): + if log is not None: + log.append(size) + return size <= limit + + return reserve + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_va_probe_finds_the_boundary(): + limit = 700 * va_reservation.GIB + found = va_reservation.largest_reservable(reserve=_fake_reserve(limit)) + + # Refinement should land within a few percent, never above the real limit. + assert found <= limit + assert found > limit * 0.9 + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_va_probe_descends_so_only_one_grant_is_paid_for(): + # Releasing a granted reservation is the expensive half; a refused one costs + # nothing. Ascending from the granularity would pay for every rung. + asked = [] + limit = 700 * va_reservation.GIB + va_reservation.largest_reservable(reserve=_fake_reserve(limit, asked), refine_steps=0) + + assert sum(1 for size in asked if size <= limit) == 1 + assert asked[0] == va_reservation.MAX_PROBE_BYTES + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_va_probe_reports_zero_when_nothing_can_be_reserved(): + assert va_reservation.largest_reservable(reserve=_fake_reserve(0)) == 0 + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_va_probe_only_asks_for_aligned_sizes(): + # Device memory is not a multiple of the 2 MiB granularity, and an unaligned + # ask fails with CUDA_ERROR_INVALID_VALUE at every size -- which would read + # as an exhausted address space rather than as a bug here. + asked = [] + va_reservation.largest_reservable(reserve=_fake_reserve(3 * 25650855936, asked)) + + assert asked and all(size % va_reservation.VA_ALIGNMENT == 0 for size in asked) + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_reservation_records_failure_without_raising(): + def boom(): + raise RuntimeError("CUDA_ERROR_OUT_OF_MEMORY: nope") + + item = va_reservation.Reservation("default device mempool", "cuDeviceGetMemPool", boom) + + assert item.run() is False + assert item.succeeded is False + assert "CUDA_ERROR_OUT_OF_MEMORY" in item.error + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_failure_message_names_the_pool_and_the_size_it_needed(): + failed = va_reservation.Reservation("graph memory pool", "cuGraphAddMemAllocNode", lambda: 1 / 0) + failed.run() + report = va_reservation.ReservationReport( + "NVIDIA Graphics Device", 191998918656, 300 * va_reservation.GIB, 300 * va_reservation.GIB, [failed], True + ) + + message = va_reservation.build_failure_message(report) + + assert "cannot run on this machine" in message + assert "graph memory pool" in message + assert "357.63 GiB" in message # 2x installed device memory + assert "address space" in message + assert "#2381" in message + + +@pytest.mark.agent_authored(model="claude-opus-5") +def test_report_lines_cover_both_driver_pools(): + ok_pools = [ + va_reservation.Reservation("default device mempool", "cuDeviceGetMemPool", lambda: None), + va_reservation.Reservation("graph memory pool", "cuGraphAddMemAllocNode", lambda: None), + ] + for item in ok_pools: + item.run() + report = va_reservation.ReservationReport( + "dev", 25650855936, 900 * va_reservation.GIB, 800 * va_reservation.GIB, ok_pools, True + ) + + text = "\n".join(report.lines()) + + assert report.failed == [] + assert "default device mempool" in text + assert "graph memory pool" in text + assert "shrank by 100.00 GiB" in text + assert "more pool-sized" in text From 51f6ff4911f4e8b0ac04e85bed46fe592fbc2d15 Mon Sep 17 00:00:00 2001 From: Ralf Juengling Date: Thu, 6 Aug 2026 11:23:13 -0700 Subject: [PATCH 2/2] test(cuda_core): skip early pool reservation when mempools are unsupported Both driver-managed pools require mempool support, so on a device without it there is nothing to reserve and nothing to pre-empt. Reporting a failed reservation there would abort the session, when the correct behaviour is to carry on: the tests that need pools already skip themselves via Device.properties.memory_pools_supported, and the rest still run. --- cuda_core/tests/helpers/va_reservation.py | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/cuda_core/tests/helpers/va_reservation.py b/cuda_core/tests/helpers/va_reservation.py index eed57251b8b..df1bdc23f36 100644 --- a/cuda_core/tests/helpers/va_reservation.py +++ b/cuda_core/tests/helpers/va_reservation.py @@ -210,7 +210,9 @@ def reservations_for(device) -> list[Reservation]: class ReservationReport: """What the early reservations cost, for the terminal.""" - def __init__(self, device_name, device_memory, before, after, reservations, measured, seconds=0.0): + def __init__( + self, device_name, device_memory, before, after, reservations, measured, seconds=0.0, unsupported=False + ): self.device_name = device_name self.device_memory = device_memory self.before = before @@ -218,6 +220,7 @@ def __init__(self, device_name, device_memory, before, after, reservations, meas self.reservations = reservations self.measured = measured self.seconds = seconds + self.unsupported = unsupported @property def failed(self) -> list[Reservation]: @@ -231,6 +234,9 @@ def pool_reservation_bytes(self) -> int | None: def lines(self) -> list[str]: out = [f"device 0: {self.device_name} ({format_bytes(self.device_memory)} device memory)"] + if self.unsupported: + out.append("device does not support memory pools; nothing to reserve") + return out if self.measured: out.append(f"largest reservable range before: {format_bytes(self.before)}") else: @@ -298,12 +304,20 @@ def build_failure_message(report: ReservationReport) -> str: def reserve_driver_pools(device, measure: bool = True) -> ReservationReport: - """Materialize both driver-managed pools, measuring address space around them.""" + """Materialize both driver-managed pools, measuring address space around them. + + Both pools require mempool support, so on a device without it there is + nothing to reserve and nothing to pre-empt. Skip rather than fail: the tests + that need pools skip themselves on such a device, and the rest still run. + """ device_memory = None err, _free, total = driver.cuMemGetInfo() if err == driver.CUresult.CUDA_SUCCESS: device_memory = int(total) + if not device.properties.memory_pools_supported: + return ReservationReport(device.name, device_memory, None, None, [], measured=False, unsupported=True) + measured = measure and vmm_supported(device.device_id) started = time.perf_counter() before = largest_reservable() if measured else None