From bb7da3653d47bea883023c3f007da85b4187b8be Mon Sep 17 00:00:00 2001 From: Ralf Juengling Date: Fri, 2 Oct 2026 13:48:31 -0700 Subject: [PATCH] test: harden test_vmm_allocate_zero_size against parallel-run flakes test_vmm_allocate_zero_size asserts that grown.close() waits on the recorded deallocation stream (done.is_done after a 200ms nanosleep). It ran as PARALLEL under the free-threaded (py3.14t) job and failed on aarch64 / A100 / CUDA 13.0.2 with assert done.is_done == False, while its sibling test_vmm_close_synchronizes_recorded_streams (which is marked thread_unsafe and wraps the close in assert_no_cuda_warning) passed. The close path never raises: if cuStreamSynchronize is skipped (capture) or fails, sync_recorded_stream reports a CUDAWarning and unmaps anyway, so the failure surfaced as a bare event-check failure with no clue why. Match the sibling test: - Mark thread_unsafe so the free-threaded job runs it serially, matching the other close-synchronizes tests that record process-global warnings. - Wrap grown.close() in assert_no_cuda_warning() so a skipped/failed sync surfaces as the warning text instead of a bare assert False. If the test still fails when run serially on that job, it is a real library bug in the VMM sync path rather than a parallel-execution artifact. --- cuda_core/tests/test_memory.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/cuda_core/tests/test_memory.py b/cuda_core/tests/test_memory.py index ac67d369575..f33e17ddccf 100644 --- a/cuda_core/tests/test_memory.py +++ b/cuda_core/tests/test_memory.py @@ -1765,6 +1765,7 @@ def test_vmm_host_modify_allocation_rejects_exportable_handle_type(init_cuda): @pytest.mark.agent_authored(model="claude-fable-5-1") +@pytest.mark.thread_unsafe(reason="records process-global warnings") def test_vmm_allocate_zero_size(init_cuda): """allocate(0) returns an empty buffer without a driver call; a grow of it inherits its stream.""" device = _vmm_device_or_skip() @@ -1780,10 +1781,12 @@ def test_vmm_allocate_zero_size(init_cuda): # The grown buffer's deallocation is ordered on the stream passed to # allocate(0): its close waits for the work queued there. The sleep # kernel does not touch the buffer, so a missing wait fails the event - # check instead of faulting. + # check instead of faulting. The close is wrapped so a skipped sync + # surfaces as a CUDAWarning instead of a bare event-check failure. NanosleepKernel(device, sleep_duration_ms=200).launch(s) done = s.record() - grown.close() + with assert_no_cuda_warning(): + grown.close() assert done.is_done buf.close() finally: