diff --git a/tests/v1/kv_offload/cpu/test_gpu_worker.py b/tests/v1/kv_offload/cpu/test_gpu_worker.py index d192b04a07b1..1d16e1c99ae7 100644 --- a/tests/v1/kv_offload/cpu/test_gpu_worker.py +++ b/tests/v1/kv_offload/cpu/test_gpu_worker.py @@ -420,3 +420,19 @@ def test_transfer_multi_group( handlers.cpu_to_gpu_handler.shutdown() handlers.gpu_to_cpu_handler.shutdown() + + +def test_max_pinnable_mmap_bytes(): + import resource + + from vllm.v1.kv_offload.cpu.gpu_worker import _max_pinnable_mmap_bytes + + # With a finite memlock limit, the bound equals that limit. + resource.setrlimit(resource.RLIMIT_MEMLOCK, (1 << 20, 2 << 20)) + assert _max_pinnable_mmap_bytes() == 2 << 20 + + # With an unlimited memlock limit, the bound is a fraction of physical RAM. + resource.setrlimit(resource.RLIMIT_MEMLOCK, (resource.RLIM_INFINITY, resource.RLIM_INFINITY)) + bound = _max_pinnable_mmap_bytes() + assert bound is not None + assert bound > 0 diff --git a/vllm/v1/kv_offload/cpu/gpu_worker.py b/vllm/v1/kv_offload/cpu/gpu_worker.py index 81545281b642..d82bc17486fe 100644 --- a/vllm/v1/kv_offload/cpu/gpu_worker.py +++ b/vllm/v1/kv_offload/cpu/gpu_worker.py @@ -1,6 +1,8 @@ # SPDX-License-Identifier: Apache-2.0 # SPDX-FileCopyrightText: Copyright contributors to the vLLM project import functools +import os +import resource import time from collections import deque from dataclasses import dataclass @@ -123,6 +125,27 @@ def compute_sub_block_ptrs( output[:] = flat[skip_count : skip_count + num_sub_blocks] +def _max_pinnable_mmap_bytes() -> int | None: + """Upper bound on how much of an mmap region we should try to pin. + + Uses RLIMIT_MEMLOCK when it is finite. When it is unlimited, fall back + to a conservative fraction of physical RAM, because cudaHostRegister can + still fail on extremely large tmpfs-backed mmap regions (e.g. >100 GiB) + and a failed call may leave the CUDA context in a bad state. + """ + hard_limit = resource.getrlimit(resource.RLIMIT_MEMLOCK)[1] + if hard_limit != resource.RLIMIT_INFINITY: + return hard_limit + try: + page_size = os.sysconf(os.sysconf_names["SC_PAGE_SIZE"]) + num_pages = os.sysconf(os.sysconf_names["SC_PHYS_PAGES"]) + total_ram = page_size * num_pages + except (AttributeError, OSError, ValueError): + return None + # Do not try to pin more than half of physical RAM. + return total_ram // 2 + + def pin_mmap_region(region: SharedOffloadRegion) -> None: """Register the entire mmap as CUDA pinned memory via cudaHostRegister.""" if not current_platform.is_cuda_alike(): @@ -133,6 +156,24 @@ def pin_mmap_region(region: SharedOffloadRegion) -> None: ) return + # Avoid cudaHostRegister when the region is larger than what the process + # can pin. On failure, cudaHostRegister can corrupt the CUDA context and + # cause later CUDA operations (e.g. warmup) to fail with errors such as + # "CUDA error: invalid argument". This is especially common with CPU KV + # cache offloading, where the mmap can be 100+ GiB while the default + # docker/container memlock limit is only a few MiB. + max_pin = _max_pinnable_mmap_bytes() + if max_pin is not None and region.total_size_bytes > max_pin: + logger.info( + "Skipping mmap host registration: region size %.2f GiB exceeds " + "the pinnable memory bound %.2f GiB. Transfers will use unpinned " + "DMA. Raise the memlock limit (e.g. --ulimit memlock=-1:-1) to " + "pin this region.", + region.total_size_bytes / (1 << 30), + max_pin / (1 << 30), + ) + return + rank = region.rank base_ptr = region._base.data_ptr()