From b9d47cc08f507c8c0008dc63b1bd6e526767b391 Mon Sep 17 00:00:00 2001 From: AI Assistant Date: Sat, 5 Sep 2026 20:09:30 +0000 Subject: [PATCH] Skip cudaHostRegister for oversized CPU offload mmap regions pin_mmap_region() calls cudaHostRegister() on the entire shared mmap used for CPU KV cache offloading. When the offload region is larger than the process memlock limit (common in containerized deployments with --kv-offloading-size above a few GiB), the call fails with cudaErrorInvalidValue and leaves the CUDA context in a bad state. Later CUDA operations then fail with errors such as "CUDA error: invalid argument", typically during model warmup. Add _max_pinnable_mmap_bytes() which uses RLIMIT_MEMLOCK when finite, and falls back to half of physical RAM when the limit is unlimited. Skip the cudaHostRegister call when the mmap exceeds the bound and log an informative message instead. Fixes startup crashes with large CPU RAM KV cache offloading. --- tests/v1/kv_offload/cpu/test_gpu_worker.py | 16 +++++++++ vllm/v1/kv_offload/cpu/gpu_worker.py | 41 ++++++++++++++++++++++ 2 files changed, 57 insertions(+) diff --git a/tests/v1/kv_offload/cpu/test_gpu_worker.py b/tests/v1/kv_offload/cpu/test_gpu_worker.py index d192b04a07b1..1d16e1c99ae7 100644 --- a/tests/v1/kv_offload/cpu/test_gpu_worker.py +++ b/tests/v1/kv_offload/cpu/test_gpu_worker.py @@ -420,3 +420,19 @@ def test_transfer_multi_group( handlers.cpu_to_gpu_handler.shutdown() handlers.gpu_to_cpu_handler.shutdown() + + +def test_max_pinnable_mmap_bytes(): + import resource + + from vllm.v1.kv_offload.cpu.gpu_worker import _max_pinnable_mmap_bytes + + # With a finite memlock limit, the bound equals that limit. + resource.setrlimit(resource.RLIMIT_MEMLOCK, (1 << 20, 2 << 20)) + assert _max_pinnable_mmap_bytes() == 2 << 20 + + # With an unlimited memlock limit, the bound is a fraction of physical RAM. + resource.setrlimit(resource.RLIMIT_MEMLOCK, (resource.RLIM_INFINITY, resource.RLIM_INFINITY)) + bound = _max_pinnable_mmap_bytes() + assert bound is not None + assert bound > 0 diff --git a/vllm/v1/kv_offload/cpu/gpu_worker.py b/vllm/v1/kv_offload/cpu/gpu_worker.py index 81545281b642..d82bc17486fe 100644 --- a/vllm/v1/kv_offload/cpu/gpu_worker.py +++ b/vllm/v1/kv_offload/cpu/gpu_worker.py @@ -1,6 +1,8 @@ # SPDX-License-Identifier: Apache-2.0 # SPDX-FileCopyrightText: Copyright contributors to the vLLM project import functools +import os +import resource import time from collections import deque from dataclasses import dataclass @@ -123,6 +125,27 @@ def compute_sub_block_ptrs( output[:] = flat[skip_count : skip_count + num_sub_blocks] +def _max_pinnable_mmap_bytes() -> int | None: + """Upper bound on how much of an mmap region we should try to pin. + + Uses RLIMIT_MEMLOCK when it is finite. When it is unlimited, fall back + to a conservative fraction of physical RAM, because cudaHostRegister can + still fail on extremely large tmpfs-backed mmap regions (e.g. >100 GiB) + and a failed call may leave the CUDA context in a bad state. + """ + hard_limit = resource.getrlimit(resource.RLIMIT_MEMLOCK)[1] + if hard_limit != resource.RLIMIT_INFINITY: + return hard_limit + try: + page_size = os.sysconf(os.sysconf_names["SC_PAGE_SIZE"]) + num_pages = os.sysconf(os.sysconf_names["SC_PHYS_PAGES"]) + total_ram = page_size * num_pages + except (AttributeError, OSError, ValueError): + return None + # Do not try to pin more than half of physical RAM. + return total_ram // 2 + + def pin_mmap_region(region: SharedOffloadRegion) -> None: """Register the entire mmap as CUDA pinned memory via cudaHostRegister.""" if not current_platform.is_cuda_alike(): @@ -133,6 +156,24 @@ def pin_mmap_region(region: SharedOffloadRegion) -> None: ) return + # Avoid cudaHostRegister when the region is larger than what the process + # can pin. On failure, cudaHostRegister can corrupt the CUDA context and + # cause later CUDA operations (e.g. warmup) to fail with errors such as + # "CUDA error: invalid argument". This is especially common with CPU KV + # cache offloading, where the mmap can be 100+ GiB while the default + # docker/container memlock limit is only a few MiB. + max_pin = _max_pinnable_mmap_bytes() + if max_pin is not None and region.total_size_bytes > max_pin: + logger.info( + "Skipping mmap host registration: region size %.2f GiB exceeds " + "the pinnable memory bound %.2f GiB. Transfers will use unpinned " + "DMA. Raise the memlock limit (e.g. --ulimit memlock=-1:-1) to " + "pin this region.", + region.total_size_bytes / (1 << 30), + max_pin / (1 << 30), + ) + return + rank = region.rank base_ptr = region._base.data_ptr()