Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions tests/v1/kv_offload/cpu/test_gpu_worker.py
Original file line number Diff line number Diff line change
Expand Up @@ -420,3 +420,19 @@ def test_transfer_multi_group(

handlers.cpu_to_gpu_handler.shutdown()
handlers.gpu_to_cpu_handler.shutdown()


def test_max_pinnable_mmap_bytes():
import resource

from vllm.v1.kv_offload.cpu.gpu_worker import _max_pinnable_mmap_bytes

# With a finite memlock limit, the bound equals that limit.
resource.setrlimit(resource.RLIMIT_MEMLOCK, (1 << 20, 2 << 20))
assert _max_pinnable_mmap_bytes() == 2 << 20

# With an unlimited memlock limit, the bound is a fraction of physical RAM.
resource.setrlimit(resource.RLIMIT_MEMLOCK, (resource.RLIM_INFINITY, resource.RLIM_INFINITY))
bound = _max_pinnable_mmap_bytes()
assert bound is not None
assert bound > 0
41 changes: 41 additions & 0 deletions vllm/v1/kv_offload/cpu/gpu_worker.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,8 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
import functools
import os
import resource
import time
from collections import deque
from dataclasses import dataclass
Expand Down Expand Up @@ -123,6 +125,27 @@ def compute_sub_block_ptrs(
output[:] = flat[skip_count : skip_count + num_sub_blocks]


def _max_pinnable_mmap_bytes() -> int | None:
"""Upper bound on how much of an mmap region we should try to pin.

Uses RLIMIT_MEMLOCK when it is finite. When it is unlimited, fall back
to a conservative fraction of physical RAM, because cudaHostRegister can
still fail on extremely large tmpfs-backed mmap regions (e.g. >100 GiB)
and a failed call may leave the CUDA context in a bad state.
"""
hard_limit = resource.getrlimit(resource.RLIMIT_MEMLOCK)[1]
if hard_limit != resource.RLIMIT_INFINITY:
return hard_limit
try:
page_size = os.sysconf(os.sysconf_names["SC_PAGE_SIZE"])
num_pages = os.sysconf(os.sysconf_names["SC_PHYS_PAGES"])
total_ram = page_size * num_pages
except (AttributeError, OSError, ValueError):
return None
# Do not try to pin more than half of physical RAM.
return total_ram // 2


def pin_mmap_region(region: SharedOffloadRegion) -> None:
"""Register the entire mmap as CUDA pinned memory via cudaHostRegister."""
if not current_platform.is_cuda_alike():
Expand All @@ -133,6 +156,24 @@ def pin_mmap_region(region: SharedOffloadRegion) -> None:
)
return

# Avoid cudaHostRegister when the region is larger than what the process
# can pin. On failure, cudaHostRegister can corrupt the CUDA context and
# cause later CUDA operations (e.g. warmup) to fail with errors such as
# "CUDA error: invalid argument". This is especially common with CPU KV
# cache offloading, where the mmap can be 100+ GiB while the default
# docker/container memlock limit is only a few MiB.
max_pin = _max_pinnable_mmap_bytes()
if max_pin is not None and region.total_size_bytes > max_pin:
logger.info(
"Skipping mmap host registration: region size %.2f GiB exceeds "
"the pinnable memory bound %.2f GiB. Transfers will use unpinned "
"DMA. Raise the memlock limit (e.g. --ulimit memlock=-1:-1) to "
"pin this region.",
region.total_size_bytes / (1 << 30),
max_pin / (1 << 30),
)
return

rank = region.rank

base_ptr = region._base.data_ptr()
Expand Down