forked from Karylab-cklius/vllm
Compare commits
23
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7fc60bb26d | ||
|
|
1f2c614c27 | ||
|
|
183a430c13 | ||
|
|
a346d589f5 | ||
|
|
7df3d7dada | ||
|
|
8dd1b702f2 | ||
|
|
f57ac274b2 | ||
|
|
6e919960af | ||
|
|
c88d3d4775 | ||
|
|
ab7fcbdd5d | ||
|
|
3b4a76b63f | ||
|
|
cc22621b51 | ||
|
|
77148992cf | ||
|
|
891cc4b9c5 | ||
|
|
1bdf9810aa | ||
|
|
ebfbcfe46a | ||
|
|
e9de72fe6c | ||
|
|
d272418f45 | ||
|
|
7ff7f5c8eb | ||
|
|
f24d8d5bb4 | ||
|
|
dced290769 | ||
|
|
93bad11912 | ||
|
|
928e13af5f |
@@ -647,7 +647,7 @@ steps:
|
||||
- pytest -v -s v1/cudagraph/test_cudagraph_mode.py
|
||||
|
||||
- label: e2e Core (1 GPU) # TBD
|
||||
timeout_in_minutes: 180
|
||||
timeout_in_minutes: 35
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
|
||||
agent_pool: mi250_1
|
||||
optional: true
|
||||
@@ -2075,19 +2075,6 @@ steps:
|
||||
- export VLLM_ALLOW_INSECURE_SERIALIZATION=1
|
||||
- pytest -v -s v1/spec_decode/test_acceptance_length.py -m slow_test
|
||||
|
||||
- label: e2e Core (1 GPU) # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_1
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/v1/
|
||||
- tests/v1/e2e/
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- pytest -v -s v1/e2e/general --ignore v1/e2e/general/test_async_scheduling.py
|
||||
|
||||
- label: e2e Scheduling (1 GPU) # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
|
||||
@@ -74,6 +74,16 @@ steps:
|
||||
- tests/v1/e2e/general/
|
||||
commands:
|
||||
- pytest -v -s v1/e2e/general --ignore v1/e2e/general/test_async_scheduling.py
|
||||
mirror:
|
||||
amd:
|
||||
device: mi250_1
|
||||
timeout_in_minutes: 35
|
||||
depends_on:
|
||||
- image-build-amd
|
||||
source_file_dependencies:
|
||||
- vllm/v1/
|
||||
- tests/v1/e2e/general/
|
||||
- vllm/platforms/rocm.py
|
||||
|
||||
- label: V1 e2e (2 GPUs)
|
||||
key: v1-e2e-2-gpus
|
||||
|
||||
@@ -109,6 +109,7 @@ steps:
|
||||
- image-build-amd
|
||||
commands:
|
||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- export PYTORCH_ROCM_ARCH=gfx942 # Limit Quark compilation to save time
|
||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-mi3xx.txt
|
||||
|
||||
- label: MoE Refactor Integration Test (H100 - TEMPORARY)
|
||||
|
||||
@@ -268,9 +268,14 @@ int64_t sm100_cutlass_mla_get_workspace_size(int64_t max_seq_len, int64_t num_ba
|
||||
using TileShapeD = typename MlaSm100Type::TileShapeD;
|
||||
arguments.problem_shape =
|
||||
cute::make_tuple(TileShapeH{}, static_cast<int>(max_seq_len), TileShapeD{}, static_cast<int>(num_batches));
|
||||
// Assumes device 0 when getting sm_count.
|
||||
arguments.hw_info.sm_count =
|
||||
sm_count <= 0 ? cutlass::KernelHardwareInfo::query_device_multiprocessor_count(/*device_id=*/0) : sm_count;
|
||||
if (sm_count <= 0) {
|
||||
int current_device = 0;
|
||||
cudaGetDevice(¤t_device);
|
||||
arguments.hw_info.sm_count =
|
||||
cutlass::KernelHardwareInfo::query_device_multiprocessor_count(current_device);
|
||||
} else {
|
||||
arguments.hw_info.sm_count = sm_count;
|
||||
}
|
||||
arguments.split_kv = static_cast<int>(num_kv_splits);
|
||||
MlaSm100Type::Fmha::set_split_kv(arguments);
|
||||
|
||||
|
||||
@@ -301,8 +301,9 @@ __global__ void per_token_group_quant_8bit_packed_register_kernel(
|
||||
|
||||
const int sf_k_local = local_group_id % kGroupsPerBlockX;
|
||||
const int row_local = local_group_id / kGroupsPerBlockX;
|
||||
const int sf_k_idx = blockIdx.x * kGroupsPerBlockX + sf_k_local;
|
||||
const int mn_idx = blockIdx.y * kRowsPerBlock + row_local;
|
||||
// Rows on grid.x: mn scales with tokens and can exceed the 65535 grid.y cap.
|
||||
const int sf_k_idx = blockIdx.y * kGroupsPerBlockX + sf_k_local;
|
||||
const int mn_idx = blockIdx.x * kRowsPerBlock + row_local;
|
||||
|
||||
#if (defined(__CUDA_ARCH__) && (__CUDA_ARCH__ >= 900))
|
||||
asm volatile("griddepcontrol.wait;");
|
||||
@@ -496,14 +497,15 @@ void per_token_group_quant_8bit_packed(const torch::stable::Tensor& input,
|
||||
" is not a multiple of 4.");
|
||||
const int kx = GetGroupsPerBlockX(padded_groups_per_row);
|
||||
const int ry = 16 / kx;
|
||||
const int64_t blocks_x = padded_groups_per_row / kx;
|
||||
const int64_t blocks_y = (tma_aligned_mn + ry - 1) / ry;
|
||||
const int64_t row_blocks = (tma_aligned_mn + ry - 1) / ry;
|
||||
const int64_t sf_k_blocks = padded_groups_per_row / kx;
|
||||
const int num_threads = (kx * ry) * THREADS_PER_GROUP;
|
||||
// CUDA caps grid.x and grid.y at 2^31 - 1; guard against pathological inputs.
|
||||
STD_TORCH_CHECK(blocks_x <= static_cast<int64_t>(INT32_MAX) &&
|
||||
blocks_y <= static_cast<int64_t>(INT32_MAX),
|
||||
// CUDA caps grid.x at 2^31 - 1 and grid.y at 2^16 - 1 (65535).
|
||||
constexpr int64_t kMaxGridDimYZ = 65535;
|
||||
STD_TORCH_CHECK(row_blocks <= static_cast<int64_t>(INT32_MAX) &&
|
||||
sf_k_blocks <= kMaxGridDimYZ,
|
||||
"per_token_group_quant_8bit_packed grid too large: (",
|
||||
blocks_x, ", ", blocks_y, ").");
|
||||
row_blocks, ", ", sf_k_blocks, ").");
|
||||
|
||||
auto dst_type = output_q.scalar_type();
|
||||
|
||||
@@ -513,8 +515,8 @@ void per_token_group_quant_8bit_packed(const torch::stable::Tensor& input,
|
||||
#define LAUNCH_REG_KERNEL_INST(T, DST_DTYPE, KX, RY) \
|
||||
do { \
|
||||
cudaLaunchConfig_t config = {}; \
|
||||
config.gridDim = dim3(static_cast<unsigned int>(blocks_x), \
|
||||
static_cast<unsigned int>(blocks_y)); \
|
||||
config.gridDim = dim3(static_cast<unsigned int>(row_blocks), \
|
||||
static_cast<unsigned int>(sf_k_blocks)); \
|
||||
config.blockDim = dim3(num_threads); \
|
||||
config.dynamicSmemBytes = 0; \
|
||||
config.stream = stream; \
|
||||
@@ -539,8 +541,8 @@ void per_token_group_quant_8bit_packed(const torch::stable::Tensor& input,
|
||||
#else
|
||||
#define LAUNCH_REG_KERNEL_INST(T, DST_DTYPE, KX, RY) \
|
||||
do { \
|
||||
dim3 grid(static_cast<unsigned int>(blocks_x), \
|
||||
static_cast<unsigned int>(blocks_y)); \
|
||||
dim3 grid(static_cast<unsigned int>(row_blocks), \
|
||||
static_cast<unsigned int>(sf_k_blocks)); \
|
||||
dim3 block(num_threads); \
|
||||
per_token_group_quant_8bit_packed_register_kernel<T, DST_DTYPE, 128, KX, \
|
||||
RY> \
|
||||
|
||||
@@ -649,3 +649,196 @@ def test_cloud_storage_tokenizer_skips_get_model_path(monkeypatch):
|
||||
args = EngineArgs(model="s3://bucket/model", tokenizer="s3://bucket/tokenizer")
|
||||
assert args.model == "s3://bucket/model"
|
||||
assert args.tokenizer == "s3://bucket/tokenizer"
|
||||
|
||||
|
||||
class TestDeviceIds:
|
||||
def test_device_ids_with_cvd_out_of_range(self, monkeypatch):
|
||||
"""--device-ids index beyond the CVD set raises ValueError."""
|
||||
from vllm.platforms import current_platform
|
||||
|
||||
key = current_platform.device_control_env_var
|
||||
monkeypatch.setenv(key, "4,5")
|
||||
args = EngineArgs(model="m", device_ids=[0, 2])
|
||||
with pytest.raises(ValueError, match="out of range"):
|
||||
args._resolve_device_ids()
|
||||
|
||||
def test_device_ids_with_cvd_resolve_to_physical_ids(self, monkeypatch):
|
||||
"""--device-ids are CVD-local indices resolved to physical ids."""
|
||||
from vllm.platforms import current_platform
|
||||
|
||||
key = current_platform.device_control_env_var
|
||||
monkeypatch.setenv(key, "4,5")
|
||||
args = EngineArgs(model="m", device_ids=[0, 1])
|
||||
assert args._resolve_device_ids() == [4, 5]
|
||||
|
||||
def test_device_ids_with_uuid_cvd_resolve_to_physical_ids(self, monkeypatch):
|
||||
"""--device-ids support UUID CVD values resolved by the platform."""
|
||||
from vllm.platforms import current_platform
|
||||
|
||||
key = current_platform.device_control_env_var
|
||||
monkeypatch.setenv(key, "GPU-abcd1234,GPU-ef567890")
|
||||
monkeypatch.setattr(
|
||||
type(current_platform),
|
||||
"device_control_id_to_physical_device_id",
|
||||
classmethod(
|
||||
lambda cls, device_id: {"GPU-abcd1234": 4, "GPU-ef567890": 5}[device_id]
|
||||
),
|
||||
)
|
||||
|
||||
args = EngineArgs(model="m", device_ids=[0, 1])
|
||||
assert args._resolve_device_ids() == [4, 5]
|
||||
|
||||
def test_device_ids_with_uuid_args_resolve_to_physical_ids(self, monkeypatch):
|
||||
"""UUID --device-ids are resolved to physical IDs immediately."""
|
||||
from vllm.platforms import current_platform
|
||||
|
||||
monkeypatch.setattr(
|
||||
type(current_platform),
|
||||
"device_control_id_to_physical_device_id",
|
||||
classmethod(lambda cls, device_id: {"GPU-abcd1234": 4}[device_id]),
|
||||
)
|
||||
|
||||
args = EngineArgs(model="m", device_ids=["GPU-abcd1234"])
|
||||
assert args._resolve_device_ids() == [4]
|
||||
|
||||
def test_device_ids_reject_mixed_integer_and_uuid_args(self):
|
||||
"""--device-ids must not mix CVD indices and UUIDs."""
|
||||
args = EngineArgs(model="m", device_ids=[0, "GPU-abcd1234"])
|
||||
with pytest.raises(ValueError, match="must not mix"):
|
||||
args._resolve_device_ids()
|
||||
|
||||
def test_no_device_ids(self):
|
||||
"""No --device-ids returns None."""
|
||||
args = EngineArgs(model="m")
|
||||
assert args._resolve_device_ids() is None
|
||||
|
||||
def test_cli_parsing(self):
|
||||
"""--device-ids parses comma-separated string from CLI."""
|
||||
parser = FlexibleArgumentParser()
|
||||
EngineArgs.add_cli_args(parser)
|
||||
parsed = parser.parse_args(["--model", "m", "--device-ids", "0,2,4"])
|
||||
assert parsed.device_ids == [0, 2, 4]
|
||||
|
||||
def test_cli_parsing_uuid(self):
|
||||
"""--device-ids parses comma-separated UUID strings from CLI."""
|
||||
parser = FlexibleArgumentParser()
|
||||
EngineArgs.add_cli_args(parser)
|
||||
parsed = parser.parse_args(
|
||||
["--model", "m", "--device-ids", "GPU-abcd1234,GPU-ef567890"]
|
||||
)
|
||||
assert parsed.device_ids == ["GPU-abcd1234", "GPU-ef567890"]
|
||||
|
||||
def test_assigned_physical_gpu_ids_are_physical_with_cvd(self, monkeypatch):
|
||||
"""assigned_physical_gpu_ids are already physical and not composed with CVD."""
|
||||
import vllm.platforms.interface as platform_interface
|
||||
from vllm.platforms import current_platform
|
||||
|
||||
monkeypatch.setattr(platform_interface, "_assigned_physical_gpu_ids", [4, 5])
|
||||
monkeypatch.setenv(current_platform.device_control_env_var, "4,5")
|
||||
|
||||
assert current_platform.device_id_to_physical_device_id(0) == 4
|
||||
assert current_platform.device_id_to_physical_device_id(1) == 5
|
||||
assert current_platform.logical_device_id_to_visible_device_id(0) == 0
|
||||
assert current_platform.logical_device_id_to_visible_device_id(1) == 1
|
||||
|
||||
def test_assigned_physical_gpu_ids_map_to_visible_uuid_cvd(self, monkeypatch):
|
||||
"""Physical IDs map back to visible ordinals when CVD uses UUIDs."""
|
||||
import vllm.platforms.interface as platform_interface
|
||||
from vllm.platforms import current_platform
|
||||
|
||||
monkeypatch.setattr(platform_interface, "_assigned_physical_gpu_ids", [5])
|
||||
monkeypatch.setenv(
|
||||
current_platform.device_control_env_var,
|
||||
"GPU-abcd1234,GPU-ef567890",
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
type(current_platform),
|
||||
"device_control_id_to_physical_device_id",
|
||||
classmethod(
|
||||
lambda cls, device_id: {"GPU-abcd1234": 4, "GPU-ef567890": 5}[device_id]
|
||||
),
|
||||
)
|
||||
|
||||
assert current_platform.logical_device_id_to_visible_device_id(0) == 1
|
||||
|
||||
def test_device_ids_reject_duplicates(self):
|
||||
"""--device-ids must not contain duplicate entries."""
|
||||
args = EngineArgs(model="m", device_ids=[2, 2])
|
||||
with pytest.raises(ValueError, match="duplicates"):
|
||||
args._resolve_device_ids()
|
||||
|
||||
def test_cli_parsing_strips_whitespace(self):
|
||||
"""--device-ids tolerates whitespace around commas."""
|
||||
parser = FlexibleArgumentParser()
|
||||
EngineArgs.add_cli_args(parser)
|
||||
parsed = parser.parse_args(["--model", "m", "--device-ids", "0, 2, 4"])
|
||||
assert parsed.device_ids == [0, 2, 4]
|
||||
|
||||
def test_visible_ordinal_to_physical_ignores_assigned_ids(self, monkeypatch):
|
||||
"""visible_device_id_to_physical_device_id maps torch device ordinals,
|
||||
independent of the logical-to-physical mapping.
|
||||
|
||||
Regression test: CustomAllreduce passes device.index (a visible
|
||||
ordinal) and must not index into assigned_physical_gpu_ids, which
|
||||
raised IndexError for non-identity --device-ids like [2, 3].
|
||||
"""
|
||||
import vllm.platforms.interface as platform_interface
|
||||
from vllm.platforms import current_platform
|
||||
|
||||
monkeypatch.setattr(platform_interface, "_assigned_physical_gpu_ids", [2, 3])
|
||||
monkeypatch.delenv(current_platform.device_control_env_var, raising=False)
|
||||
|
||||
# CVD unset: visible ordinal == physical ID, even beyond the
|
||||
# assigned list's length.
|
||||
assert current_platform.visible_device_id_to_physical_device_id(2) == 2
|
||||
assert current_platform.visible_device_id_to_physical_device_id(3) == 3
|
||||
|
||||
monkeypatch.setenv(current_platform.device_control_env_var, "4,5")
|
||||
assert current_platform.visible_device_id_to_physical_device_id(1) == 5
|
||||
with pytest.raises(IndexError, match="out of range"):
|
||||
current_platform.visible_device_id_to_physical_device_id(2)
|
||||
|
||||
|
||||
class TestDpDeviceIdSharding:
|
||||
def test_dp_supervisor_device_ids_stay_env_relative(self):
|
||||
"""Regression test: the DP supervisor must pass env-relative indices,
|
||||
not physical IDs, because each child re-resolves --device-ids
|
||||
against its inherited device-control env var."""
|
||||
import argparse
|
||||
|
||||
from vllm.entrypoints.openai.dp_supervisor import _build_device_ids
|
||||
|
||||
args = argparse.Namespace(
|
||||
tensor_parallel_size=2, pipeline_parallel_size=1, device_ids=None
|
||||
)
|
||||
assert _build_device_ids(args, local_rank=0) == [0, 1]
|
||||
assert _build_device_ids(args, local_rank=1) == [2, 3]
|
||||
|
||||
def test_dp_supervisor_shards_user_device_ids(self):
|
||||
"""User-provided --device-ids are sharded across DP children."""
|
||||
import argparse
|
||||
|
||||
from vllm.entrypoints.openai.dp_supervisor import _build_device_ids
|
||||
|
||||
args = argparse.Namespace(
|
||||
tensor_parallel_size=2, pipeline_parallel_size=1, device_ids=[4, 5, 6, 7]
|
||||
)
|
||||
assert _build_device_ids(args, local_rank=0) == [4, 5]
|
||||
assert _build_device_ids(args, local_rank=1) == [6, 7]
|
||||
with pytest.raises(ValueError, match="needs devices"):
|
||||
_build_device_ids(args, local_rank=2)
|
||||
|
||||
def test_dp_rank_shards_user_assigned_gpu_ids(self):
|
||||
"""get_physical_gpu_ids_for_local_dp_rank slices the user-provided
|
||||
--device-ids list instead of recomputing from the env var."""
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.v1.engine.utils import get_physical_gpu_ids_for_local_dp_rank
|
||||
|
||||
evar = current_platform.device_control_env_var
|
||||
assert get_physical_gpu_ids_for_local_dp_rank(
|
||||
evar, local_dp_rank=1, world_size=2, user_assigned_gpu_ids=[4, 5, 6, 7]
|
||||
) == [6, 7]
|
||||
with pytest.raises(ValueError, match="needs devices"):
|
||||
get_physical_gpu_ids_for_local_dp_rank(
|
||||
evar, local_dp_rank=2, world_size=2, user_assigned_gpu_ids=[4, 5, 6, 7]
|
||||
)
|
||||
|
||||
@@ -8,6 +8,8 @@ AnthropicServingMessages._convert_anthropic_to_openai_request().
|
||||
Also covers extended-thinking edge cases such as ``redacted_thinking``
|
||||
blocks echoed back by Anthropic clients, and streaming conversion in
|
||||
``message_stream_converter``.
|
||||
|
||||
Also covers cache usage computation in ``_build_anthropic_usage``.
|
||||
"""
|
||||
|
||||
import json
|
||||
@@ -18,7 +20,11 @@ import pytest
|
||||
from vllm.entrypoints.anthropic.protocol import (
|
||||
AnthropicMessagesRequest,
|
||||
)
|
||||
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
|
||||
from vllm.entrypoints.anthropic.serving import (
|
||||
AnthropicServingMessages,
|
||||
_build_anthropic_usage,
|
||||
_get_cached_tokens,
|
||||
)
|
||||
from vllm.entrypoints.openai.chat_completion.protocol import (
|
||||
ChatCompletionResponseStreamChoice,
|
||||
ChatCompletionStreamResponse,
|
||||
@@ -27,6 +33,7 @@ from vllm.entrypoints.openai.engine.protocol import (
|
||||
DeltaFunctionCall,
|
||||
DeltaMessage,
|
||||
DeltaToolCall,
|
||||
PromptTokenUsageInfo,
|
||||
UsageInfo,
|
||||
)
|
||||
|
||||
@@ -653,6 +660,108 @@ class TestThinkingBlockConversion:
|
||||
assert asst.get("content") == "Hi!"
|
||||
|
||||
|
||||
# ======================================================================
|
||||
# Cache usage computation
|
||||
# ======================================================================
|
||||
|
||||
|
||||
class TestGetCachedTokens:
|
||||
"""Tests for _get_cached_tokens helper."""
|
||||
|
||||
def test_none_usage(self):
|
||||
assert _get_cached_tokens(None) is None
|
||||
|
||||
def test_no_prompt_tokens_details(self):
|
||||
usage = UsageInfo(prompt_tokens=100, completion_tokens=10)
|
||||
assert _get_cached_tokens(usage) is None
|
||||
|
||||
def test_cached_tokens_present(self):
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=80),
|
||||
)
|
||||
assert _get_cached_tokens(usage) == 80
|
||||
|
||||
def test_cached_tokens_zero(self):
|
||||
"""Zero cached tokens should return 0, not None."""
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=0),
|
||||
)
|
||||
assert _get_cached_tokens(usage) == 0
|
||||
|
||||
def test_cached_tokens_none_in_details(self):
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=None),
|
||||
)
|
||||
assert _get_cached_tokens(usage) is None
|
||||
|
||||
|
||||
class TestBuildAnthropicUsage:
|
||||
"""Tests for _build_anthropic_usage helper.
|
||||
|
||||
Anthropic defines: total_input = input_tokens + cache_read + cache_creation
|
||||
vLLM's prompt_tokens is the total.
|
||||
"""
|
||||
|
||||
def test_no_cache_info(self):
|
||||
"""When cache info is unavailable, return raw prompt_tokens."""
|
||||
result = _build_anthropic_usage(100, 10, None)
|
||||
assert result.input_tokens == 100
|
||||
assert result.output_tokens == 10
|
||||
assert result.cache_read_input_tokens is None
|
||||
assert result.cache_creation_input_tokens is None
|
||||
|
||||
def test_cache_hit(self):
|
||||
"""When cache is hit, input_tokens excludes cached tokens."""
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=80),
|
||||
)
|
||||
result = _build_anthropic_usage(100, 10, usage)
|
||||
assert result.input_tokens == 20 # 100 - 80
|
||||
assert result.output_tokens == 10
|
||||
assert result.cache_read_input_tokens == 80
|
||||
assert result.cache_creation_input_tokens == 0
|
||||
|
||||
def test_zero_cached_tokens(self):
|
||||
"""Zero cached tokens should still set cache_creation to 0."""
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=0),
|
||||
)
|
||||
result = _build_anthropic_usage(100, 10, usage)
|
||||
assert result.input_tokens == 100 # 100 - 0
|
||||
assert result.cache_read_input_tokens == 0
|
||||
assert result.cache_creation_input_tokens == 0
|
||||
|
||||
def test_all_tokens_cached(self):
|
||||
"""When all tokens are cached, input_tokens should be 0."""
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=100),
|
||||
)
|
||||
result = _build_anthropic_usage(100, 10, usage)
|
||||
assert result.input_tokens == 0
|
||||
assert result.cache_read_input_tokens == 100
|
||||
assert result.cache_creation_input_tokens == 0
|
||||
|
||||
def test_no_prompt_tokens_details(self):
|
||||
"""UsageInfo without prompt_tokens_details returns no cache info."""
|
||||
usage = UsageInfo(prompt_tokens=100, completion_tokens=10)
|
||||
result = _build_anthropic_usage(100, 10, usage)
|
||||
assert result.input_tokens == 100
|
||||
assert result.cache_read_input_tokens is None
|
||||
assert result.cache_creation_input_tokens is None
|
||||
|
||||
|
||||
class TestInlineSystemMessageInMessagesArray:
|
||||
"""Verify that ``role: system`` messages embedded inside the ``messages``
|
||||
array are preserved in their original position.
|
||||
@@ -1098,6 +1207,135 @@ class TestMessageStartIncludesTypeAndRole:
|
||||
assert message["role"] == "assistant"
|
||||
|
||||
|
||||
class TestStreamingCacheUsageSemantics:
|
||||
"""Locks in the documented streaming behavior of cache usage fields.
|
||||
|
||||
vLLM's OpenAI chat completion streaming only attaches
|
||||
``prompt_tokens_details`` to the terminal usage chunk. The Anthropic layer
|
||||
mirrors that contract: cache fields are omitted on ``message_start`` (key
|
||||
absence signals "unknown") and populated on ``message_delta`` (the final
|
||||
cumulative count). This is intentionally consistent with vLLM's OpenAI
|
||||
behavior, even though Anthropic's upstream API populates cache fields on
|
||||
``message_start``; closing that gap requires plumbing cache info into the
|
||||
first chunk at the OpenAI layer, which is out of scope here.
|
||||
"""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_streaming_cache_fields_absent_then_populated(self):
|
||||
"""First chunk lacks prompt_tokens_details (vLLM contract);
|
||||
message_start omits cache fields. The final chunk carries
|
||||
prompt_tokens_details, so message_delta carries resolved values."""
|
||||
|
||||
async def sse_input():
|
||||
yield _make_stream_chunk(
|
||||
delta=DeltaMessage(role="assistant", content="hi"),
|
||||
usage=UsageInfo(prompt_tokens=100, total_tokens=100),
|
||||
)
|
||||
yield _make_stream_chunk(finish_reason="stop")
|
||||
yield _make_stream_chunk(
|
||||
choices=[],
|
||||
usage=UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=5,
|
||||
total_tokens=105,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=80),
|
||||
),
|
||||
)
|
||||
yield "data: [DONE]"
|
||||
|
||||
converter = _make_stream_converter()
|
||||
output = []
|
||||
async for event in converter.message_stream_converter(sse_input()):
|
||||
output.append(event)
|
||||
events = _parse_sse_events(output)
|
||||
|
||||
# message_start: cache fields unknown → omitted from JSON entirely.
|
||||
start_usage = events[0][1]["message"]["usage"]
|
||||
assert events[0][0] == "message_start"
|
||||
assert start_usage["input_tokens"] == 100
|
||||
assert "cache_read_input_tokens" not in start_usage
|
||||
assert "cache_creation_input_tokens" not in start_usage
|
||||
|
||||
# message_delta: authoritative usage with cache fields populated.
|
||||
delta_usage = next(
|
||||
data["usage"] for ev, data in events if ev == "message_delta"
|
||||
)
|
||||
assert delta_usage["input_tokens"] == 20 # 100 - 80
|
||||
assert delta_usage["cache_read_input_tokens"] == 80
|
||||
assert delta_usage["cache_creation_input_tokens"] == 0
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_streaming_no_cache_hit(self):
|
||||
"""When the final chunk reports cached_tokens=0, message_delta carries
|
||||
cache fields = 0 (cache miss); message_start still omits them."""
|
||||
|
||||
async def sse_input():
|
||||
yield _make_stream_chunk(
|
||||
delta=DeltaMessage(role="assistant"),
|
||||
usage=UsageInfo(prompt_tokens=50, total_tokens=50),
|
||||
)
|
||||
yield _make_stream_chunk(finish_reason="stop")
|
||||
yield _make_stream_chunk(
|
||||
choices=[],
|
||||
usage=UsageInfo(
|
||||
prompt_tokens=50,
|
||||
completion_tokens=5,
|
||||
total_tokens=55,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=0),
|
||||
),
|
||||
)
|
||||
yield "data: [DONE]"
|
||||
|
||||
converter = _make_stream_converter()
|
||||
output = []
|
||||
async for event in converter.message_stream_converter(sse_input()):
|
||||
output.append(event)
|
||||
events = _parse_sse_events(output)
|
||||
|
||||
start_usage = events[0][1]["message"]["usage"]
|
||||
delta_usage = next(
|
||||
data["usage"] for ev, data in events if ev == "message_delta"
|
||||
)
|
||||
assert start_usage["input_tokens"] == 50
|
||||
assert "cache_read_input_tokens" not in start_usage
|
||||
assert "cache_creation_input_tokens" not in start_usage
|
||||
assert delta_usage["input_tokens"] == 50 # 50 - 0
|
||||
assert delta_usage["cache_read_input_tokens"] == 0
|
||||
assert delta_usage["cache_creation_input_tokens"] == 0
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_streaming_no_prompt_tokens_details_at_all(self):
|
||||
"""If --enable-prompt-tokens-details is off, no chunk carries cache
|
||||
info; both message_start and message_delta omit cache fields."""
|
||||
|
||||
async def sse_input():
|
||||
yield _make_stream_chunk(
|
||||
delta=DeltaMessage(role="assistant"),
|
||||
usage=UsageInfo(prompt_tokens=30, total_tokens=30),
|
||||
)
|
||||
yield _make_stream_chunk(finish_reason="stop")
|
||||
yield _make_stream_chunk(
|
||||
choices=[],
|
||||
usage=UsageInfo(prompt_tokens=30, completion_tokens=2, total_tokens=32),
|
||||
)
|
||||
yield "data: [DONE]"
|
||||
|
||||
converter = _make_stream_converter()
|
||||
output = []
|
||||
async for event in converter.message_stream_converter(sse_input()):
|
||||
output.append(event)
|
||||
events = _parse_sse_events(output)
|
||||
|
||||
start_usage = events[0][1]["message"]["usage"]
|
||||
delta_usage = next(
|
||||
data["usage"] for ev, data in events if ev == "message_delta"
|
||||
)
|
||||
assert "cache_read_input_tokens" not in start_usage
|
||||
assert "cache_creation_input_tokens" not in start_usage
|
||||
assert "cache_read_input_tokens" not in delta_usage
|
||||
assert "cache_creation_input_tokens" not in delta_usage
|
||||
|
||||
|
||||
# ======================================================================
|
||||
# Auto-detection of system-first template requirement
|
||||
# ======================================================================
|
||||
|
||||
@@ -364,7 +364,7 @@ class MockVLLMServer:
|
||||
await self._serve_task
|
||||
|
||||
|
||||
def launch_mock_vllm(child_args: argparse.Namespace, env_updates: dict[str, str]):
|
||||
def launch_mock_vllm(child_args: argparse.Namespace):
|
||||
logger.info("Launching mock vLLM on port %s", child_args.port)
|
||||
mock_vllm = MockVLLMServer(
|
||||
port=child_args.port,
|
||||
@@ -375,7 +375,7 @@ def launch_mock_vllm(child_args: argparse.Namespace, env_updates: dict[str, str]
|
||||
|
||||
|
||||
def launch_mock_vllm_with_drain(
|
||||
child_args: argparse.Namespace, env_updates: dict[str, str]
|
||||
child_args: argparse.Namespace,
|
||||
):
|
||||
logger.info("Launching mock vLLM with 15s drain on port %s", child_args.port)
|
||||
mock_vllm = MockVLLMServer(
|
||||
|
||||
@@ -8,6 +8,7 @@ import pytest
|
||||
import pytest_asyncio
|
||||
|
||||
from tests.utils import RemoteLaunchRenderServer
|
||||
from vllm.tokenizers import get_tokenizer
|
||||
|
||||
MODEL_NAME = "hmellor/tiny-random-LlamaForCausalLM"
|
||||
|
||||
@@ -486,3 +487,438 @@ async def test_derender_completion_kv_transfer_params_passthrough(client):
|
||||
)
|
||||
assert response.status_code == 200
|
||||
assert response.json()["kv_transfer_params"] == kv
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# E2E: render -> derender roundtrip with parser (reasoning + tool calls)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
PARSER_MODEL = "deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B"
|
||||
|
||||
_E2E_TOOLS = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"description": "Get weather for a city",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"city": {"type": "string"}},
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def parser_server():
|
||||
args = [
|
||||
"--enable-auto-tool-choice",
|
||||
"--tool-call-parser",
|
||||
"hermes",
|
||||
"--reasoning-parser",
|
||||
"deepseek_r1",
|
||||
]
|
||||
with RemoteLaunchRenderServer(PARSER_MODEL, args) as remote_server:
|
||||
yield remote_server
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def parser_client(parser_server):
|
||||
async with httpx.AsyncClient(
|
||||
base_url=parser_server.url_for(""), timeout=60.0
|
||||
) as http_client:
|
||||
yield http_client
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def parser_tokenizer():
|
||||
return get_tokenizer(PARSER_MODEL)
|
||||
|
||||
|
||||
def _encode(tokenizer, text: str) -> list[int]:
|
||||
return tokenizer.encode(text, add_special_tokens=False)
|
||||
|
||||
|
||||
def _decoded(tokenizer, token_ids: list[int]) -> str:
|
||||
return tokenizer.decode(token_ids, skip_special_tokens=True)
|
||||
|
||||
|
||||
def _require_markers_survive(tokenizer, text: str, *markers: str) -> list[int]:
|
||||
"""Encode text and skip the test if any marker is lost in roundtrip."""
|
||||
ids = _encode(tokenizer, text)
|
||||
decoded = tokenizer.decode(ids, skip_special_tokens=False)
|
||||
for m in markers:
|
||||
if m not in decoded:
|
||||
pytest.skip(f"Marker {m!r} lost in encode->decode roundtrip")
|
||||
return ids
|
||||
|
||||
|
||||
async def _e2e_render_chat(
|
||||
client: httpx.AsyncClient,
|
||||
model: str,
|
||||
messages: list[dict],
|
||||
) -> dict:
|
||||
resp = await client.post(
|
||||
"/v1/chat/completions/render",
|
||||
json={"model": model, "messages": messages},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
return resp.json()
|
||||
|
||||
|
||||
def _e2e_generate_response(
|
||||
token_ids: list[int],
|
||||
request_id: str = "chatcmpl-e2e-test",
|
||||
) -> dict:
|
||||
return {
|
||||
"request_id": request_id,
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"token_ids": token_ids,
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_plain_roundtrip(parser_client, parser_tokenizer):
|
||||
"""Plain text without reasoning markers roundtrips correctly."""
|
||||
messages = [{"role": "user", "content": "What is 2+2?"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
answer = "The answer is four."
|
||||
output_ids = _encode(parser_tokenizer, answer)
|
||||
expected = _decoded(parser_tokenizer, output_ids)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
content = resp.json()["choices"][0]["message"]["content"]
|
||||
assert content == expected
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_token_identity(parser_client, parser_tokenizer):
|
||||
"""encode(derender(token_ids)) == token_ids (RL invariant)."""
|
||||
messages = [{"role": "user", "content": "Hi"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
answer = "Hello! How can I help?"
|
||||
output_ids = _encode(parser_tokenizer, answer)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
content = resp.json()["choices"][0]["message"]["content"]
|
||||
re_encoded = _encode(parser_tokenizer, content)
|
||||
assert output_ids == re_encoded
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_non_ascii_roundtrip(parser_client, parser_tokenizer):
|
||||
"""CJK + emoji roundtrip without U+FFFD."""
|
||||
messages = [{"role": "user", "content": "Reply in Chinese"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
answer = "你好世界 😀"
|
||||
output_ids = _encode(parser_tokenizer, answer)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
content = resp.json()["choices"][0]["message"]["content"]
|
||||
assert "�" not in content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_parsed_reasoning(parser_client, parser_tokenizer):
|
||||
"""<think>...</think> splits into reasoning + content."""
|
||||
messages = [{"role": "user", "content": "What is 2+3?"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
reasoning_text = "The user wants 2 plus 3. That is 5."
|
||||
answer_text = "The answer is 5."
|
||||
output_text = f"<think>{reasoning_text}</think>{answer_text}"
|
||||
output_ids = _require_markers_survive(parser_tokenizer, output_text, "</think>")
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
"chat_request": {
|
||||
"model": PARSER_MODEL,
|
||||
"messages": messages,
|
||||
"include_reasoning": True,
|
||||
},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
msg = resp.json()["choices"][0]["message"]
|
||||
assert msg["reasoning"] is not None
|
||||
assert reasoning_text in msg["reasoning"]
|
||||
assert answer_text in msg["content"]
|
||||
assert "<think>" not in msg["content"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_parsed_tool_call(parser_client, parser_tokenizer):
|
||||
"""<tool_call> extracted into tool_calls field."""
|
||||
messages = [{"role": "user", "content": "Weather in Paris?"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
output_text = (
|
||||
"<think>Let me check the weather.</think>"
|
||||
'<tool_call>\n{"name": "get_weather", '
|
||||
'"arguments": {"city": "Paris"}}\n</tool_call>'
|
||||
)
|
||||
output_ids = _require_markers_survive(
|
||||
parser_tokenizer,
|
||||
output_text,
|
||||
"</think>",
|
||||
"<tool_call>",
|
||||
"</tool_call>",
|
||||
)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
"chat_request": {
|
||||
"model": PARSER_MODEL,
|
||||
"messages": messages,
|
||||
"tools": _E2E_TOOLS,
|
||||
"tool_choice": "auto",
|
||||
},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
choice = resp.json()["choices"][0]
|
||||
assert choice["message"]["tool_calls"]
|
||||
assert choice["message"]["tool_calls"][0]["function"]["name"] == "get_weather"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_parsed_reasoning_and_tool_call(parser_client, parser_tokenizer):
|
||||
"""Reasoning + tool call in the same output."""
|
||||
messages = [{"role": "user", "content": "Weather in Paris?"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
reasoning_text = "I should look up the weather."
|
||||
tool_text = (
|
||||
'<tool_call>\n{"name": "get_weather", '
|
||||
'"arguments": {"city": "Paris"}}\n</tool_call>'
|
||||
)
|
||||
output_text = f"<think>{reasoning_text}</think>{tool_text}"
|
||||
output_ids = _require_markers_survive(
|
||||
parser_tokenizer, output_text, "</think>", "<tool_call>"
|
||||
)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
"chat_request": {
|
||||
"model": PARSER_MODEL,
|
||||
"messages": messages,
|
||||
"tools": _E2E_TOOLS,
|
||||
"tool_choice": "auto",
|
||||
"include_reasoning": True,
|
||||
},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
choice = resp.json()["choices"][0]
|
||||
assert choice["message"]["reasoning"] is not None
|
||||
assert reasoning_text in choice["message"]["reasoning"]
|
||||
assert choice["message"]["tool_calls"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_no_chat_request_fallback(parser_client, parser_tokenizer):
|
||||
"""Without chat_request, derender falls back to plain detokenization."""
|
||||
messages = [{"role": "user", "content": "Hello"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
answer = "Hi there!"
|
||||
output_ids = _encode(parser_tokenizer, answer)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
content = resp.json()["choices"][0]["message"]["content"]
|
||||
assert "Hi" in content
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# E2E: HarmonyParser + GPT-OSS
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
HARMONY_MODEL = "openai/gpt-oss-20b"
|
||||
|
||||
|
||||
def _ensure_harmony_vocab():
|
||||
"""Pre-cache the o200k_base BPE file needed by openai-harmony.
|
||||
|
||||
The Rust tiktoken-rs backend downloads from Azure Blob Storage, which
|
||||
may be unreachable in some environments. When the cache is cold we
|
||||
fetch the file ourselves and place it in ``/tmp/tiktoken-rs-cache/``
|
||||
using the SHA-1(URL) filename that tiktoken-rs expects.
|
||||
"""
|
||||
import hashlib
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
url = "https://openaipublic.blob.core.windows.net/encodings/o200k_base.tiktoken"
|
||||
cache_dir = Path("/tmp/tiktoken-rs-cache")
|
||||
cache_key = hashlib.sha1(url.encode()).hexdigest()
|
||||
cache_file = cache_dir / cache_key
|
||||
if not cache_file.exists():
|
||||
cache_dir.mkdir(parents=True, exist_ok=True)
|
||||
urllib.request.urlretrieve(url, cache_file)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def harmony_server():
|
||||
_ensure_harmony_vocab()
|
||||
args = [
|
||||
"--trust-remote-code",
|
||||
"--enable-auto-tool-choice",
|
||||
"--tool-call-parser",
|
||||
"openai",
|
||||
"--reasoning-parser",
|
||||
"openai_gptoss",
|
||||
]
|
||||
with RemoteLaunchRenderServer(HARMONY_MODEL, args) as remote_server:
|
||||
yield remote_server
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def harmony_client(harmony_server):
|
||||
async with httpx.AsyncClient(
|
||||
base_url=harmony_server.url_for(""), timeout=60.0
|
||||
) as http_client:
|
||||
yield http_client
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def harmony_tokenizer():
|
||||
return get_tokenizer(HARMONY_MODEL, trust_remote_code=True)
|
||||
|
||||
|
||||
def _harmony_extract_assistant_ids(
|
||||
tokenizer, assistant_msg: dict, user_content: str = "test"
|
||||
) -> list[int]:
|
||||
"""Extract assistant token IDs via apply_chat_template diff."""
|
||||
prompt = [{"role": "user", "content": user_content}]
|
||||
full = prompt + [assistant_msg]
|
||||
text_prompt = tokenizer.apply_chat_template(
|
||||
prompt, add_generation_prompt=True, tokenize=False
|
||||
)
|
||||
text_full = tokenizer.apply_chat_template(
|
||||
full, add_generation_prompt=False, tokenize=False
|
||||
)
|
||||
prompt_ids = tokenizer.encode(text_prompt)
|
||||
full_ids = tokenizer.encode(text_full)
|
||||
assistant_ids = list(full_ids[len(prompt_ids) :])
|
||||
if not assistant_ids:
|
||||
pytest.skip("Could not extract assistant tokens for Harmony")
|
||||
return assistant_ids
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_harmony_plain_roundtrip(harmony_client, harmony_tokenizer):
|
||||
"""GPT-OSS content-only roundtrip."""
|
||||
messages = [{"role": "user", "content": "What is 2+2?"}]
|
||||
gen_req = await _e2e_render_chat(harmony_client, HARMONY_MODEL, messages)
|
||||
|
||||
assistant_msg = {"role": "assistant", "content": "Four."}
|
||||
output_ids = _harmony_extract_assistant_ids(harmony_tokenizer, assistant_msg)
|
||||
|
||||
resp = await harmony_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": HARMONY_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
"chat_request": {
|
||||
"model": HARMONY_MODEL,
|
||||
"messages": messages,
|
||||
},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
content = resp.json()["choices"][0]["message"]["content"]
|
||||
assert content is not None and len(content) > 0
|
||||
assert "Four" in content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_harmony_reasoning(harmony_client, harmony_tokenizer):
|
||||
"""GPT-OSS reasoning: analysis channel extracted."""
|
||||
messages = [{"role": "user", "content": "Add 2 and 3."}]
|
||||
gen_req = await _e2e_render_chat(harmony_client, HARMONY_MODEL, messages)
|
||||
|
||||
reasoning_text = "The user wants 2 plus 3."
|
||||
answer_text = "The answer is 5."
|
||||
assistant_msg = {
|
||||
"role": "assistant",
|
||||
"thinking": reasoning_text,
|
||||
"content": answer_text,
|
||||
}
|
||||
output_ids = _harmony_extract_assistant_ids(harmony_tokenizer, assistant_msg)
|
||||
|
||||
decoded = harmony_tokenizer.decode(output_ids)
|
||||
if reasoning_text not in decoded:
|
||||
pytest.skip("Harmony template did not render thinking")
|
||||
|
||||
resp = await harmony_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": HARMONY_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
"chat_request": {
|
||||
"model": HARMONY_MODEL,
|
||||
"messages": messages,
|
||||
"include_reasoning": True,
|
||||
},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
msg = resp.json()["choices"][0]["message"]
|
||||
assert msg["reasoning"] is not None
|
||||
assert reasoning_text in msg["reasoning"]
|
||||
assert answer_text in (msg["content"] or "")
|
||||
|
||||
@@ -345,6 +345,63 @@ def test_per_token_group_quant_fp8_packed_zero_fills_padded_output_q(
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
not current_platform.is_cuda_alike(),
|
||||
reason="packed FP8 per-token-group quant kernel requires a CUDA-alike GPU",
|
||||
)
|
||||
def test_per_token_group_quant_fp8_packed_large_mn():
|
||||
"""Regression test for https://github.com/vllm-project/vllm/issues/45099.
|
||||
|
||||
Some background: gridDim.x and gridDim.y have different limits of 2^31 - 1 and
|
||||
2^16 - 1, respectively.
|
||||
Prior code introduced a bug where it incorrectly assumed grid.x and y both have
|
||||
2^31 - 1 limits and mixed them up, which doesn't surface until the kernel is
|
||||
launched with a large mn that exceeds grid.y limit (2^16 - 1).
|
||||
|
||||
This issue doesn't surface often because each forward pass only processes a
|
||||
bounded token batch, not the full context.
|
||||
Quantizing tensors with more rows than that will fail at launch with
|
||||
"CUDA error: invalid argument".
|
||||
This is a differential test that compares fp8 output against Triton output
|
||||
reference when token size sits just above the gridDim.y 2^16 - 1 limit.
|
||||
"""
|
||||
|
||||
device = "cuda"
|
||||
group_size = 128
|
||||
# hidden 2048 -> 2048/128 = 16 groups per row -> kx=16, ry=1: one grid row per mn
|
||||
# row, so any mn > 65535 overflowed grid.y before the fix.
|
||||
num_tokens, hidden_dim = 65537, 2048
|
||||
torch.manual_seed(42)
|
||||
x = torch.randn((num_tokens, hidden_dim), device=device, dtype=torch.bfloat16) * 8
|
||||
|
||||
out_q, out_s_packed = fp8_utils.per_token_group_quant_fp8_packed_for_deepgemm(
|
||||
x,
|
||||
group_size=group_size,
|
||||
use_ue8m0=True,
|
||||
)
|
||||
|
||||
with patch("vllm.platforms.current_platform.is_cuda_alike", return_value=False):
|
||||
ref_q, ref_s = fp8_utils.per_token_group_quant_fp8(
|
||||
x, group_size, use_ue8m0=True
|
||||
)
|
||||
|
||||
assert torch.equal(out_q, ref_q), "Quantized output mismatch"
|
||||
|
||||
# Vectorized packed-scale check; the per-element loop used by the smaller
|
||||
# tests is too slow at this size. groups_per_row is a multiple of 4 here,
|
||||
# so there is no K padding and the packed view lines up.
|
||||
mn = num_tokens
|
||||
groups_per_row = hidden_dim // group_size
|
||||
k_num_packed = (groups_per_row + 3) // 4
|
||||
assert groups_per_row % 4 == 0
|
||||
ref_exponents = (ref_s.reshape(mn, groups_per_row).view(torch.int32) >> 23) & 0xFF
|
||||
exp = ref_exponents.view(mn, k_num_packed, 4)
|
||||
expected = (
|
||||
exp[..., 0] | (exp[..., 1] << 8) | (exp[..., 2] << 16) | (exp[..., 3] << 24)
|
||||
)
|
||||
assert torch.equal(out_s_packed.cpu(), expected.cpu()), "Packed scale mismatch"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("shape", [(32, 128), (64, 256), (16, 512)])
|
||||
@pytest.mark.parametrize("group_size", [64, 128])
|
||||
@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
|
||||
|
||||
@@ -92,3 +92,49 @@ def test_processor_num_frames_timestamp(
|
||||
assert len(video_phs) == 1, (
|
||||
f"Expected exactly 1 video placeholder, got {len(video_phs)}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model_id", [MODEL_ID])
|
||||
@pytest.mark.parametrize("num_videos", [2, 4])
|
||||
def test_processor_multi_video(
|
||||
model_id: str,
|
||||
num_videos: int,
|
||||
) -> None:
|
||||
"""Verify that multi-video processing produces correct placeholders.
|
||||
|
||||
This exercises the token-level replacement path in
|
||||
``_call_hf_processor`` which avoids the quadratic text-level
|
||||
prompt expansion.
|
||||
"""
|
||||
ctx = build_model_context(
|
||||
model_id,
|
||||
limit_mm_per_prompt={"image": 0, "video": num_videos},
|
||||
)
|
||||
processor = MULTIMODAL_REGISTRY.create_processor(ctx.model_config)
|
||||
|
||||
prompt = "<|vision_start|><|video_pad|><|vision_end|>" * num_videos
|
||||
mm_data = {"video": [_build_video_mm_data(num_frames=8)["video"][0]] * num_videos}
|
||||
|
||||
processed = processor(
|
||||
prompt,
|
||||
mm_items=processor.info.parse_mm_data(mm_data),
|
||||
hf_processor_mm_kwargs={"num_frames": 8},
|
||||
)
|
||||
|
||||
token_ids = processed["prompt_token_ids"]
|
||||
assert len(token_ids) > 0
|
||||
|
||||
video_phs = processed["mm_placeholders"].get("video", [])
|
||||
assert len(video_phs) == num_videos, (
|
||||
f"Expected {num_videos} video placeholders, got {len(video_phs)}"
|
||||
)
|
||||
|
||||
# All placeholders should have the same length (same video params)
|
||||
# and must not overlap.
|
||||
lengths = {ph.length for ph in video_phs}
|
||||
assert len(lengths) == 1, f"Placeholder lengths differ: {lengths}"
|
||||
for i in range(1, len(video_phs)):
|
||||
prev_end = video_phs[i - 1].offset + video_phs[i - 1].length
|
||||
assert video_phs[i].offset >= prev_end, (
|
||||
f"Placeholder {i} overlaps with placeholder {i - 1}"
|
||||
)
|
||||
|
||||
@@ -83,6 +83,20 @@ class MiniMaxM3Tokenizer:
|
||||
return "".join(tokens)
|
||||
|
||||
|
||||
class SplitMiniMaxM3Tokenizer(MiniMaxM3Tokenizer):
|
||||
"""Tokenizer that exposes marker vocab entries but encodes them as text."""
|
||||
|
||||
def tokenize(self, text: str) -> list[str]:
|
||||
return list(text)
|
||||
|
||||
|
||||
class RuntimeSplitMiniMaxM3Tokenizer(MiniMaxM3Tokenizer):
|
||||
"""Tokenizer whose runtime output splits markers despite atomic encodes."""
|
||||
|
||||
def encode_runtime(self, text: str) -> list[int]:
|
||||
return [self._add_token(token) for token in list(text)]
|
||||
|
||||
|
||||
def make_parser(
|
||||
chat_template_kwargs: dict[str, str] | None = None,
|
||||
) -> tuple[MiniMaxM3ReasoningParser, MiniMaxM3Tokenizer]:
|
||||
@@ -105,7 +119,8 @@ def run_streaming(
|
||||
reasoning_end_states: list[bool] = []
|
||||
|
||||
for chunk in chunks:
|
||||
delta_token_ids = tokenizer.encode(chunk, add_special_tokens=False)
|
||||
encode_runtime = getattr(tokenizer, "encode_runtime", tokenizer.encode)
|
||||
delta_token_ids = encode_runtime(chunk)
|
||||
current_text = previous_text + chunk
|
||||
current_token_ids = previous_token_ids + delta_token_ids
|
||||
delta = parser.extract_reasoning_streaming(
|
||||
@@ -174,14 +189,14 @@ def test_nonstreaming_drops_leading_end_tag():
|
||||
assert content == "answer"
|
||||
|
||||
|
||||
def test_nonstreaming_non_leading_end_tag_is_content():
|
||||
def test_nonstreaming_end_tag_in_content_state_is_dropped():
|
||||
parser, _ = make_parser()
|
||||
request = ChatCompletionRequest(messages=[], model="test-model")
|
||||
|
||||
reasoning, content = parser.extract_reasoning("XXX</mm:think>YYY", request)
|
||||
|
||||
assert reasoning is None
|
||||
assert content == "XXX</mm:think>YYY"
|
||||
assert content == "XXXYYY"
|
||||
|
||||
|
||||
def test_nonstreaming_enabled_mode_starts_in_reasoning():
|
||||
@@ -246,7 +261,7 @@ def test_streaming_drops_leading_end_tag():
|
||||
assert end_states == [True, True]
|
||||
|
||||
|
||||
def test_streaming_non_leading_end_tag_is_content():
|
||||
def test_streaming_end_tag_in_content_state_is_dropped():
|
||||
parser, tokenizer = make_parser()
|
||||
|
||||
reasoning, content, end_states = run_streaming(
|
||||
@@ -256,7 +271,7 @@ def test_streaming_non_leading_end_tag_is_content():
|
||||
)
|
||||
|
||||
assert reasoning is None
|
||||
assert content == "XXX</mm:think>YYY"
|
||||
assert content == "XXXYYY"
|
||||
assert end_states == [True]
|
||||
|
||||
|
||||
@@ -288,6 +303,110 @@ def test_streaming_plain_content_ends_reasoning_phase():
|
||||
assert end_states == [True, True]
|
||||
|
||||
|
||||
def test_streaming_split_marker_tokens_are_not_returned():
|
||||
tokenizer = RuntimeSplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(tokenizer)
|
||||
|
||||
reasoning, content, end_states = run_streaming(
|
||||
parser,
|
||||
tokenizer,
|
||||
["<mm:think>", "Reasoning", " content", "</mm:think>", "content"],
|
||||
)
|
||||
|
||||
assert reasoning == "Reasoning content"
|
||||
assert content == "content"
|
||||
assert end_states == [False, False, False, True, True]
|
||||
|
||||
|
||||
def test_streaming_split_marker_text_drives_end_state():
|
||||
tokenizer = RuntimeSplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(tokenizer)
|
||||
previous_text = ""
|
||||
previous_token_ids: list[int] = []
|
||||
|
||||
for chunk in ["<mm:think>", "Reasoning", " content", "</mm:think>"]:
|
||||
delta_token_ids = tokenizer.encode_runtime(chunk)
|
||||
current_text = previous_text + chunk
|
||||
current_token_ids = previous_token_ids + delta_token_ids
|
||||
parser.extract_reasoning_streaming(
|
||||
previous_text=previous_text,
|
||||
current_text=current_text,
|
||||
delta_text=chunk,
|
||||
previous_token_ids=previous_token_ids,
|
||||
current_token_ids=current_token_ids,
|
||||
delta_token_ids=delta_token_ids,
|
||||
)
|
||||
previous_text = current_text
|
||||
previous_token_ids = current_token_ids
|
||||
|
||||
assert parser.is_reasoning_end_streaming(previous_token_ids, []) is True
|
||||
|
||||
|
||||
def test_streaming_split_marker_tokens_enabled_mode():
|
||||
tokenizer = RuntimeSplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(
|
||||
tokenizer, chat_template_kwargs={"thinking_mode": "enabled"}
|
||||
)
|
||||
|
||||
reasoning, content, end_states = run_streaming(
|
||||
parser,
|
||||
tokenizer,
|
||||
["Reasoning", " content", "</mm:think>", "content"],
|
||||
)
|
||||
|
||||
assert reasoning == "Reasoning content"
|
||||
assert content == "content"
|
||||
assert end_states == [False, False, True, True]
|
||||
|
||||
|
||||
def test_streaming_split_marker_text_across_deltas():
|
||||
tokenizer = RuntimeSplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(tokenizer)
|
||||
|
||||
reasoning, content, end_states = run_streaming(
|
||||
parser,
|
||||
tokenizer,
|
||||
["<mm:", "think>", "Reasoning", " content", "</mm:", "think>", "content"],
|
||||
)
|
||||
|
||||
assert reasoning == "Reasoning content"
|
||||
assert content == "content"
|
||||
assert end_states == [False, False, False, False, False, True, True]
|
||||
|
||||
|
||||
def test_streaming_split_leading_end_marker_text_across_deltas():
|
||||
tokenizer = RuntimeSplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(tokenizer)
|
||||
|
||||
reasoning, content, end_states = run_streaming(
|
||||
parser,
|
||||
tokenizer,
|
||||
["</mm:", "think>", "content"],
|
||||
)
|
||||
|
||||
assert reasoning is None
|
||||
assert content == "content"
|
||||
assert end_states == [False, True, True]
|
||||
|
||||
|
||||
def test_token_id_helpers_with_split_marker_tokens():
|
||||
tokenizer = SplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(tokenizer)
|
||||
output_ids = tokenizer.encode(
|
||||
"<mm:think>abc</mm:think>def", add_special_tokens=False
|
||||
)
|
||||
open_reasoning_ids = tokenizer.encode("<mm:think>abc", add_special_tokens=False)
|
||||
content_ids = tokenizer.encode("plain", add_special_tokens=False)
|
||||
|
||||
assert parser.is_reasoning_end(output_ids)
|
||||
assert not parser.is_reasoning_end(open_reasoning_ids)
|
||||
assert not parser.is_reasoning_end(content_ids)
|
||||
assert tokenizer.decode(parser.extract_content_ids(output_ids)) == "def"
|
||||
assert parser.extract_content_ids(open_reasoning_ids) == []
|
||||
assert parser.extract_content_ids(content_ids) == content_ids
|
||||
assert parser.count_reasoning_tokens(output_ids) == len(tokenizer.encode("abc"))
|
||||
|
||||
|
||||
def test_token_id_helpers():
|
||||
parser, tokenizer = make_parser()
|
||||
output_ids = tokenizer.encode(
|
||||
|
||||
@@ -1,16 +1,23 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
"""Tests for contiguous KV cache packing in _get_kv_cache_config_deepseek_v4."""
|
||||
"""Tests for contiguous KV cache packing."""
|
||||
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from vllm.v1.core.kv_cache_utils import _get_kv_cache_config_deepseek_v4
|
||||
from vllm import envs
|
||||
from vllm.v1.core.kv_cache_utils import (
|
||||
_get_kv_cache_config_deepseek_v4,
|
||||
get_kv_cache_config_from_groups,
|
||||
)
|
||||
from vllm.v1.kv_cache_interface import (
|
||||
FullAttentionSpec,
|
||||
KVCacheGroupSpec,
|
||||
KVCacheTensor,
|
||||
MLAAttentionSpec,
|
||||
SlidingWindowSpec,
|
||||
UniformTypeKVCacheSpecs,
|
||||
)
|
||||
|
||||
@@ -28,6 +35,25 @@ def _make_mla_spec(page_size: int, block_size: int = 256) -> MLAAttentionSpec:
|
||||
)
|
||||
|
||||
|
||||
def _make_full_spec() -> FullAttentionSpec:
|
||||
return FullAttentionSpec(
|
||||
block_size=16,
|
||||
num_kv_heads=2,
|
||||
head_size=64,
|
||||
dtype=torch.float16,
|
||||
)
|
||||
|
||||
|
||||
def _make_sw_spec() -> SlidingWindowSpec:
|
||||
return SlidingWindowSpec(
|
||||
block_size=16,
|
||||
num_kv_heads=2,
|
||||
head_size=64,
|
||||
dtype=torch.float16,
|
||||
sliding_window=128,
|
||||
)
|
||||
|
||||
|
||||
def _make_groups(n_c4, n_c128, n_swa):
|
||||
PS_C4_MLA = 37440
|
||||
PS_C4_IDX = 8640
|
||||
@@ -130,6 +156,73 @@ class TestInterleavedPacking:
|
||||
for i, v in enumerate(views):
|
||||
assert (v == i + 1).all(), f"View {i} was corrupted"
|
||||
|
||||
def test_hma_attention_groups_keep_default_backing(self, monkeypatch):
|
||||
monkeypatch.setattr(envs, "VLLM_USE_PACKED_HMA_KV_CACHE", False, raising=False)
|
||||
full = _make_full_spec()
|
||||
sw = _make_sw_spec()
|
||||
page_size = full.page_size_bytes
|
||||
groups = [
|
||||
KVCacheGroupSpec(["full.0", "full.1"], full),
|
||||
KVCacheGroupSpec(["sw.0", "sw.2"], sw),
|
||||
KVCacheGroupSpec(["sw.1", "sw.3"], sw),
|
||||
]
|
||||
|
||||
config = get_kv_cache_config_from_groups(
|
||||
_mock_vllm_config(), groups, available_memory=page_size * 2 * 32
|
||||
)
|
||||
|
||||
assert config.num_blocks == 32
|
||||
assert sum(t.size for t in config.kv_cache_tensors) == page_size * 2 * 32
|
||||
assert config.kv_cache_tensors == [
|
||||
KVCacheTensor(size=page_size * 32, shared_by=["full.0", "sw.0", "sw.1"]),
|
||||
KVCacheTensor(size=page_size * 32, shared_by=["full.1", "sw.2", "sw.3"]),
|
||||
]
|
||||
|
||||
def test_hma_attention_groups_use_packed_backing_with_flag(self, monkeypatch):
|
||||
monkeypatch.setattr(envs, "VLLM_USE_PACKED_HMA_KV_CACHE", True, raising=False)
|
||||
full = _make_full_spec()
|
||||
sw = _make_sw_spec()
|
||||
page_size = full.page_size_bytes
|
||||
groups = [
|
||||
KVCacheGroupSpec(["full.0", "full.1"], full),
|
||||
KVCacheGroupSpec(["sw.0", "sw.2"], sw),
|
||||
KVCacheGroupSpec(["sw.1", "sw.3"], sw),
|
||||
]
|
||||
|
||||
config = get_kv_cache_config_from_groups(
|
||||
_mock_vllm_config(), groups, available_memory=page_size * 2 * 32
|
||||
)
|
||||
|
||||
assert config.num_blocks == 32
|
||||
assert {t.size for t in config.kv_cache_tensors} == {page_size * 2 * 32}
|
||||
assert config.kv_cache_tensors == [
|
||||
KVCacheTensor(
|
||||
size=page_size * 2 * 32,
|
||||
shared_by=["full.0", "sw.0", "sw.1"],
|
||||
offset=0,
|
||||
block_stride=page_size * 2,
|
||||
),
|
||||
KVCacheTensor(
|
||||
size=page_size * 2 * 32,
|
||||
shared_by=["full.1", "sw.2", "sw.3"],
|
||||
offset=page_size,
|
||||
block_stride=page_size * 2,
|
||||
),
|
||||
]
|
||||
|
||||
def test_single_group_attention_keeps_unpacked_layout(self):
|
||||
spec = _make_full_spec()
|
||||
groups = [KVCacheGroupSpec(["full.0", "full.1"], spec)]
|
||||
|
||||
config = get_kv_cache_config_from_groups(
|
||||
_mock_vllm_config(), groups, available_memory=spec.page_size_bytes * 2 * 32
|
||||
)
|
||||
|
||||
assert sum(t.size for t in config.kv_cache_tensors) == (
|
||||
spec.page_size_bytes * 2 * 32
|
||||
)
|
||||
assert [t.block_stride for t in config.kv_cache_tensors] == [0, 0]
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
pytest.main([__file__, "-v"])
|
||||
|
||||
@@ -144,6 +144,43 @@ def test_async_scheduling_pp_allows_rescheduling_with_output_placeholders():
|
||||
assert req.request_id in output.num_scheduled_tokens
|
||||
|
||||
|
||||
def test_cached_request_data_resumed_all_token_ids_mrv1_only():
|
||||
"""all_token_ids carries a resumed request's token ids to the connector
|
||||
for the V1 model runner, but is skipped entirely for the V2 model runner.
|
||||
"""
|
||||
from vllm.v1.core.kv_cache_manager import KVCacheBlocks
|
||||
|
||||
scheduler = create_scheduler()
|
||||
(req,) = create_requests(num_requests=1, num_tokens=8)
|
||||
req.append_output_token_ids([101, 102, 103])
|
||||
|
||||
# A resumed request was not scheduled in the previous step.
|
||||
assert req.request_id not in scheduler.prev_step_scheduled_req_ids
|
||||
|
||||
empty_blocks = KVCacheBlocks(blocks=((),))
|
||||
|
||||
def make_cached():
|
||||
return scheduler._make_cached_request_data(
|
||||
running_reqs=[],
|
||||
resumed_reqs=[req],
|
||||
num_scheduled_tokens={req.request_id: 1},
|
||||
spec_decode_tokens={},
|
||||
req_to_new_blocks={req.request_id: empty_blocks},
|
||||
)
|
||||
|
||||
# V1 model runner: the full token id list is propagated.
|
||||
assert not scheduler.use_v2_model_runner
|
||||
cached = make_cached()
|
||||
assert req.request_id in cached.resumed_req_ids
|
||||
assert cached.all_token_ids[req.request_id] == list(req.all_token_ids)
|
||||
|
||||
# V2 model runner: all_token_ids is skipped entirely.
|
||||
scheduler.use_v2_model_runner = True
|
||||
cached = make_cached()
|
||||
assert req.request_id in cached.resumed_req_ids
|
||||
assert cached.all_token_ids == {}
|
||||
|
||||
|
||||
def test_schedule_partial_requests():
|
||||
"""Test scheduling behavior with partial requests.
|
||||
|
||||
|
||||
@@ -4,9 +4,16 @@
|
||||
import pytest
|
||||
|
||||
from vllm import LLM, SamplingParams
|
||||
from vllm.platforms import current_platform
|
||||
|
||||
from ....utils import create_new_process_for_each_test
|
||||
|
||||
if current_platform.is_rocm():
|
||||
pytest.skip(
|
||||
"Cascade attention backends FLASH_ATTN and FLASHINFER are notsupported on ROCm",
|
||||
allow_module_level=True,
|
||||
)
|
||||
|
||||
|
||||
@create_new_process_for_each_test()
|
||||
@pytest.mark.parametrize("attn_backend", ["FLASH_ATTN", "FLASHINFER"])
|
||||
|
||||
@@ -7,7 +7,10 @@ from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.coordinator imp
|
||||
ExternalCachedBlockPool,
|
||||
MooncakeStoreCoordinator,
|
||||
)
|
||||
from vllm.v1.core.kv_cache_utils import BlockHash, BlockHashListWithBlockSize
|
||||
from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.data import (
|
||||
chunk_hashes_for_block_size,
|
||||
)
|
||||
from vllm.v1.core.kv_cache_utils import BlockHash
|
||||
from vllm.v1.kv_cache_interface import (
|
||||
FullAttentionSpec,
|
||||
KVCacheGroupSpec,
|
||||
@@ -182,7 +185,7 @@ def test_coordinator_group_block_size_double_hash():
|
||||
]
|
||||
coord = _make_coord(groups, hash_block_size=16)
|
||||
hs = _hashes(4)
|
||||
big_hashes = list(BlockHashListWithBlockSize(hs, 16, 32))
|
||||
big_hashes = list(chunk_hashes_for_block_size(hs, 16, 32))
|
||||
exists = {(0, bytes(h)) for h in hs}
|
||||
exists |= {(1, bytes(bh)) for bh in big_hashes}
|
||||
cmap = ExternalCachedBlockPool(exists)
|
||||
|
||||
@@ -323,8 +323,8 @@ def test_recv_skips_swa_blocks_before_window():
|
||||
|
||||
def test_chunked_token_database_hash_block_size_smaller_than_block_size():
|
||||
"""DSv4-style: hash_block_size=4, group block_size=16 — process_tokens
|
||||
must merge every 4 fine hashes into one chunk hash via
|
||||
BlockHashListWithBlockSize."""
|
||||
keys each 16-token chunk by its last fine hash, keeping the Mooncake key
|
||||
at one digest instead of concatenating all 4 fine hashes."""
|
||||
md = KeyMetadata("m", 0, 0, 0, 0, group_id=3)
|
||||
db = ChunkedTokenDatabase(md, block_size=16, hash_block_size=4)
|
||||
db.set_kv_caches_base_addr([0])
|
||||
@@ -335,8 +335,7 @@ def test_chunked_token_database_hash_block_size_smaller_than_block_size():
|
||||
assert len(out) == 2
|
||||
assert out[0][0] == 0 and out[0][1] == 16
|
||||
assert out[1][0] == 16 and out[1][1] == 32
|
||||
# Each chunk's hash is the concatenation of 4 fine hashes.
|
||||
expected0 = b"".join(fine_hashes[0:4]).hex()
|
||||
expected1 = b"".join(fine_hashes[4:8]).hex()
|
||||
assert out[0][2].chunk_hash == expected0
|
||||
assert out[1][2].chunk_hash == expected1
|
||||
# Each chunk's hash is its last (4th) fine hash, which already chains the
|
||||
# prior three.
|
||||
assert out[0][2].chunk_hash == fine_hashes[3].hex()
|
||||
assert out[1][2].chunk_hash == fine_hashes[7].hex()
|
||||
|
||||
@@ -23,6 +23,7 @@ from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store import (
|
||||
worker as mooncake_store_worker,
|
||||
)
|
||||
from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.data import (
|
||||
BlobBlockHashes,
|
||||
ChunkedTokenDatabase,
|
||||
KeyMetadata,
|
||||
LoadSpec,
|
||||
@@ -32,6 +33,7 @@ from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.data import (
|
||||
from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.metrics import (
|
||||
MooncakeStoreConnectorStats,
|
||||
)
|
||||
from vllm.v1.core.kv_cache_utils import BlockHash
|
||||
|
||||
|
||||
def _default_send_coord() -> mooncake_store_worker.MooncakeStoreCoordinator:
|
||||
@@ -1179,9 +1181,9 @@ def test_store_sending_thread_kv_events_use_group_chunk_metadata():
|
||||
assert full_event.group_idx == 0
|
||||
assert full_event.block_size == 32
|
||||
assert full_event.token_ids == list(range(32))
|
||||
assert full_event.block_hashes == [
|
||||
maybe_convert_block_hash(BlockHash(b"".join(hs)))
|
||||
]
|
||||
# block_size=32 over hash_block_size=8 (scale 4): the chunk is keyed by its
|
||||
# last sub-hash, not the concatenation of all four.
|
||||
assert full_event.block_hashes == [maybe_convert_block_hash(BlockHash(hs[3]))]
|
||||
|
||||
assert swa_event.group_idx == 1
|
||||
assert swa_event.block_size == 8
|
||||
@@ -1749,3 +1751,33 @@ def test_store_worker_close_swallows_store_errors():
|
||||
worker.close()
|
||||
|
||||
assert worker.store is None
|
||||
|
||||
|
||||
def test_blob_block_hashes_wire_roundtrip():
|
||||
"""The lookup wire format sends a ``hash_len`` frame plus the raw hashes
|
||||
concatenated back-to-back; the server rebuilds them through a zero-copy
|
||||
``BlobBlockHashes`` view over the frame buffer."""
|
||||
hashes = [BlockHash(bytes([i]) * 16) for i in range(5)]
|
||||
hash_len = len(hashes[0])
|
||||
|
||||
# Client side (LookupKeyClient._lookup): flat payload frame.
|
||||
blob = b"".join(hashes)
|
||||
|
||||
# Server side (LookupKeyServer): view over the frame buffer (a memoryview),
|
||||
# never materializing the full hash list upfront.
|
||||
view = BlobBlockHashes(memoryview(blob), hash_len)
|
||||
|
||||
assert len(view) == 5
|
||||
assert list(view) == hashes # default Sequence iter terminates via IndexError
|
||||
assert [bytes(h) for h in view] == hashes
|
||||
assert bytes(view[-1]) == hashes[-1]
|
||||
assert [bytes(h) for h in view[1:3]] == hashes[1:3]
|
||||
with pytest.raises(IndexError):
|
||||
_ = view[5]
|
||||
|
||||
|
||||
def test_blob_block_hashes_empty():
|
||||
"""Empty lookups send hash_len=0 and an empty payload."""
|
||||
view = BlobBlockHashes(memoryview(b""), 0)
|
||||
assert len(view) == 0
|
||||
assert list(view) == []
|
||||
|
||||
@@ -14,12 +14,13 @@ from vllm.v1.kv_offload.base import (
|
||||
ReqContext,
|
||||
make_offload_key,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.common import CPULoadStoreSpec
|
||||
from vllm.v1.kv_offload.cpu.common import (
|
||||
CPULoadStoreSpec,
|
||||
CPUOffloadingMetrics,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.manager import CPUOffloadingManager
|
||||
from vllm.v1.kv_offload.cpu.policies.arc import ARCCachePolicy
|
||||
|
||||
STORES_SKIPPED = "vllm:kv_offload_stores_skipped"
|
||||
|
||||
|
||||
def make_req_context(
|
||||
req_id: str = "", kv_transfer_params: dict | None = None
|
||||
@@ -181,10 +182,45 @@ def test_filter_reused_manager_reports_stores_skipped_counter():
|
||||
)
|
||||
stats = manager.get_stats()
|
||||
assert stats is not None
|
||||
assert stats.reduce()[STORES_SKIPPED] == 3
|
||||
assert stats.reduce()[CPUOffloadingMetrics.STORES_SKIPPED] == 3
|
||||
stats = manager.get_stats()
|
||||
assert stats is not None
|
||||
assert stats.reduce()[STORES_SKIPPED] == 0
|
||||
assert stats.reduce()[CPUOffloadingMetrics.STORES_SKIPPED] == 0
|
||||
|
||||
|
||||
def test_cpu_manager_reports_cache_usage_gauge():
|
||||
def check_usage_stats(manager: CPUOffloadingManager, value: float):
|
||||
stats = manager.get_stats()
|
||||
assert stats is not None
|
||||
assert stats.reduce()[
|
||||
CPUOffloadingMetrics.CPU_CACHE_USAGE_PERC
|
||||
] == pytest.approx(value)
|
||||
|
||||
# Zero-capacity manager always reports 0.0
|
||||
manager = make_cpu_manager(num_blocks=0)
|
||||
check_usage_stats(manager, 0.0)
|
||||
|
||||
# Empty manager (4 blocks, none allocated): usage = 0.0
|
||||
manager = make_cpu_manager(num_blocks=4)
|
||||
check_usage_stats(manager, 0.0)
|
||||
|
||||
# After allocating 2 of 4 blocks: usage = 0.5
|
||||
manager.prepare_store(to_keys([1, 2]), _EMPTY_REQ_CTX)
|
||||
check_usage_stats(manager, 0.5)
|
||||
|
||||
# After filling all 4 blocks: usage = 1.0
|
||||
manager.prepare_store(to_keys([3, 4]), _EMPTY_REQ_CTX)
|
||||
check_usage_stats(manager, 1.0)
|
||||
|
||||
# After completing store, the blocks becomes evictable as it is not actively used
|
||||
# and usage drops.
|
||||
manager.complete_store(to_keys([1, 2]), _EMPTY_REQ_CTX)
|
||||
check_usage_stats(manager, 0.5)
|
||||
|
||||
# After completing store, the blocks becomes evictable as it is not actively used
|
||||
# and usage drops.
|
||||
manager.complete_store(to_keys([3, 4]), _EMPTY_REQ_CTX)
|
||||
check_usage_stats(manager, 0.0)
|
||||
|
||||
|
||||
def test_cpu_manager():
|
||||
|
||||
@@ -145,7 +145,6 @@ def _generate_fake_sampling_metadata(
|
||||
vllm_config.scheduler_config.max_num_seqs,
|
||||
num_spec,
|
||||
device,
|
||||
PIN_MEMORY_AVAILABLE,
|
||||
)
|
||||
fake_sampling_metadata = SamplingMetadata(
|
||||
temperature=torch.full((batch_size,), 0.0),
|
||||
@@ -880,7 +879,6 @@ def test_maybe_create_thinking_budget_holder_without_reasoning():
|
||||
cfg.scheduler_config.max_num_seqs,
|
||||
0,
|
||||
torch.device("cpu"),
|
||||
False,
|
||||
)
|
||||
is None
|
||||
)
|
||||
|
||||
@@ -6,6 +6,7 @@ from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from vllm import SamplingParams
|
||||
@@ -1528,3 +1529,232 @@ def test_reset_pending_loads() -> None:
|
||||
# All GPU blocks free
|
||||
num_used = gpu_pool.num_gpu_blocks - gpu_pool.get_num_free_blocks()
|
||||
assert num_used == 1, f"Expected only null block in use, got {num_used}"
|
||||
|
||||
|
||||
def _make_cp_vllm_config(
|
||||
dcp_world_size: int = 1,
|
||||
pcp_world_size: int = 1,
|
||||
) -> VllmConfig:
|
||||
"""VllmConfig with context-parallel sizes set for scheduler-only tests."""
|
||||
cfg = _make_vllm_config()
|
||||
|
||||
cfg.parallel_config.decode_context_parallel_size = dcp_world_size
|
||||
cfg.parallel_config.prefill_context_parallel_size = pcp_world_size
|
||||
return cfg
|
||||
|
||||
|
||||
def _make_cp_scheduler(
|
||||
*,
|
||||
dcp_world_size: int = 1,
|
||||
pcp_world_size: int = 1,
|
||||
num_cpu_blocks: int = 8,
|
||||
num_gpu_blocks: int = 16,
|
||||
lazy: bool = False,
|
||||
) -> SchedulerFixture:
|
||||
"""Build a SimpleCPUOffloadScheduler with CP-scaled virtual block size."""
|
||||
cp_world_size = dcp_world_size * pcp_world_size
|
||||
virtual_block_size = BLOCK_SIZE * cp_world_size
|
||||
|
||||
kv_cache_config = _make_kv_cache_config(num_gpu_blocks)
|
||||
vllm_config = _make_cp_vllm_config(dcp_world_size, pcp_world_size)
|
||||
cpu_capacity_bytes = _BYTES_PER_BLOCK * num_cpu_blocks
|
||||
|
||||
sched = SimpleCPUOffloadScheduler(
|
||||
vllm_config=vllm_config,
|
||||
kv_cache_config=kv_cache_config,
|
||||
cpu_capacity_bytes=cpu_capacity_bytes,
|
||||
scheduler_block_size=virtual_block_size,
|
||||
hash_block_size=virtual_block_size,
|
||||
lazy_offload=lazy,
|
||||
)
|
||||
|
||||
gpu_block_pool = BlockPool(
|
||||
num_gpu_blocks=num_gpu_blocks,
|
||||
enable_caching=True,
|
||||
hash_block_size=virtual_block_size,
|
||||
)
|
||||
sched.bind_gpu_block_pool(gpu_block_pool)
|
||||
|
||||
return SchedulerFixture(
|
||||
scheduler=sched,
|
||||
gpu_block_pool=gpu_block_pool,
|
||||
vllm_config=vllm_config,
|
||||
kv_cache_config=kv_cache_config,
|
||||
)
|
||||
|
||||
|
||||
def _make_cp_request(
|
||||
num_blocks: int,
|
||||
virtual_block_size: int,
|
||||
request_id: str | None = None,
|
||||
) -> Request:
|
||||
"""Create a request whose block hashes are computed at the virtual
|
||||
(CP-scaled) block size, matching what the real scheduler does.
|
||||
"""
|
||||
global _req_counter
|
||||
_req_counter += 1
|
||||
if request_id is None:
|
||||
request_id = f"req-cp-{_req_counter}"
|
||||
|
||||
num_tokens = num_blocks * virtual_block_size + 1
|
||||
start = _req_counter * 10000
|
||||
prompt_token_ids = list(range(start, start + num_tokens))
|
||||
sampling_params = SamplingParams(max_tokens=1)
|
||||
|
||||
return Request(
|
||||
request_id=request_id,
|
||||
prompt_token_ids=prompt_token_ids,
|
||||
sampling_params=sampling_params,
|
||||
pooling_params=None,
|
||||
mm_features=None,
|
||||
block_hasher=get_request_block_hasher(virtual_block_size, sha256),
|
||||
)
|
||||
|
||||
|
||||
def _allocate_cp_gpu_blocks(
|
||||
gpu_block_pool: BlockPool,
|
||||
request: Request,
|
||||
num_blocks: int,
|
||||
virtual_block_size: int,
|
||||
group_id: int = 0,
|
||||
) -> list:
|
||||
"""Allocate GPU blocks and cache them using the CP-scaled block size."""
|
||||
blocks = gpu_block_pool.get_new_blocks(num_blocks)
|
||||
num_full = min(num_blocks, len(request.block_hashes))
|
||||
if num_full > 0:
|
||||
gpu_block_pool.cache_full_blocks(
|
||||
request=request,
|
||||
blocks=blocks,
|
||||
num_cached_blocks=0,
|
||||
num_full_blocks=num_full,
|
||||
block_size=virtual_block_size,
|
||||
kv_cache_group_id=group_id,
|
||||
)
|
||||
return blocks
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Test 15: CP block size scaling is correct
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize(
|
||||
"dcp_world_size, pcp_world_size",
|
||||
[
|
||||
(2, 1), # DCP only
|
||||
(1, 2), # PCP only
|
||||
(2, 2), # DCP + PCP
|
||||
],
|
||||
)
|
||||
def test_cp_block_size_scaling(dcp_world_size: int, pcp_world_size: int) -> None:
|
||||
"""Verify that the scheduler's block_size and cp_world_size are correctly
|
||||
scaled when context parallelism is enabled."""
|
||||
fix = _make_cp_scheduler(
|
||||
dcp_world_size=dcp_world_size, pcp_world_size=pcp_world_size
|
||||
)
|
||||
sched = fix.scheduler
|
||||
|
||||
expected_cp = dcp_world_size * pcp_world_size
|
||||
assert sched.cp_world_size == expected_cp
|
||||
assert sched.block_size == BLOCK_SIZE * expected_cp
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Test 16: CP eager store-and-load roundtrip
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize(
|
||||
"dcp_world_size, pcp_world_size",
|
||||
[
|
||||
(2, 1),
|
||||
(1, 2),
|
||||
],
|
||||
)
|
||||
def test_cp_eager_store_and_load_roundtrip(
|
||||
dcp_world_size: int, pcp_world_size: int
|
||||
) -> None:
|
||||
"""With CP enabled, store blocks to CPU and reload them for a new request
|
||||
with matching tokens. Verifies that hash matching and transfer-pair
|
||||
construction work with the virtual block size."""
|
||||
fix = _make_cp_scheduler(
|
||||
dcp_world_size=dcp_world_size,
|
||||
pcp_world_size=pcp_world_size,
|
||||
num_cpu_blocks=8,
|
||||
num_gpu_blocks=16,
|
||||
lazy=False,
|
||||
)
|
||||
sched = fix.scheduler
|
||||
cp = dcp_world_size * pcp_world_size
|
||||
vbs = BLOCK_SIZE * cp
|
||||
|
||||
num_blocks = 2
|
||||
req = _make_cp_request(num_blocks, vbs)
|
||||
|
||||
# Allocate GPU blocks and register hashes
|
||||
gpu_blocks = _allocate_cp_gpu_blocks(fix.gpu_block_pool, req, num_blocks, vbs)
|
||||
kv_blocks = KVCacheBlocks(blocks=(gpu_blocks,))
|
||||
req.num_computed_tokens = num_blocks * vbs
|
||||
sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
|
||||
|
||||
block_ids = kv_blocks.get_block_ids()
|
||||
sched_out = make_scheduler_output(
|
||||
{req.request_id: num_blocks * vbs},
|
||||
new_reqs={req.request_id: block_ids},
|
||||
)
|
||||
|
||||
meta = sched.build_connector_meta(sched_out)
|
||||
assert meta.store_event >= 0, "Expected a store event"
|
||||
assert len(meta.store_gpu_blocks) == num_blocks
|
||||
assert len(meta.store_cpu_blocks) == num_blocks
|
||||
simulate_store_completion(sched, meta.store_event)
|
||||
|
||||
# New request with same tokens — should get a full CPU cache hit.
|
||||
req2 = Request(
|
||||
request_id="req-cp-load",
|
||||
prompt_token_ids=req.prompt_token_ids,
|
||||
sampling_params=req.sampling_params,
|
||||
pooling_params=None,
|
||||
mm_features=None,
|
||||
block_hasher=req._block_hasher,
|
||||
)
|
||||
|
||||
hit_tokens, is_async = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
|
||||
assert hit_tokens == num_blocks * vbs
|
||||
assert is_async is True
|
||||
|
||||
# Allocate fresh GPU blocks for the load.
|
||||
gpu_blocks2 = fix.gpu_block_pool.get_new_blocks(num_blocks)
|
||||
kv_blocks2 = KVCacheBlocks(blocks=(gpu_blocks2,))
|
||||
sched.update_state_after_alloc(req2, kv_blocks2, num_external_tokens=hit_tokens)
|
||||
|
||||
sched_out2 = make_scheduler_output(
|
||||
{req2.request_id: 1},
|
||||
new_reqs={req2.request_id: kv_blocks2.get_block_ids()},
|
||||
)
|
||||
meta2 = sched.build_connector_meta(sched_out2)
|
||||
assert meta2.load_event >= 0, "Expected a load event"
|
||||
assert len(meta2.load_gpu_blocks) == num_blocks
|
||||
assert len(meta2.load_cpu_blocks) == num_blocks
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Test 17: CP lazy target blocks are scaled correctly
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize("cp_world_size", [1, 2, 4])
|
||||
def test_cp_lazy_target_blocks_scaling(cp_world_size: int) -> None:
|
||||
"""_estimate_lazy_target_blocks returns fewer blocks when cp_world_size > 1
|
||||
because each virtual block covers more tokens."""
|
||||
kv_cache_config = _make_kv_cache_config(num_blocks=16)
|
||||
max_batched = 64
|
||||
|
||||
target_base = SimpleCPUOffloadScheduler._estimate_lazy_target_blocks(
|
||||
kv_cache_config, max_batched, cp_world_size=1
|
||||
)
|
||||
target_cp = SimpleCPUOffloadScheduler._estimate_lazy_target_blocks(
|
||||
kv_cache_config, max_batched, cp_world_size=cp_world_size
|
||||
)
|
||||
|
||||
if cp_world_size == 1:
|
||||
assert target_cp == target_base
|
||||
else:
|
||||
assert target_cp < target_base, (
|
||||
f"cp_world_size={cp_world_size}: target_cp={target_cp} should be "
|
||||
f"less than target_base={target_base}"
|
||||
)
|
||||
|
||||
@@ -35,7 +35,6 @@ def mock_model_runner_with_input_batch():
|
||||
max_model_len=1024,
|
||||
max_num_batched_tokens=1024,
|
||||
device="cpu",
|
||||
pin_memory=False,
|
||||
vocab_size=32000,
|
||||
block_sizes=[16],
|
||||
kernel_block_sizes=[16],
|
||||
|
||||
@@ -10,7 +10,6 @@ import torch
|
||||
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.sampling_params import SamplingParams
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import make_tensor_with_pad
|
||||
from vllm.v1.pool.metadata import PoolingMetadata
|
||||
from vllm.v1.sample.logits_processor import LogitsProcessors
|
||||
@@ -236,7 +235,6 @@ def test_sampling_metadata_in_input_batch(device: str, batch_size: int):
|
||||
max_model_len=1024,
|
||||
max_num_batched_tokens=1024,
|
||||
device=torch.device(device),
|
||||
pin_memory=is_pin_memory_available(),
|
||||
vocab_size=1024,
|
||||
block_sizes=[1],
|
||||
kernel_block_sizes=[1],
|
||||
@@ -331,7 +329,6 @@ def test_swap_states_in_input_batch(device: str, batch_size: int, swap_list: lis
|
||||
max_model_len=1024,
|
||||
max_num_batched_tokens=1024,
|
||||
device=torch.device(device),
|
||||
pin_memory=is_pin_memory_available(),
|
||||
vocab_size=1024,
|
||||
block_sizes=[1],
|
||||
kernel_block_sizes=[1],
|
||||
@@ -341,7 +338,6 @@ def test_swap_states_in_input_batch(device: str, batch_size: int, swap_list: lis
|
||||
max_model_len=1024,
|
||||
max_num_batched_tokens=1024,
|
||||
device=torch.device(device),
|
||||
pin_memory=is_pin_memory_available(),
|
||||
vocab_size=1024,
|
||||
block_sizes=[1],
|
||||
kernel_block_sizes=[1],
|
||||
@@ -410,7 +406,6 @@ def test_pooling_prompt_lens_not_aliased(device: str):
|
||||
max_model_len=MAX_PROMPT_SIZE + NUM_OUTPUT_TOKENS,
|
||||
max_num_batched_tokens=batch_size * (MAX_PROMPT_SIZE + NUM_OUTPUT_TOKENS),
|
||||
device=torch.device(device),
|
||||
pin_memory=is_pin_memory_available(),
|
||||
vocab_size=VOCAB_SIZE,
|
||||
block_sizes=[16],
|
||||
kernel_block_sizes=[16],
|
||||
@@ -459,7 +454,6 @@ def test_pooling_metadata_token_id_buffers(
|
||||
max_model_len=MAX_PROMPT_SIZE + NUM_OUTPUT_TOKENS,
|
||||
max_num_batched_tokens=MAX_PROMPT_SIZE + NUM_OUTPUT_TOKENS,
|
||||
device=torch.device("cpu"),
|
||||
pin_memory=False,
|
||||
vocab_size=VOCAB_SIZE,
|
||||
block_sizes=[16],
|
||||
kernel_block_sizes=[16],
|
||||
|
||||
@@ -85,7 +85,6 @@ def initialize_kv_cache(runner: GPUModelRunner):
|
||||
max_model_len=runner.max_model_len,
|
||||
max_num_batched_tokens=runner.max_num_tokens,
|
||||
device=runner.device,
|
||||
pin_memory=runner.pin_memory,
|
||||
vocab_size=runner.model_config.get_vocab_size(),
|
||||
block_sizes=[kv_cache_config.kv_cache_groups[0].kv_cache_spec.block_size],
|
||||
kernel_block_sizes=[
|
||||
@@ -1405,7 +1404,6 @@ def test_input_batch_with_kernel_block_sizes():
|
||||
max_model_len = 512
|
||||
max_num_batched_tokens = 512
|
||||
device = torch.device(DEVICE_TYPE)
|
||||
pin_memory = False
|
||||
vocab_size = 50272
|
||||
|
||||
# Test with different kernel block sizes
|
||||
@@ -1417,7 +1415,6 @@ def test_input_batch_with_kernel_block_sizes():
|
||||
max_model_len=max_model_len,
|
||||
max_num_batched_tokens=max_num_batched_tokens,
|
||||
device=device,
|
||||
pin_memory=pin_memory,
|
||||
vocab_size=vocab_size,
|
||||
block_sizes=block_sizes,
|
||||
kernel_block_sizes=kernel_block_sizes,
|
||||
@@ -1478,7 +1475,6 @@ def test_hybrid_cache_integration(default_vllm_config, dist_init):
|
||||
max_model_len=runner.max_model_len,
|
||||
max_num_batched_tokens=runner.max_num_tokens,
|
||||
device=runner.device,
|
||||
pin_memory=runner.pin_memory,
|
||||
vocab_size=runner.model_config.get_vocab_size(),
|
||||
block_sizes=[kv_cache_config.kv_cache_groups[0].kv_cache_spec.block_size],
|
||||
kernel_block_sizes=[16],
|
||||
|
||||
@@ -991,7 +991,7 @@ class VllmBackend:
|
||||
},
|
||||
payload_fn=lambda: json.dumps(
|
||||
{
|
||||
"model": self.vllm_config.model_config.model,
|
||||
"model": getattr(self.vllm_config.model_config, "model", "unknown"),
|
||||
"prefix": self.prefix,
|
||||
"mode": str(cc.mode),
|
||||
"backend": cc.backend,
|
||||
|
||||
@@ -302,6 +302,14 @@ class ParallelConfig:
|
||||
Each entry must use `numactl --physcpubind` CPU-list syntax, for example
|
||||
`"0-3"` or `"0,2,4-7"`.
|
||||
"""
|
||||
assigned_physical_gpu_ids: list[int] | None = None
|
||||
"""Mapping from vLLM-local logical GPU IDs to physical GPU IDs.
|
||||
|
||||
For example, ``[2, 3]`` means logical GPU 0 maps to physical GPU 2,
|
||||
and logical GPU 1 maps to physical GPU 3. Physical IDs are used only
|
||||
at platform/topology boundaries such as NVML, NIC affinity, P2P
|
||||
checks, and final CUDA device selection when needed. When None,
|
||||
logical IDs map to visible device IDs in order."""
|
||||
|
||||
distributed_timeout_seconds: int | None = None
|
||||
"""Timeout in seconds for distributed operations (e.g., init_process_group).
|
||||
@@ -772,6 +780,7 @@ class ParallelConfig:
|
||||
"numa_bind",
|
||||
"numa_bind_nodes",
|
||||
"numa_bind_cpus",
|
||||
"assigned_physical_gpu_ids",
|
||||
}
|
||||
|
||||
from vllm.config.utils import get_hash_factors, hash_factors
|
||||
|
||||
@@ -18,8 +18,8 @@ import torch
|
||||
|
||||
from vllm.device_allocator import AllocationData, HandleType
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.system_utils import find_loaded_library
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
@@ -196,7 +196,7 @@ class CuMemAllocator:
|
||||
size_in_bytes,
|
||||
dtype=torch.uint8,
|
||||
device="cpu",
|
||||
pin_memory=is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
cpu_ptr = cpu_backup_tensor.data_ptr()
|
||||
libcudart.cudaMemcpy(cpu_ptr, ptr, size_in_bytes)
|
||||
|
||||
@@ -11,7 +11,7 @@ import torch
|
||||
|
||||
from vllm.device_allocator import AllocationData, HandleType
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
@@ -188,7 +188,7 @@ class XpuMemAllocator:
|
||||
size_in_bytes,
|
||||
dtype=torch.uint8,
|
||||
device="cpu",
|
||||
pin_memory=is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
cpu_ptr = cpu_backup_tensor.data_ptr()
|
||||
_xpu_memcpy_sync(
|
||||
|
||||
@@ -704,7 +704,14 @@ class FlashInferNVLinkOneSidedManager(All2AllManagerBase):
|
||||
self.num_experts = num_experts
|
||||
|
||||
self.cleanup()
|
||||
gpus_per_node = torch.accelerator.device_count()
|
||||
from vllm.platforms.interface import get_assigned_physical_gpu_ids
|
||||
|
||||
assigned_physical_gpu_ids = get_assigned_physical_gpu_ids()
|
||||
gpus_per_node = (
|
||||
len(assigned_physical_gpu_ids)
|
||||
if assigned_physical_gpu_ids is not None
|
||||
else torch.accelerator.device_count()
|
||||
)
|
||||
logger.debug(
|
||||
"Making One-sided NVLink mapping: rank=%d, world size=%d",
|
||||
self.rank,
|
||||
|
||||
@@ -320,13 +320,21 @@ def gpu_p2p_access_check(src: int, tgt: int) -> bool:
|
||||
|
||||
is_distributed = dist.is_initialized()
|
||||
|
||||
num_dev = current_platform.device_count()
|
||||
cuda_visible_devices = envs.CUDA_VISIBLE_DEVICES
|
||||
if cuda_visible_devices is None:
|
||||
cuda_visible_devices = ",".join(str(i) for i in range(num_dev))
|
||||
from vllm.platforms.interface import get_assigned_physical_gpu_ids
|
||||
|
||||
assigned_physical_gpu_ids = get_assigned_physical_gpu_ids()
|
||||
if assigned_physical_gpu_ids is not None:
|
||||
# Key by the ordered list: the cache stores directed local-index
|
||||
# pairs, so permutations of the same set are distinct mappings.
|
||||
cache_key = ",".join(str(i) for i in assigned_physical_gpu_ids)
|
||||
num_dev = len(assigned_physical_gpu_ids)
|
||||
else:
|
||||
num_dev = current_platform.device_count()
|
||||
cuda_visible_devices = envs.CUDA_VISIBLE_DEVICES
|
||||
cache_key = cuda_visible_devices or ",".join(str(i) for i in range(num_dev))
|
||||
|
||||
path = os.path.join(
|
||||
envs.VLLM_CACHE_ROOT, f"gpu_p2p_access_cache_for_{cuda_visible_devices}.json"
|
||||
envs.VLLM_CACHE_ROOT, f"gpu_p2p_access_cache_for_{cache_key}.json"
|
||||
)
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
from vllm.distributed.parallel_state import get_world_group
|
||||
@@ -338,7 +346,15 @@ def gpu_p2p_access_check(src: int, tgt: int) -> bool:
|
||||
# enter this block to calculate the cache
|
||||
logger.info("generating GPU P2P access cache in %s", path)
|
||||
cache: dict[str, bool] = {}
|
||||
ids = list(range(num_dev))
|
||||
# The probe subprocesses inherit this process's device-control env
|
||||
# var, so they must be given visible ordinals, not physical IDs.
|
||||
if assigned_physical_gpu_ids is not None:
|
||||
ids = [
|
||||
current_platform.logical_device_id_to_visible_device_id(local)
|
||||
for local in range(num_dev)
|
||||
]
|
||||
else:
|
||||
ids = list(range(num_dev))
|
||||
# batch of all pairs of GPUs
|
||||
batch_src, batch_tgt = zip(*list(product(ids, ids)))
|
||||
# NOTE: we use `subprocess` rather than `multiprocessing` here
|
||||
@@ -368,8 +384,11 @@ def gpu_p2p_access_check(src: int, tgt: int) -> bool:
|
||||
) from e
|
||||
with open(output_file.name, "rb") as f:
|
||||
result = pickle.load(f)
|
||||
# Cache entries must be keyed by local indices (0..N-1) because
|
||||
# gpu_p2p_access_check() is called with local ranks.
|
||||
id_to_local = {device_id: local for local, device_id in enumerate(ids)}
|
||||
for _i, _j, r in zip(batch_src, batch_tgt, result):
|
||||
cache[f"{_i}->{_j}"] = r
|
||||
cache[f"{id_to_local[_i]}->{id_to_local[_j]}"] = r
|
||||
with open(path, "w") as f:
|
||||
json.dump(cache, f, indent=4)
|
||||
if is_distributed:
|
||||
|
||||
@@ -34,7 +34,12 @@ def _can_p2p(rank: int, world_size: int) -> bool:
|
||||
continue
|
||||
if envs.VLLM_SKIP_P2P_CHECK:
|
||||
logger.debug("Skipping P2P check and trusting the driver's P2P report.")
|
||||
return torch.cuda.can_device_access_peer(rank, i)
|
||||
# can_device_access_peer takes visible device ordinals, while
|
||||
# rank and i are logical local IDs.
|
||||
return torch.cuda.can_device_access_peer(
|
||||
current_platform.logical_device_id_to_visible_device_id(rank),
|
||||
current_platform.logical_device_id_to_visible_device_id(i),
|
||||
)
|
||||
if not gpu_p2p_access_check(rank, i):
|
||||
return False
|
||||
return True
|
||||
@@ -126,13 +131,10 @@ class CustomAllreduce:
|
||||
CUSTOM_ALL_REDUCE_MAX_SIZES[device_capability_str][world_size],
|
||||
max_size,
|
||||
)
|
||||
cuda_visible_devices = envs.CUDA_VISIBLE_DEVICES
|
||||
if cuda_visible_devices:
|
||||
device_ids = list(map(int, cuda_visible_devices.split(",")))
|
||||
else:
|
||||
device_ids = list(range(current_platform.device_count()))
|
||||
|
||||
physical_device_id = device_ids[device.index]
|
||||
# device.index is a visible ordinal, not a logical local ID.
|
||||
physical_device_id = current_platform.visible_device_id_to_physical_device_id(
|
||||
device.index
|
||||
)
|
||||
tensor = torch.tensor([physical_device_id], dtype=torch.int, device="cpu")
|
||||
gather_list = [
|
||||
torch.tensor([0], dtype=torch.int, device="cpu") for _ in range(world_size)
|
||||
|
||||
@@ -129,12 +129,10 @@ class QuickAllReduce:
|
||||
assert isinstance(device, torch.device)
|
||||
self.device = device
|
||||
|
||||
cuda_visible_devices = envs.CUDA_VISIBLE_DEVICES
|
||||
if cuda_visible_devices:
|
||||
device_ids = list(map(int, cuda_visible_devices.split(",")))
|
||||
else:
|
||||
device_ids = list(range(current_platform.device_count()))
|
||||
physical_device_id = device_ids[device.index]
|
||||
# device.index is a visible ordinal, not a logical local ID.
|
||||
physical_device_id = current_platform.visible_device_id_to_physical_device_id(
|
||||
device.index
|
||||
)
|
||||
tensor = torch.tensor([physical_device_id], dtype=torch.int, device="cpu")
|
||||
gather_list = [
|
||||
torch.tensor([0], dtype=torch.int, device="cpu")
|
||||
|
||||
@@ -840,7 +840,13 @@ class MessageQueue:
|
||||
The MessageQueue instance for the calling process,
|
||||
and a list of handles (only non-empty for the reader process).
|
||||
"""
|
||||
local_size = current_platform.device_count()
|
||||
from vllm.platforms.interface import get_assigned_physical_gpu_ids
|
||||
|
||||
assigned_physical_gpu_ids = get_assigned_physical_gpu_ids()
|
||||
if assigned_physical_gpu_ids is not None:
|
||||
local_size = len(assigned_physical_gpu_ids)
|
||||
else:
|
||||
local_size = current_platform.device_count()
|
||||
rank = dist.get_rank()
|
||||
same_node = rank // local_size == reader_rank // local_size
|
||||
buffer_io = MessageQueue(
|
||||
|
||||
@@ -482,10 +482,11 @@ def _init_lmcache_engine(
|
||||
)
|
||||
|
||||
# Change current device.
|
||||
num_gpus = torch.accelerator.device_count()
|
||||
local_rank = parallel_config.rank % num_gpus
|
||||
torch.accelerator.set_device_index(local_rank)
|
||||
device = torch.device(f"cuda:{local_rank}")
|
||||
from vllm.distributed.parallel_state import get_world_group
|
||||
|
||||
device_index = get_world_group().device_index
|
||||
torch.accelerator.set_device_index(device_index)
|
||||
device = torch.device(f"cuda:{device_index}")
|
||||
metadata = LMCacheEngineMetadata(
|
||||
model_config.model,
|
||||
parallel_config.world_size,
|
||||
|
||||
@@ -2,13 +2,15 @@
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
"""External-store cache-hit coordinator for MooncakeStoreConnector."""
|
||||
|
||||
from collections.abc import Sequence
|
||||
from typing import cast
|
||||
|
||||
from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.data import (
|
||||
chunk_hashes_for_block_size,
|
||||
)
|
||||
from vllm.v1.core.block_pool import BlockPool
|
||||
from vllm.v1.core.kv_cache_utils import (
|
||||
BlockHash,
|
||||
BlockHashList,
|
||||
BlockHashListWithBlockSize,
|
||||
KVCacheBlock,
|
||||
)
|
||||
from vllm.v1.core.single_type_kv_cache_manager import (
|
||||
@@ -120,7 +122,7 @@ class MooncakeStoreCoordinator:
|
||||
|
||||
def find_longest_cache_hit(
|
||||
self,
|
||||
block_hashes: list[BlockHash],
|
||||
block_hashes: Sequence[BlockHash],
|
||||
max_length: int,
|
||||
cached_block_pool: ExternalCachedBlockPool,
|
||||
*,
|
||||
@@ -147,7 +149,7 @@ class MooncakeStoreCoordinator:
|
||||
|
||||
def load_mask(
|
||||
self,
|
||||
block_hashes: list[BlockHash],
|
||||
block_hashes: Sequence[BlockHash],
|
||||
token_len: int,
|
||||
) -> tuple[list[bool], ...]:
|
||||
"""Per-group load masks: ``mask[g][i]`` is True iff group ``g``'s
|
||||
@@ -236,17 +238,15 @@ class MooncakeStoreCoordinator:
|
||||
return tuple(masks)
|
||||
|
||||
def block_hashes_for_spec(
|
||||
self, block_hashes: list[BlockHash], spec: KVCacheSpec
|
||||
) -> BlockHashList:
|
||||
if spec.block_size == self.hash_block_size:
|
||||
return block_hashes
|
||||
return BlockHashListWithBlockSize(
|
||||
self, block_hashes: Sequence[BlockHash], spec: KVCacheSpec
|
||||
) -> Sequence[BlockHash]:
|
||||
return chunk_hashes_for_block_size(
|
||||
block_hashes, self.hash_block_size, spec.block_size
|
||||
)
|
||||
|
||||
def _find_hit_blocks(
|
||||
self,
|
||||
block_hashes: list[BlockHash],
|
||||
block_hashes: Sequence[BlockHash],
|
||||
max_length: int,
|
||||
cached_block_pool: ExternalCachedBlockPool,
|
||||
*,
|
||||
@@ -264,7 +264,7 @@ class MooncakeStoreCoordinator:
|
||||
spec, group_ids, manager_cls = self.attention_groups[0]
|
||||
hashes = self.block_hashes_for_spec(block_hashes, spec)
|
||||
hit_blocks = manager_cls.find_longest_cache_hit(
|
||||
block_hashes=hashes,
|
||||
block_hashes=hashes, # type: ignore[arg-type]
|
||||
max_length=max_length,
|
||||
kv_cache_group_ids=group_ids,
|
||||
block_pool=cast(BlockPool, cached_block_pool),
|
||||
@@ -304,7 +304,7 @@ class MooncakeStoreCoordinator:
|
||||
_max_length = min(curr_hit_length + spec.block_size, max_length)
|
||||
hashes = self.block_hashes_for_spec(block_hashes, spec)
|
||||
hit_blocks = manager_cls.find_longest_cache_hit(
|
||||
block_hashes=hashes,
|
||||
block_hashes=hashes, # type: ignore[arg-type]
|
||||
max_length=_max_length,
|
||||
kv_cache_group_ids=group_ids,
|
||||
block_pool=cast(BlockPool, cached_block_pool),
|
||||
|
||||
@@ -5,8 +5,9 @@
|
||||
# (vllm_ascend/distributed/kv_transfer/kv_pool/ascend_store/).
|
||||
"""Data classes for MooncakeStoreConnector."""
|
||||
|
||||
from collections.abc import Iterable
|
||||
from collections.abc import Iterable, Sequence
|
||||
from dataclasses import dataclass
|
||||
from typing import cast
|
||||
|
||||
import torch
|
||||
|
||||
@@ -23,6 +24,77 @@ from vllm.v1.core.kv_cache_utils import (
|
||||
logger = init_logger(__name__)
|
||||
|
||||
|
||||
class BlobBlockHashes(Sequence[BlockHash]):
|
||||
"""Lazy view over a flat buffer of fixed-size block hashes to avoid the overhead
|
||||
of materializing all hashes upfront.
|
||||
"""
|
||||
|
||||
def __init__(self, blob: memoryview, hash_len: int):
|
||||
self._blob = blob
|
||||
self._hash_len = hash_len
|
||||
self._n = len(blob) // hash_len if hash_len else 0
|
||||
|
||||
def __len__(self) -> int:
|
||||
return self._n
|
||||
|
||||
def __getitem__(self, idx):
|
||||
if isinstance(idx, slice):
|
||||
return [self[i] for i in range(*idx.indices(self._n))]
|
||||
if idx < 0:
|
||||
idx += self._n
|
||||
if not 0 <= idx < self._n:
|
||||
raise IndexError(idx)
|
||||
off = idx * self._hash_len
|
||||
return BlockHash(self._blob[off : off + self._hash_len])
|
||||
|
||||
|
||||
class _CompactChunkHashList(BlockHashListWithBlockSize):
|
||||
"""View that keys each ``block_size`` chunk by the last constituent
|
||||
``hash_block_size`` hash instead of concatenating all of them.
|
||||
|
||||
The engine chains block hashes (each hash folds in the previous one), so the
|
||||
final sub-block hash of a chunk already uniquely identifies the whole chunk
|
||||
and its prefix. Using it keeps a Mooncake key at a single hash digest
|
||||
regardless of the ``block_size`` / ``hash_block_size`` ratio, instead of
|
||||
growing the key linearly with it (e.g. 64x for ``block_size=256``,
|
||||
``hash_block_size=4``).
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
block_hashes: Sequence[BlockHash],
|
||||
hash_block_size: int,
|
||||
target_block_size: int,
|
||||
):
|
||||
# Accept any indexable sequence (e.g. the lazy ``BlobBlockHashes``), not
|
||||
# just ``list``; the base only indexes/sizes it.
|
||||
assert target_block_size % hash_block_size == 0
|
||||
self.block_hashes = block_hashes # type: ignore[assignment]
|
||||
self.scale_factor = target_block_size // hash_block_size
|
||||
|
||||
def _get_value_at(self, idx: int) -> BlockHash:
|
||||
return self.block_hashes[idx * self.scale_factor + self.scale_factor - 1]
|
||||
|
||||
|
||||
def chunk_hashes_for_block_size(
|
||||
block_hashes: Sequence[BlockHash],
|
||||
hash_block_size: int,
|
||||
block_size: int,
|
||||
) -> Sequence[BlockHash]:
|
||||
"""Map ``hash_block_size``-granular block hashes to one compact hash per
|
||||
``block_size`` chunk (the chunk's last sub-hash). Returns ``block_hashes``
|
||||
unchanged when the two sizes are equal.
|
||||
"""
|
||||
if block_size == hash_block_size:
|
||||
return block_hashes
|
||||
# Structurally a Sequence[BlockHash] (indexable + sized); the base class
|
||||
# just isn't declared as one.
|
||||
return cast(
|
||||
"Sequence[BlockHash]",
|
||||
_CompactChunkHashList(block_hashes, hash_block_size, block_size),
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class KeyMetadata:
|
||||
"""Metadata for constructing pool keys."""
|
||||
@@ -138,18 +210,15 @@ class ChunkedTokenDatabase:
|
||||
Args:
|
||||
token_len: Total number of tokens.
|
||||
block_hashes: Block hashes computed at ``hash_block_size`` granularity.
|
||||
When ``block_size > hash_block_size`` consecutive hashes are merged
|
||||
up to the group's ``block_size`` via ``BlockHashListWithBlockSize``.
|
||||
When ``block_size > hash_block_size`` each group's ``block_size`` chunk
|
||||
is keyed by its last sub-hash via ``chunk_hashes_for_block_size``.
|
||||
mask_num: Number of tokens to skip from the beginning.
|
||||
"""
|
||||
if not block_hashes:
|
||||
return
|
||||
if self.block_size == self.hash_block_size:
|
||||
chunk_hashes: Iterable[BlockHash] = block_hashes
|
||||
else:
|
||||
chunk_hashes = BlockHashListWithBlockSize(
|
||||
block_hashes, self.hash_block_size, self.block_size
|
||||
)
|
||||
chunk_hashes: Iterable[BlockHash] = chunk_hashes_for_block_size(
|
||||
block_hashes, self.hash_block_size, self.block_size
|
||||
)
|
||||
for chunk_id, h in enumerate(chunk_hashes):
|
||||
start_idx = chunk_id * self.block_size
|
||||
if start_idx >= token_len:
|
||||
|
||||
@@ -11,7 +11,10 @@ Wire format (REQ/REP over IPC):
|
||||
|
||||
msg_type == LOOKUP_MSG:
|
||||
frame 1: token_len (u32 big-endian, 4 bytes)
|
||||
frame 2..n: msgpack-encoded list[str] of block-hash hex digests
|
||||
frame 2: hash_len (u16 big-endian, 2 bytes) — byte length of each
|
||||
fixed-size block hash (0 when there are no hashes)
|
||||
frame 3: raw block hashes concatenated back-to-back (each hash_len
|
||||
bytes); the server splits on hash_len
|
||||
Response: [hit_count: u32 big-endian, 4 bytes]
|
||||
|
||||
msg_type == RESET_MSG:
|
||||
|
||||
@@ -18,7 +18,7 @@ import socket
|
||||
import threading
|
||||
import time
|
||||
from collections import defaultdict
|
||||
from collections.abc import Callable
|
||||
from collections.abc import Callable, Sequence
|
||||
from concurrent.futures import Future, ThreadPoolExecutor
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Literal, TypeVar
|
||||
@@ -45,6 +45,7 @@ from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.coordinator imp
|
||||
MooncakeStoreCoordinator,
|
||||
)
|
||||
from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.data import ( # noqa: E501
|
||||
BlobBlockHashes,
|
||||
ChunkedTokenDatabase,
|
||||
KeyMetadata,
|
||||
MooncakeStoreConnectorMetadata,
|
||||
@@ -65,7 +66,6 @@ from vllm.v1.core.kv_cache_utils import (
|
||||
resolve_kv_cache_block_sizes,
|
||||
)
|
||||
from vllm.v1.kv_cache_interface import KVCacheConfig, KVCacheGroupSpec
|
||||
from vllm.v1.serial_utils import MsgpackDecoder, MsgpackEncoder
|
||||
|
||||
from .metrics import MooncakeStoreConnectorStats
|
||||
|
||||
@@ -1372,7 +1372,7 @@ class MooncakeStoreWorker:
|
||||
|
||||
return finished_sending
|
||||
|
||||
def lookup(self, token_len: int, block_hashes: list[BlockHash]) -> int:
|
||||
def lookup(self, token_len: int, block_hashes: Sequence[BlockHash]) -> int:
|
||||
"""Check how many prefix tokens exist in the store.
|
||||
|
||||
Checks across all TP ranks and PP ranks.
|
||||
@@ -1392,6 +1392,11 @@ class MooncakeStoreWorker:
|
||||
group_hashes = self.coord.block_hashes_for_spec(
|
||||
block_hashes, self._kv_cache_groups[g_idx].kv_cache_spec
|
||||
)
|
||||
metadata_templates = [
|
||||
dataclasses.replace(db.metadata, tp_rank=tp, pp_rank=pp)
|
||||
for tp in range(tp_count)
|
||||
for pp in range(self.pp_size)
|
||||
]
|
||||
for chunk_id, h in enumerate(group_hashes):
|
||||
start_idx = chunk_id * spec_block_size
|
||||
if start_idx >= token_len:
|
||||
@@ -1400,11 +1405,11 @@ class MooncakeStoreWorker:
|
||||
chunk_id >= len(lookup_mask) or not lookup_mask[chunk_id]
|
||||
):
|
||||
continue
|
||||
for tp in range(tp_count):
|
||||
for pp in range(self.pp_size):
|
||||
md = dataclasses.replace(db.metadata, tp_rank=tp, pp_rank=pp)
|
||||
candidate_keys.append(PoolKey(md, h.hex()).to_string())
|
||||
candidate_meta.append((g_idx, bytes(h)))
|
||||
h_hex = h.hex()
|
||||
h_bytes = bytes(h)
|
||||
for md in metadata_templates:
|
||||
candidate_keys.append(PoolKey(md, h_hex).to_string())
|
||||
candidate_meta.append((g_idx, h_bytes))
|
||||
|
||||
if not candidate_keys:
|
||||
return 0
|
||||
@@ -1483,7 +1488,6 @@ class LookupKeyServer:
|
||||
store_worker: MooncakeStoreWorker,
|
||||
vllm_config: VllmConfig,
|
||||
):
|
||||
self.decoder = MsgpackDecoder()
|
||||
self.ctx = zmq.Context() # type: ignore[attr-defined]
|
||||
socket_path = get_zmq_rpc_path_lookup(vllm_config)
|
||||
self._ipc_path = socket_path.removeprefix("ipc://")
|
||||
@@ -1506,9 +1510,9 @@ class LookupKeyServer:
|
||||
|
||||
if msg_type == LOOKUP_MSG:
|
||||
token_len = int.from_bytes(all_frames[1], byteorder="big")
|
||||
hash_frames = all_frames[2:]
|
||||
hashes_str = self.decoder.decode(hash_frames)
|
||||
block_hashes = [BlockHash(bytes.fromhex(s)) for s in hashes_str]
|
||||
hash_len = int.from_bytes(all_frames[2], byteorder="big")
|
||||
blob = all_frames[3].buffer
|
||||
block_hashes = BlobBlockHashes(blob, hash_len)
|
||||
result = self.store_worker.lookup(token_len, block_hashes)
|
||||
self.socket.send(result.to_bytes(4, "big"))
|
||||
|
||||
@@ -1557,7 +1561,6 @@ class LookupKeyClient:
|
||||
"""
|
||||
|
||||
def __init__(self, vllm_config: VllmConfig):
|
||||
self.encoder = MsgpackEncoder()
|
||||
self.ctx = zmq.Context() # type: ignore[attr-defined]
|
||||
socket_path = get_zmq_rpc_path_lookup(vllm_config)
|
||||
self.socket = make_zmq_socket(
|
||||
@@ -1574,14 +1577,16 @@ class LookupKeyClient:
|
||||
self.futures: dict[str, Future[int]] = {}
|
||||
|
||||
def _lookup(self, token_len: int, block_hashes: list[BlockHash]) -> int:
|
||||
hash_strs = [h.hex() for h in block_hashes]
|
||||
hash_frames = self.encoder.encode(hash_strs)
|
||||
token_len_bytes = token_len.to_bytes(4, byteorder="big")
|
||||
all_frames = [LOOKUP_MSG, token_len_bytes] + list(hash_frames)
|
||||
hash_len = len(block_hashes[0]) if block_hashes else 0
|
||||
all_frames = (
|
||||
LOOKUP_MSG,
|
||||
token_len.to_bytes(4, byteorder="big"),
|
||||
hash_len.to_bytes(2, byteorder="big"),
|
||||
b"".join(block_hashes),
|
||||
)
|
||||
self.socket.send_multipart(all_frames, copy=False)
|
||||
resp = self.socket.recv()
|
||||
result = int.from_bytes(resp, "big")
|
||||
return result
|
||||
return int.from_bytes(resp, "big")
|
||||
|
||||
def lookup(
|
||||
self,
|
||||
|
||||
@@ -50,7 +50,8 @@ class OffloadingConnectorWorker:
|
||||
def register_kv_caches(
|
||||
self, kv_caches: dict[str, torch.Tensor | list[torch.Tensor]]
|
||||
):
|
||||
num_blocks = self.spec.kv_cache_config.num_blocks
|
||||
kv_cache_config = self.spec.kv_cache_config
|
||||
num_blocks = kv_cache_config.num_blocks
|
||||
|
||||
# layer_name -> (num_blocks, page_size_bytes) tensor
|
||||
tensors_per_block: dict[str, tuple[torch.Tensor, ...]] = {}
|
||||
@@ -58,7 +59,7 @@ class OffloadingConnectorWorker:
|
||||
unpadded_page_size_bytes: dict[str, int] = {}
|
||||
# layer_name -> size of page in bytes
|
||||
page_size_bytes: dict[str, int] = {}
|
||||
for kv_cache_group in self.spec.kv_cache_config.kv_cache_groups:
|
||||
for kv_cache_group in kv_cache_config.kv_cache_groups:
|
||||
group_layer_names = kv_cache_group.layer_names
|
||||
group_kv_cache_spec = kv_cache_group.kv_cache_spec
|
||||
if isinstance(group_kv_cache_spec, UniformTypeKVCacheSpecs):
|
||||
@@ -122,9 +123,35 @@ class OffloadingConnectorWorker:
|
||||
else:
|
||||
raise NotImplementedError
|
||||
|
||||
packed_kv_cache_tensor = next(
|
||||
(t for t in kv_cache_config.kv_cache_tensors if t.block_stride), None
|
||||
)
|
||||
is_dsv4 = all(
|
||||
isinstance(group.kv_cache_spec, UniformTypeKVCacheSpecs)
|
||||
for group in kv_cache_config.kv_cache_groups
|
||||
)
|
||||
if packed_kv_cache_tensor is not None and not is_dsv4:
|
||||
(tensor,) = tensors_per_block[packed_kv_cache_tensor.shared_by[0]]
|
||||
block_stride = tensor.stride(0)
|
||||
packed_tensor = tensor.as_strided(
|
||||
(num_blocks, block_stride),
|
||||
(block_stride, 1),
|
||||
storage_offset=0,
|
||||
)
|
||||
self._register_handlers(
|
||||
CanonicalKVCaches(
|
||||
[CanonicalKVCacheTensor(packed_tensor, block_stride)],
|
||||
[
|
||||
[CanonicalKVCacheRef(0, block_stride)]
|
||||
for _ in kv_cache_config.kv_cache_groups
|
||||
],
|
||||
)
|
||||
)
|
||||
return
|
||||
|
||||
block_tensors: list[CanonicalKVCacheTensor] = []
|
||||
block_data_refs: dict[str, list[CanonicalKVCacheRef]] = defaultdict(list)
|
||||
for kv_cache_tensor in self.spec.kv_cache_config.kv_cache_tensors:
|
||||
for kv_cache_tensor in kv_cache_config.kv_cache_tensors:
|
||||
# Filter to layers that were actually processed above.
|
||||
# _get_kv_cache_config_deepseek_v4 emits KVCacheTensor entries for
|
||||
# every (tuple_idx, page_size) slot; slots where no group has a
|
||||
@@ -166,7 +193,7 @@ class OffloadingConnectorWorker:
|
||||
)
|
||||
|
||||
group_data_refs: list[list[CanonicalKVCacheRef]] = []
|
||||
for kv_cache_group in self.spec.kv_cache_config.kv_cache_groups:
|
||||
for kv_cache_group in kv_cache_config.kv_cache_groups:
|
||||
group_refs: list[CanonicalKVCacheRef] = []
|
||||
for layer_name in kv_cache_group.layer_names:
|
||||
group_refs += block_data_refs[layer_name]
|
||||
|
||||
@@ -392,6 +392,14 @@ class GroupCoordinator:
|
||||
|
||||
self.rank = torch.distributed.get_rank()
|
||||
self.local_rank = local_rank
|
||||
self.device_index: int
|
||||
if _WORLD is not None:
|
||||
self.device_index = _WORLD.device_index
|
||||
else:
|
||||
assert local_rank >= 0, (
|
||||
"local_rank must be provided when creating the world group"
|
||||
)
|
||||
self.device_index = local_rank
|
||||
|
||||
self_device_group = None
|
||||
self_cpu_group = None
|
||||
@@ -442,11 +450,18 @@ class GroupCoordinator:
|
||||
from vllm.platforms import current_platform
|
||||
|
||||
if current_platform.is_cuda_alike():
|
||||
self.device = torch.device(f"cuda:{local_rank}")
|
||||
visible_device_index = (
|
||||
current_platform.logical_device_id_to_visible_device_id(
|
||||
self.device_index
|
||||
)
|
||||
)
|
||||
self.device = torch.device(f"cuda:{visible_device_index}")
|
||||
elif current_platform.is_xpu():
|
||||
self.device = torch.device(f"xpu:{local_rank}")
|
||||
self.device = torch.device(f"xpu:{self.device_index}")
|
||||
elif current_platform.is_out_of_tree():
|
||||
self.device = torch.device(f"{current_platform.device_name}:{local_rank}")
|
||||
self.device = torch.device(
|
||||
f"{current_platform.device_name}:{self.device_index}"
|
||||
)
|
||||
else:
|
||||
self.device = torch.device("cpu")
|
||||
|
||||
@@ -1438,7 +1453,12 @@ def _init_process_group_for_split_group(
|
||||
"""
|
||||
if torch.accelerator.is_available() and backend != "gloo":
|
||||
init_backend = "cpu:gloo,cuda:nccl"
|
||||
device_id: torch.device | None = torch.device(f"cuda:{local_rank}")
|
||||
from vllm.platforms import current_platform
|
||||
|
||||
visible_device_index = current_platform.logical_device_id_to_visible_device_id(
|
||||
local_rank
|
||||
)
|
||||
device_id: torch.device | None = torch.device(f"cuda:{visible_device_index}")
|
||||
else:
|
||||
init_backend = "gloo"
|
||||
device_id = None
|
||||
|
||||
@@ -86,6 +86,15 @@ class StatelessGroupCoordinator(GroupCoordinator):
|
||||
|
||||
self.rank = global_rank
|
||||
self.local_rank = local_rank
|
||||
from vllm.distributed.parallel_state import _WORLD
|
||||
|
||||
if _WORLD is not None:
|
||||
self.device_index = _WORLD.device_index
|
||||
else:
|
||||
assert local_rank >= 0, (
|
||||
"local_rank must be provided when creating the world group"
|
||||
)
|
||||
self.device_index = local_rank
|
||||
|
||||
self_device_group = None
|
||||
self_cpu_group = None
|
||||
@@ -152,11 +161,18 @@ class StatelessGroupCoordinator(GroupCoordinator):
|
||||
self.tcp_store_group = self_tcp_store_group
|
||||
|
||||
if current_platform.is_cuda_alike():
|
||||
self.device = torch.device(f"cuda:{local_rank}")
|
||||
visible_device_index = (
|
||||
current_platform.logical_device_id_to_visible_device_id(
|
||||
self.device_index
|
||||
)
|
||||
)
|
||||
self.device = torch.device(f"cuda:{visible_device_index}")
|
||||
elif current_platform.is_xpu():
|
||||
self.device = torch.device(f"xpu:{local_rank}")
|
||||
self.device = torch.device(f"xpu:{self.device_index}")
|
||||
elif current_platform.is_out_of_tree():
|
||||
self.device = torch.device(f"{current_platform.device_name}:{local_rank}")
|
||||
self.device = torch.device(
|
||||
f"{current_platform.device_name}:{self.device_index}"
|
||||
)
|
||||
else:
|
||||
self.device = torch.device("cpu")
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ import copy
|
||||
import dataclasses
|
||||
import functools
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from collections.abc import Callable
|
||||
from dataclasses import MISSING, asdict, dataclass, fields, is_dataclass
|
||||
@@ -465,6 +466,7 @@ class EngineArgs:
|
||||
numa_bind: bool = ParallelConfig.numa_bind
|
||||
numa_bind_nodes: list[int] | None = ParallelConfig.numa_bind_nodes
|
||||
numa_bind_cpus: list[str] | None = ParallelConfig.numa_bind_cpus
|
||||
device_ids: list[int | str] | None = None
|
||||
tensor_parallel_size: int = ParallelConfig.tensor_parallel_size
|
||||
prefill_context_parallel_size: int = ParallelConfig.prefill_context_parallel_size
|
||||
decode_context_parallel_size: int = ParallelConfig.decode_context_parallel_size
|
||||
@@ -979,6 +981,20 @@ class EngineArgs:
|
||||
parallel_group.add_argument(
|
||||
"--numa-bind-cpus", **parallel_kwargs["numa_bind_cpus"]
|
||||
)
|
||||
parallel_group.add_argument(
|
||||
"--device-ids",
|
||||
type=lambda s: [
|
||||
int(device_id) if device_id.isdigit() else device_id
|
||||
for device_id in (part.strip() for part in s.split(","))
|
||||
],
|
||||
default=None,
|
||||
help="Comma-separated physical GPU device IDs or UUIDs to use "
|
||||
'(e.g. --device-ids "2,3,5,7"). Avoids setting '
|
||||
"CUDA_VISIBLE_DEVICES, preserving full GPU topology "
|
||||
"visibility for GPU-NIC affinity and DeepGEMM. "
|
||||
"Note: has no effect with Ray executors; use Ray "
|
||||
"placement groups for GPU selection instead.",
|
||||
)
|
||||
parallel_group.add_argument(
|
||||
"--tensor-parallel-size", "-tp", **parallel_kwargs["tensor_parallel_size"]
|
||||
)
|
||||
@@ -1716,6 +1732,47 @@ class EngineArgs:
|
||||
)
|
||||
return SpeculativeConfig(**self.speculative_config)
|
||||
|
||||
def _resolve_device_ids(self) -> list[int] | None:
|
||||
if not self.device_ids:
|
||||
return None
|
||||
if self.distributed_executor_backend == "ray":
|
||||
logger.warning(
|
||||
"--device-ids has no effect when using the Ray executor. "
|
||||
"Use Ray placement groups for GPU selection instead."
|
||||
)
|
||||
ids = self.device_ids
|
||||
if len(set(ids)) != len(ids):
|
||||
raise ValueError(f"--device-ids must not contain duplicates: {ids}")
|
||||
if all(isinstance(i, str) for i in ids):
|
||||
return [
|
||||
current_platform.device_control_id_to_physical_device_id(i)
|
||||
for i in cast(list[str], ids)
|
||||
]
|
||||
if any(isinstance(i, str) for i in ids):
|
||||
raise ValueError("--device-ids must not mix integer IDs and UUIDs")
|
||||
int_ids = cast(list[int], ids)
|
||||
# Compose with CUDA_VISIBLE_DEVICES: if CVD is set, treat
|
||||
# --device-ids values as indices into the CVD-visible set.
|
||||
cvd = getattr(
|
||||
envs,
|
||||
current_platform.device_control_env_var,
|
||||
os.environ.get(current_platform.device_control_env_var),
|
||||
)
|
||||
if cvd:
|
||||
cvd_ids = [
|
||||
current_platform.device_control_id_to_physical_device_id(x)
|
||||
for x in cvd.split(",")
|
||||
]
|
||||
for i in int_ids:
|
||||
if i >= len(cvd_ids):
|
||||
raise ValueError(
|
||||
f"--device-ids index {i} is out of range for "
|
||||
f"{current_platform.device_control_env_var}"
|
||||
f"={cvd} ({len(cvd_ids)} devices visible)"
|
||||
)
|
||||
return [cvd_ids[i] for i in int_ids]
|
||||
return int_ids
|
||||
|
||||
def create_diffusion_config(self) -> DiffusionConfig | None:
|
||||
if self.diffusion_config is None:
|
||||
return None
|
||||
@@ -2029,6 +2086,7 @@ class EngineArgs:
|
||||
cp_kv_cache_interleave_size=self.cp_kv_cache_interleave_size,
|
||||
_api_process_count=self._api_process_count,
|
||||
_api_process_rank=self._api_process_rank,
|
||||
assigned_physical_gpu_ids=self._resolve_device_ids(),
|
||||
numa_bind=self.numa_bind,
|
||||
numa_bind_nodes=self.numa_bind_nodes,
|
||||
numa_bind_cpus=self.numa_bind_cpus,
|
||||
|
||||
@@ -43,6 +43,7 @@ from vllm.entrypoints.openai.engine.protocol import (
|
||||
JsonSchemaResponseFormat,
|
||||
ResponseFormat,
|
||||
StreamOptions,
|
||||
UsageInfo,
|
||||
)
|
||||
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
|
||||
from vllm.entrypoints.serve.utils.api_utils import sanitize_message
|
||||
@@ -54,6 +55,49 @@ if TYPE_CHECKING:
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _get_cached_tokens(usage: UsageInfo | None) -> int | None:
|
||||
"""Extract cached token count from OpenAI UsageInfo."""
|
||||
if usage is None or usage.prompt_tokens_details is None:
|
||||
return None
|
||||
return usage.prompt_tokens_details.cached_tokens
|
||||
|
||||
|
||||
def _build_anthropic_usage(
|
||||
prompt_tokens: int,
|
||||
completion_tokens: int | None,
|
||||
usage: UsageInfo | None,
|
||||
) -> AnthropicUsage:
|
||||
"""Build an AnthropicUsage from OpenAI-style token counts.
|
||||
|
||||
Anthropic defines ``total_input == input_tokens + cache_read +
|
||||
cache_creation``. vLLM's ``prompt_tokens`` is the total, so
|
||||
``input_tokens = prompt_tokens - cached_tokens``.
|
||||
|
||||
OpenAI usage only exposes ``cached_tokens`` (hits); there is no
|
||||
cache-creation analog, so ``cache_creation_input_tokens`` is ``0``
|
||||
when cache info is present. When cache info is absent (e.g.
|
||||
``--enable-prompt-tokens-details`` off, or a streaming chunk that
|
||||
hasn't carried it yet), cache fields are left **unset** so
|
||||
``exclude_unset=True`` serialization omits them entirely.
|
||||
|
||||
``completion_tokens`` follows ``UsageInfo`` and may be ``None`` on
|
||||
intermediate stream chunks; we coerce to ``0`` for the wire format.
|
||||
"""
|
||||
output_tokens = completion_tokens or 0
|
||||
cached = _get_cached_tokens(usage)
|
||||
if cached is not None:
|
||||
return AnthropicUsage(
|
||||
input_tokens=prompt_tokens - cached,
|
||||
output_tokens=output_tokens,
|
||||
cache_read_input_tokens=cached,
|
||||
cache_creation_input_tokens=0,
|
||||
)
|
||||
return AnthropicUsage(
|
||||
input_tokens=prompt_tokens,
|
||||
output_tokens=output_tokens,
|
||||
)
|
||||
|
||||
|
||||
def wrap_data_with_event(data: str, event: str):
|
||||
return f"event: {event}\ndata: {data}\n\n"
|
||||
|
||||
@@ -582,9 +626,10 @@ class AnthropicServingMessages(OpenAIServingChat):
|
||||
id=generator.id,
|
||||
content=[],
|
||||
model=generator.model,
|
||||
usage=AnthropicUsage(
|
||||
input_tokens=generator.usage.prompt_tokens,
|
||||
output_tokens=generator.usage.completion_tokens,
|
||||
usage=_build_anthropic_usage(
|
||||
generator.usage.prompt_tokens,
|
||||
generator.usage.completion_tokens,
|
||||
generator.usage,
|
||||
),
|
||||
kv_transfer_params=generator.kv_transfer_params,
|
||||
)
|
||||
@@ -765,11 +810,12 @@ class AnthropicServingMessages(OpenAIServingChat):
|
||||
model=origin_chunk.model,
|
||||
stop_reason=None,
|
||||
stop_sequence=None,
|
||||
usage=AnthropicUsage(
|
||||
input_tokens=origin_chunk.usage.prompt_tokens
|
||||
usage=_build_anthropic_usage(
|
||||
origin_chunk.usage.prompt_tokens
|
||||
if origin_chunk.usage
|
||||
else 0,
|
||||
output_tokens=0,
|
||||
0,
|
||||
origin_chunk.usage,
|
||||
),
|
||||
),
|
||||
)
|
||||
@@ -788,13 +834,14 @@ class AnthropicServingMessages(OpenAIServingChat):
|
||||
chunk = AnthropicStreamEvent(
|
||||
type="message_delta",
|
||||
delta=AnthropicDelta(stop_reason=stop_reason),
|
||||
usage=AnthropicUsage(
|
||||
input_tokens=origin_chunk.usage.prompt_tokens
|
||||
usage=_build_anthropic_usage(
|
||||
origin_chunk.usage.prompt_tokens
|
||||
if origin_chunk.usage
|
||||
else 0,
|
||||
output_tokens=origin_chunk.usage.completion_tokens
|
||||
origin_chunk.usage.completion_tokens
|
||||
if origin_chunk.usage
|
||||
else 0,
|
||||
origin_chunk.usage,
|
||||
),
|
||||
)
|
||||
data = chunk.model_dump_json(exclude_unset=True)
|
||||
|
||||
@@ -898,12 +898,6 @@ class LLM(BeamSearchOfflineMixin, PoolingOfflineMixin, OfflineInferenceMixin):
|
||||
def finish_weight_update(self) -> None:
|
||||
"""Finish the current weight update."""
|
||||
self.llm_engine.collective_rpc("finish_weight_update")
|
||||
# Invalidate cached state computed with the old weights so it isn't
|
||||
# reused for subsequent requests:
|
||||
# - prefix cache: KV blocks computed with the old weights
|
||||
# - encoder cache: multimodal embeddings keyed only by mm_hash
|
||||
self.llm_engine.reset_prefix_cache()
|
||||
self.llm_engine.reset_encoder_cache()
|
||||
|
||||
def __repr__(self) -> str:
|
||||
"""Return a transformers-style hierarchical view of the model."""
|
||||
|
||||
@@ -455,7 +455,7 @@ async def init_render_app_state(
|
||||
enable_auto_tools=args.enable_auto_tool_choice,
|
||||
exclude_tools_when_tool_choice_none=args.exclude_tools_when_tool_choice_none,
|
||||
tool_parser=args.tool_call_parser,
|
||||
reasoning_parser=args.structured_outputs_config.reasoning_parser,
|
||||
reasoning_parser=args.reasoning_parser,
|
||||
default_chat_template_kwargs=args.default_chat_template_kwargs,
|
||||
log_error_stack=args.log_error_stack,
|
||||
)
|
||||
|
||||
@@ -23,12 +23,10 @@ import uvloop
|
||||
from fastapi import FastAPI, Response
|
||||
|
||||
from vllm.logger import init_logger
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.utils.system_utils import (
|
||||
decorate_logs,
|
||||
kill_process_tree,
|
||||
set_process_title,
|
||||
update_environment_variables,
|
||||
)
|
||||
|
||||
logger = init_logger(__name__)
|
||||
@@ -127,22 +125,29 @@ def _build_vllm_dp_server_args(
|
||||
child_args.data_parallel_multi_port_external_lb = False
|
||||
child_args.data_parallel_supervisor_port = None
|
||||
child_args.api_server_count = 1
|
||||
child_args.device_ids = _build_device_ids(args, local_rank)
|
||||
return child_args
|
||||
|
||||
|
||||
def _build_vllm_dp_server_env(
|
||||
args: argparse.Namespace, local_rank: int
|
||||
) -> dict[str, str]:
|
||||
# set visible devices for the child process
|
||||
def _build_device_ids(args: argparse.Namespace, local_rank: int) -> list[int | str]:
|
||||
"""Build the --device-ids value for a DP child process.
|
||||
|
||||
The child resolves these against its own inherited device-control env
|
||||
var (e.g. CUDA_VISIBLE_DEVICES), so integer IDs must stay env-relative
|
||||
here rather than being translated to physical IDs.
|
||||
"""
|
||||
devices_per_rank = args.tensor_parallel_size * args.pipeline_parallel_size
|
||||
start = local_rank * devices_per_rank
|
||||
stop = start + devices_per_rank
|
||||
device_env = current_platform.device_control_env_var
|
||||
visible_devices = ",".join(
|
||||
str(current_platform.device_id_to_physical_device_id(idx))
|
||||
for idx in range(start, stop)
|
||||
)
|
||||
return {device_env: visible_devices}
|
||||
device_ids = getattr(args, "device_ids", None)
|
||||
if device_ids is not None:
|
||||
if stop > len(device_ids):
|
||||
raise ValueError(
|
||||
f"--device-ids has {len(device_ids)} entries, but DP rank "
|
||||
f"{local_rank} needs devices [{start}, {stop})"
|
||||
)
|
||||
return device_ids[start:stop]
|
||||
return list(range(start, stop))
|
||||
|
||||
|
||||
def _child_base_url(args: argparse.Namespace, port: int) -> str:
|
||||
@@ -228,9 +233,7 @@ def _build_dp_supervisor_app(supervisor: DPSupervisor) -> FastAPI:
|
||||
return app
|
||||
|
||||
|
||||
def _run_vllm_dp_server(
|
||||
child_args: argparse.Namespace, env_updates: dict[str, str]
|
||||
) -> None:
|
||||
def _run_vllm_dp_server(child_args: argparse.Namespace) -> None:
|
||||
"""
|
||||
Entrypoint function for the vLLM DP Server.
|
||||
"""
|
||||
@@ -241,7 +244,6 @@ def _run_vllm_dp_server(
|
||||
os.setpgrp()
|
||||
|
||||
name = f"APIServer_DP{child_args.data_parallel_rank}"
|
||||
update_environment_variables(env_updates)
|
||||
set_process_title(name)
|
||||
decorate_logs(name)
|
||||
uvloop.run(run_server(child_args))
|
||||
@@ -345,11 +347,10 @@ class DPSupervisor:
|
||||
context = multiprocessing.get_context("spawn")
|
||||
for local_rank in range(self.args.data_parallel_size_local):
|
||||
child_args = _build_vllm_dp_server_args(self.args, local_rank)
|
||||
child_env = _build_vllm_dp_server_env(self.args, local_rank)
|
||||
process = context.Process(
|
||||
target=_run_vllm_dp_server,
|
||||
name=f"APIServer_DPRank_{child_args.data_parallel_rank}",
|
||||
args=(child_args, child_env),
|
||||
args=(child_args,),
|
||||
)
|
||||
process.start()
|
||||
self._processes.append(process)
|
||||
|
||||
@@ -219,10 +219,14 @@ class GenerateResponse(BaseModel):
|
||||
|
||||
|
||||
class DerenderChatRequest(BaseModel):
|
||||
"""Request for the /v1/chat/completions/derender endpoint.
|
||||
"""Request for the /v1/chat/completions/derender endpoint (non-streaming).
|
||||
|
||||
Wraps a GenerateResponse and caller-supplied metadata needed to produce
|
||||
a fully-formed ChatCompletionResponse without a GPU.
|
||||
Wraps a complete GenerateResponse and caller-supplied metadata needed to
|
||||
produce a fully-formed ChatCompletionResponse without a GPU.
|
||||
|
||||
Streaming derender would require a separate endpoint design with
|
||||
incremental token delivery, ``OutputProcessor``-based detokenization,
|
||||
and ``parser.parse_delta()`` instead of ``parser.parse()``.
|
||||
"""
|
||||
|
||||
model: str
|
||||
@@ -244,7 +248,7 @@ class DerenderChatRequest(BaseModel):
|
||||
|
||||
|
||||
class DerenderCompletionRequest(BaseModel):
|
||||
"""Request for the /v1/completions/derender endpoint.
|
||||
"""Request for the /v1/completions/derender endpoint (non-streaming).
|
||||
|
||||
Parallel to DerenderChatRequest but handles the multi-prompt completions
|
||||
case: one GenerateResponse per prompt, mirroring the list[GenerateRequest]
|
||||
|
||||
@@ -27,6 +27,7 @@ from vllm.entrypoints.openai.completion.protocol import (
|
||||
)
|
||||
from vllm.entrypoints.openai.engine.protocol import (
|
||||
ErrorResponse,
|
||||
ToolCall,
|
||||
UsageInfo,
|
||||
)
|
||||
from vllm.entrypoints.openai.engine.serving import resolve_token_id_placeholder
|
||||
@@ -43,7 +44,6 @@ from vllm.entrypoints.serve.disagg.protocol import (
|
||||
DerenderChatRequest,
|
||||
DerenderCompletionRequest,
|
||||
GenerateRequest,
|
||||
GenerateResponseChoice,
|
||||
MultiModalFeatures,
|
||||
PlaceholderRangeInfo,
|
||||
)
|
||||
@@ -76,21 +76,83 @@ from vllm.utils.mistral import mt as _mt
|
||||
logger = init_logger(__name__)
|
||||
|
||||
|
||||
def _parse_token_id_placeholder(token: str) -> int | None:
|
||||
"""Extract token ID from a 'token_id:N' placeholder string."""
|
||||
if not token.startswith("token_id:"):
|
||||
return None
|
||||
try:
|
||||
return int(token[len("token_id:") :])
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def _correct_decoded_token(
|
||||
token_id: int, context_token_ids: list[int], tokenizer: TokenizerLike
|
||||
) -> str:
|
||||
"""Use preceding tokens as context to fix U+FFFD from byte-fallback.
|
||||
|
||||
Mirrors LogprobsProcessor._correct_decoded_token in v1/engine/logprobs.py.
|
||||
"""
|
||||
max_ctx = min(len(context_token_ids), 4)
|
||||
|
||||
for num_ctx in range(1, max_ctx + 1):
|
||||
context = context_token_ids[-num_ctx:]
|
||||
full_decoded = tokenizer.decode(context + [token_id])
|
||||
|
||||
if full_decoded.endswith("�"):
|
||||
continue
|
||||
|
||||
clean_end = len(context)
|
||||
for j in range(len(context) - 1, -1, -1):
|
||||
if tokenizer.decode([context[j]]).endswith("�"):
|
||||
clean_end = j
|
||||
else:
|
||||
break
|
||||
|
||||
clean_prefix = tokenizer.decode(context[:clean_end]) if clean_end > 0 else ""
|
||||
|
||||
if full_decoded.startswith(clean_prefix):
|
||||
return full_decoded[len(clean_prefix) :]
|
||||
|
||||
common_len = 0
|
||||
for a, b in zip(clean_prefix, full_decoded):
|
||||
if a != b:
|
||||
break
|
||||
common_len += 1
|
||||
return full_decoded[common_len:]
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def _resolve_logprobs(
|
||||
logprobs: ChatCompletionLogProbs, tokenizer: TokenizerLike
|
||||
) -> ChatCompletionLogProbs:
|
||||
"""Resolve all token_id:N placeholders in a ChatCompletionLogProbs object."""
|
||||
"""Resolve token_id:N placeholders in a ChatCompletionLogProbs object."""
|
||||
if logprobs.content is None:
|
||||
return logprobs
|
||||
|
||||
context_token_ids: list[int] = []
|
||||
resolved_content = []
|
||||
|
||||
for entry in logprobs.content:
|
||||
token_str, token_bytes = resolve_token_id_placeholder(entry.token, tokenizer)
|
||||
sampled_id = _parse_token_id_placeholder(entry.token)
|
||||
|
||||
if token_str.endswith("�") and sampled_id is not None:
|
||||
token_str = _correct_decoded_token(sampled_id, context_token_ids, tokenizer)
|
||||
token_bytes = list(token_str.encode("utf-8"))
|
||||
|
||||
resolved_top = []
|
||||
for top in entry.top_logprobs:
|
||||
top_str, top_bytes = resolve_token_id_placeholder(top.token, tokenizer)
|
||||
top_id = _parse_token_id_placeholder(top.token)
|
||||
if top_str.endswith("�") and top_id is not None:
|
||||
top_str = _correct_decoded_token(top_id, context_token_ids, tokenizer)
|
||||
top_bytes = list(top_str.encode("utf-8"))
|
||||
resolved_top.append(
|
||||
top.model_copy(update={"token": top_str, "bytes": top_bytes})
|
||||
)
|
||||
|
||||
resolved_content.append(
|
||||
entry.model_copy(
|
||||
update={
|
||||
@@ -100,6 +162,10 @@ def _resolve_logprobs(
|
||||
}
|
||||
)
|
||||
)
|
||||
|
||||
if sampled_id is not None:
|
||||
context_token_ids.append(sampled_id)
|
||||
|
||||
return ChatCompletionLogProbs(content=resolved_content)
|
||||
|
||||
|
||||
@@ -136,30 +202,6 @@ def _convert_chat_logprobs_to_completion_logprobs(
|
||||
)
|
||||
|
||||
|
||||
def _build_chat_choice(
|
||||
choice: GenerateResponseChoice, tokenizer: TokenizerLike
|
||||
) -> ChatCompletionResponseChoice:
|
||||
"""Detokenize and resolve logprobs for a single GenerateResponseChoice.
|
||||
|
||||
Raises:
|
||||
ValueError: if choice.token_ids is empty or None.
|
||||
"""
|
||||
if not choice.token_ids:
|
||||
raise ValueError(f"choice {choice.index} has empty or null token_ids")
|
||||
decoded_text = tokenizer.decode(choice.token_ids, skip_special_tokens=True)
|
||||
resolved_logprobs = (
|
||||
_resolve_logprobs(choice.logprobs, tokenizer)
|
||||
if choice.logprobs is not None
|
||||
else None
|
||||
)
|
||||
return ChatCompletionResponseChoice(
|
||||
index=choice.index,
|
||||
message=ChatMessage(role="assistant", content=decoded_text),
|
||||
logprobs=resolved_logprobs,
|
||||
finish_reason=choice.finish_reason,
|
||||
)
|
||||
|
||||
|
||||
class OpenAIServingRender:
|
||||
def __init__(
|
||||
self,
|
||||
@@ -536,9 +578,12 @@ class OpenAIServingRender:
|
||||
) -> ChatCompletionResponse | ErrorResponse:
|
||||
"""Postprocess a GenerateResponse into a ChatCompletionResponse.
|
||||
|
||||
This is the symmetric inverse of render_chat_request: it detokenizes
|
||||
output token IDs, resolves token_id:N logprob placeholders, and
|
||||
formats the result as an OpenAI-compatible chat completion response.
|
||||
Non-streaming only: expects the complete GenerateResponse with all
|
||||
token IDs present. Uses ``parser.parse()`` for one-shot extraction.
|
||||
|
||||
When ``request.chat_request`` is provided, the parser splits the
|
||||
output into (reasoning, content, tool_calls). Otherwise falls
|
||||
back to plain detokenization.
|
||||
"""
|
||||
error_check_ret = await self._check_model(request)
|
||||
if error_check_ret is not None:
|
||||
@@ -546,11 +591,89 @@ class OpenAIServingRender:
|
||||
|
||||
tokenizer = self.renderer.get_tokenizer()
|
||||
gen = request.generate_response
|
||||
chat_request = request.chat_request
|
||||
choices: list[ChatCompletionResponseChoice] = []
|
||||
|
||||
try:
|
||||
for choice in gen.choices:
|
||||
choices.append(_build_chat_choice(choice, tokenizer))
|
||||
if not choice.token_ids:
|
||||
raise ValueError(
|
||||
f"choice {choice.index} has empty or null token_ids"
|
||||
)
|
||||
|
||||
resolved_logprobs = (
|
||||
_resolve_logprobs(choice.logprobs, tokenizer)
|
||||
if choice.logprobs is not None
|
||||
else None
|
||||
)
|
||||
|
||||
if self.parser is not None and chat_request is not None:
|
||||
# Parser path: decode with special tokens preserved
|
||||
# so the parser can see markers like </think>,
|
||||
# <tool_call>, or Harmony channel tokens.
|
||||
decoded_text = tokenizer.decode(
|
||||
choice.token_ids, skip_special_tokens=False
|
||||
)
|
||||
|
||||
chat_template_kwargs: dict[str, Any] = {}
|
||||
if not self.use_harmony:
|
||||
chat_template_kwargs = (
|
||||
chat_request.build_chat_params(
|
||||
self.chat_template,
|
||||
self.chat_template_content_format,
|
||||
)
|
||||
.with_defaults(self.default_chat_template_kwargs)
|
||||
.chat_template_kwargs
|
||||
)
|
||||
|
||||
parser = self.parser(
|
||||
tokenizer,
|
||||
chat_request.tools,
|
||||
chat_template_kwargs=chat_template_kwargs,
|
||||
)
|
||||
reasoning, content, tool_calls = parser.parse(
|
||||
decoded_text,
|
||||
chat_request,
|
||||
enable_auto_tools=self.enable_auto_tools,
|
||||
model_output_token_ids=choice.token_ids,
|
||||
)
|
||||
|
||||
if not getattr(chat_request, "include_reasoning", True):
|
||||
reasoning = None
|
||||
|
||||
tc_items = (
|
||||
[
|
||||
ToolCall(
|
||||
id=random_uuid(),
|
||||
function=tc,
|
||||
)
|
||||
for tc in tool_calls
|
||||
]
|
||||
if tool_calls
|
||||
else []
|
||||
)
|
||||
|
||||
message = ChatMessage(
|
||||
role="assistant",
|
||||
reasoning=reasoning,
|
||||
content=content,
|
||||
tool_calls=tc_items,
|
||||
)
|
||||
else:
|
||||
# No parser: plain detokenization.
|
||||
decoded_text = tokenizer.decode(
|
||||
choice.token_ids, skip_special_tokens=True
|
||||
)
|
||||
message = ChatMessage(role="assistant", content=decoded_text)
|
||||
|
||||
choices.append(
|
||||
ChatCompletionResponseChoice(
|
||||
index=choice.index,
|
||||
message=message,
|
||||
logprobs=resolved_logprobs,
|
||||
finish_reason=choice.finish_reason,
|
||||
)
|
||||
)
|
||||
except ValueError as exc:
|
||||
return self.create_error_response(str(exc))
|
||||
|
||||
@@ -587,8 +710,9 @@ class OpenAIServingRender:
|
||||
) -> CompletionResponse | ErrorResponse:
|
||||
"""Postprocess a list of GenerateResponses into a CompletionResponse.
|
||||
|
||||
Mirrors the multi-prompt completions case: one GenerateResponse per
|
||||
prompt, parallel to the list[GenerateRequest] from /v1/completions/render.
|
||||
Non-streaming only. Mirrors the multi-prompt completions case: one
|
||||
GenerateResponse per prompt, parallel to the list[GenerateRequest]
|
||||
from /v1/completions/render.
|
||||
"""
|
||||
error_check_ret = await self._check_model(request)
|
||||
if error_check_ret is not None:
|
||||
|
||||
+7
-1
@@ -209,6 +209,7 @@ if TYPE_CHECKING:
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: int = 300
|
||||
VLLM_WORKER_SHUTDOWN_TIMEOUT_SECONDS: int = 5
|
||||
VLLM_KV_CACHE_LAYOUT: Literal["NHD", "HND"] | None = None
|
||||
VLLM_USE_PACKED_HMA_KV_CACHE: bool = False
|
||||
VLLM_SSM_CONV_STATE_LAYOUT: Literal["SD", "DS"] | None = None
|
||||
VLLM_COMPUTE_NANS_IN_LOGITS: bool = False
|
||||
VLLM_ROCM_QUICK_REDUCE_QUANTIZATION: Literal[
|
||||
@@ -485,7 +486,7 @@ def get_vllm_port() -> int | None:
|
||||
raise ValueError(
|
||||
f"VLLM_PORT '{port}' appears to be a URI. "
|
||||
"This may be caused by a Kubernetes service discovery issue,"
|
||||
"check the warning in: https://docs.vllm.ai/en/stable/serving/env_vars.html"
|
||||
"check the warning in: https://docs.vllm.ai/en/latest/configuration/env_vars.html"
|
||||
) from None
|
||||
raise ValueError(f"VLLM_PORT '{port}' must be a valid integer") from err
|
||||
|
||||
@@ -1608,6 +1609,11 @@ environment_variables: dict[str, Callable[[], Any]] = {
|
||||
"VLLM_KV_CACHE_LAYOUT": env_with_choices(
|
||||
"VLLM_KV_CACHE_LAYOUT", None, ["NHD", "HND"]
|
||||
),
|
||||
# Opt into packed per-block KV cache allocation for multi-group
|
||||
# attention-only HMA models (e.g. gpt-oss, Gemma 3/4).
|
||||
"VLLM_USE_PACKED_HMA_KV_CACHE": lambda: bool(
|
||||
int(os.getenv("VLLM_USE_PACKED_HMA_KV_CACHE", "0"))
|
||||
),
|
||||
# SSM conv state layout used for Mamba models.
|
||||
# - SD: (state_len, dim) — dim contiguous (default)
|
||||
# - DS: (dim, state_len) — TP-sharded dim on dim1,
|
||||
|
||||
@@ -17,7 +17,7 @@ from vllm.lora.utils import (
|
||||
)
|
||||
from vllm.model_executor.model_loader.tensorizer import TensorizerConfig
|
||||
from vllm.model_executor.models.utils import WeightsMapper
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
@@ -126,7 +126,7 @@ class LoRAModel:
|
||||
skip_prefixes: list[str] | None = None,
|
||||
) -> "LoRAModel":
|
||||
"""Create a LoRAModel from a dictionary of tensors."""
|
||||
pin_memory = str(device) == "cpu" and is_pin_memory_available()
|
||||
pin_memory = str(device) == "cpu" and PIN_MEMORY
|
||||
loras: dict[str, LoRALayerWeights] = {}
|
||||
for tensor_name, tensor in tensors.items():
|
||||
if is_base_embedding_weights(tensor_name):
|
||||
|
||||
@@ -7,7 +7,7 @@ import torch
|
||||
import torch.types
|
||||
|
||||
from vllm.lora.peft_helper import PEFTHelper
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
|
||||
class LoRALayerWeights:
|
||||
@@ -79,7 +79,7 @@ class LoRALayerWeights:
|
||||
dtype: torch.dtype,
|
||||
device: torch.types.Device,
|
||||
) -> "LoRALayerWeights":
|
||||
pin_memory = str(device) == "cpu" and is_pin_memory_available()
|
||||
pin_memory = str(device) == "cpu" and PIN_MEMORY
|
||||
lora_a = torch.zeros(
|
||||
[rank, input_dim], dtype=dtype, device=device, pin_memory=pin_memory
|
||||
)
|
||||
|
||||
@@ -42,7 +42,7 @@ from vllm.model_executor.models.utils import PPMissingLayer
|
||||
from vllm.multimodal import MULTIMODAL_REGISTRY
|
||||
from vllm.multimodal.encoder_budget import MultiModalBudget
|
||||
from vllm.utils.cache import LRUCache
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
@@ -801,7 +801,7 @@ class LoRAModelManager:
|
||||
# 2. The weight packing above (e.g., pack_moe) may invalidate the
|
||||
# pin_memory allocation, so we execute it after packing.
|
||||
|
||||
pin_memory = str(lora_device) == "cpu" and is_pin_memory_available()
|
||||
pin_memory = str(lora_device) == "cpu" and PIN_MEMORY
|
||||
if pin_memory:
|
||||
for lora in lora_model.loras.values():
|
||||
if isinstance(lora.lora_a, list):
|
||||
|
||||
@@ -1684,12 +1684,13 @@ class MLACommonMetadataBuilder(AttentionMetadataBuilder[M]):
|
||||
# [[0, 0, 0, 0], [256, 256, 256, 256], [512, 512, 512, 512]]
|
||||
# Note(simon): this is done in CPU because of downstream's
|
||||
# of `to_list`.
|
||||
chunk_starts = (
|
||||
chunk_starts = torch.empty(
|
||||
num_chunks, num_prefills, dtype=torch.int32, pin_memory=True
|
||||
).copy_(
|
||||
torch.arange(num_chunks, dtype=torch.int32)
|
||||
.multiply_(max_context_chunk)
|
||||
.unsqueeze(1)
|
||||
.expand(-1, num_prefills)
|
||||
* max_context_chunk
|
||||
).pin_memory()
|
||||
)
|
||||
chunk_ends = torch.min(
|
||||
context_lens_cpu.unsqueeze(0), chunk_starts + max_context_chunk
|
||||
)
|
||||
@@ -1746,12 +1747,13 @@ class MLACommonMetadataBuilder(AttentionMetadataBuilder[M]):
|
||||
)
|
||||
* self.dcp_local_block_size
|
||||
)
|
||||
local_chunk_starts = (
|
||||
local_chunk_starts = torch.empty(
|
||||
num_chunks, num_prefills, dtype=torch.int32, pin_memory=True
|
||||
).copy_(
|
||||
torch.arange(num_chunks, dtype=torch.int32)
|
||||
.multiply_(padded_local_max_context_chunk_across_ranks)
|
||||
.unsqueeze(1)
|
||||
.expand(-1, num_prefills)
|
||||
* padded_local_max_context_chunk_across_ranks
|
||||
).pin_memory()
|
||||
)
|
||||
local_chunk_ends = torch.min(
|
||||
padded_local_context_lens_cpu.unsqueeze(0),
|
||||
local_chunk_starts
|
||||
|
||||
@@ -28,6 +28,7 @@ from vllm.utils.flashinfer import (
|
||||
is_flashinfer_cudnn_fp8_prefill_attn_supported,
|
||||
)
|
||||
from vllm.utils.math_utils import round_up
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.v1.attention.backends.fa_utils import get_flash_attn_version
|
||||
from vllm.v1.attention.backends.registry import AttentionBackendEnum
|
||||
from vllm.v1.attention.ops.vit_attn_wrappers import (
|
||||
@@ -311,7 +312,7 @@ class MMEncoderAttention(CustomOp):
|
||||
)
|
||||
cu_seqlens = np.concatenate([cu_seqlens_qko, cu_seqlens_v])
|
||||
|
||||
cu_seqlens = torch.from_numpy(cu_seqlens).to(device, non_blocking=True)
|
||||
cu_seqlens = async_tensor_h2d(cu_seqlens, device=device)
|
||||
return cu_seqlens
|
||||
|
||||
def __init__(
|
||||
|
||||
@@ -21,7 +21,6 @@ from vllm.model_executor.layers.fused_moe.config import (
|
||||
FusedMoEQuantConfig,
|
||||
)
|
||||
from vllm.model_executor.layers.fused_moe.experts.triton_moe import TritonExperts
|
||||
from vllm.model_executor.layers.fused_moe.utils import moe_kernel_quantize_input
|
||||
from vllm.model_executor.layers.quantization.utils.nvfp4_emulation_utils import (
|
||||
dequantize_to_dtype,
|
||||
)
|
||||
@@ -135,14 +134,6 @@ class Nvfp4QuantizationEmulationTritonExperts(TritonExperts):
|
||||
swizzle=False,
|
||||
)
|
||||
|
||||
hidden_states, _ = moe_kernel_quantize_input(
|
||||
A=hidden_states,
|
||||
A_scale=self.quant_config.a1_gscale,
|
||||
quant_dtype="nvfp4",
|
||||
per_act_token_quant=False,
|
||||
quantization_emulation=True,
|
||||
)
|
||||
|
||||
# Activation quantization/dequantization is deferred to
|
||||
# `moe_kernel_quantize_input` in TritonExperts.apply.
|
||||
super().apply(
|
||||
|
||||
@@ -21,7 +21,6 @@ from vllm.model_executor.layers.fused_moe.config import (
|
||||
FusedMoEQuantConfig,
|
||||
)
|
||||
from vllm.model_executor.layers.fused_moe.experts.triton_moe import TritonExperts
|
||||
from vllm.model_executor.layers.fused_moe.utils import moe_kernel_quantize_input
|
||||
from vllm.model_executor.layers.quantization.utils.mxfp4_utils import dequant_mxfp4
|
||||
from vllm.model_executor.layers.quantization.utils.mxfp6_utils import dequant_mxfp6
|
||||
from vllm.model_executor.layers.quantization.utils.ocp_mx_utils import (
|
||||
@@ -155,16 +154,6 @@ class OCP_MXQuantizationEmulationTritonExperts(TritonExperts):
|
||||
w2, self.w2_scale_val, hidden_states.dtype
|
||||
)
|
||||
|
||||
# Apply activation QDQ if needed by the OCP MX scheme
|
||||
hidden_states, _ = moe_kernel_quantize_input(
|
||||
A=hidden_states,
|
||||
A_scale=None,
|
||||
quant_dtype=self.quant_config.quant_dtype,
|
||||
per_act_token_quant=False,
|
||||
ocp_mx_scheme=self.ocp_mx_scheme,
|
||||
quantization_emulation=True,
|
||||
)
|
||||
|
||||
# Activation quantization/dequantization is deferred to
|
||||
# `moe_kernel_quantize_input` in TritonExperts.apply.
|
||||
super().apply(
|
||||
|
||||
@@ -245,7 +245,7 @@ class TritonExperts(LoRAExpertsMixin, mk.FusedMoEExpertsModular):
|
||||
lora_unquantized_hidden_states = hidden_states
|
||||
hidden_states, a1q_scale = moe_kernel_quantize_input(
|
||||
hidden_states,
|
||||
self.a1_scale,
|
||||
self.a1_scale or self.a1_gscale,
|
||||
self.quant_dtype,
|
||||
self.per_act_token_quant,
|
||||
self.block_shape,
|
||||
|
||||
@@ -296,6 +296,7 @@ def moe_kernel_quantize_input(
|
||||
if not quantization_emulation:
|
||||
return _nvfp4_quantize(A, A_scale, is_sf_swizzled_layout=is_scale_swizzled)
|
||||
else:
|
||||
assert A_scale is not None
|
||||
A = ref_nvfp4_quant_dequant(A, A_scale, block_size=16)
|
||||
return A, None
|
||||
elif quant_dtype == "mxfp4":
|
||||
|
||||
@@ -10,6 +10,7 @@ import torch.nn as nn
|
||||
from vllm.config.pooler import SequencePoolingType
|
||||
from vllm.model_executor.layers.pooler import PoolingParamsUpdate
|
||||
from vllm.tasks import PoolingTask
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.v1.pool.metadata import PoolingMetadata
|
||||
|
||||
SequencePoolingMethodOutput: TypeAlias = torch.Tensor | list[torch.Tensor]
|
||||
@@ -74,15 +75,14 @@ class MeanPool(SequencePoolingMethod):
|
||||
# early return for empty batch
|
||||
return hidden_states.new_empty((0, hidden_size), dtype=torch.float32)
|
||||
|
||||
# Build segment_ids on CPU so repeat_interleave doesn't need to sync
|
||||
# GPU->CPU to learn its data-dependent output length, then upload
|
||||
# non-blocking. eg. [2, 1, 3] -> [0, 0, 1, 2, 2, 2]
|
||||
prompt_lens = async_tensor_h2d(
|
||||
prompt_lens_cpu, device=hidden_states.device, dtype=torch.int64
|
||||
)
|
||||
# eg. [2, 1, 3] -> [0, 0, 1, 2, 2, 2]
|
||||
segment_ids = torch.repeat_interleave(
|
||||
torch.arange(num_seqs, dtype=torch.long),
|
||||
prompt_lens_cpu,
|
||||
).to(hidden_states.device, non_blocking=True)
|
||||
prompt_lens = prompt_lens_cpu.to(
|
||||
hidden_states.device, dtype=torch.int64, non_blocking=True
|
||||
torch.arange(num_seqs, device=hidden_states.device, dtype=torch.long),
|
||||
prompt_lens,
|
||||
output_size=int(prompt_lens_cpu.sum()),
|
||||
)
|
||||
segment_sums = torch.zeros(
|
||||
(num_seqs, hidden_size),
|
||||
|
||||
@@ -1001,24 +1001,30 @@ class DeepseekV2MLAAttention(nn.Module):
|
||||
# IndexCache config
|
||||
# Refer: https://arxiv.org/abs/2603.12201 for more details.
|
||||
_skip_topk = False
|
||||
_index_topk_freq = getattr(config, "index_topk_freq", 1)
|
||||
_index_topk_pattern = getattr(config, "index_topk_pattern", None)
|
||||
_index_skip_topk_offset = getattr(config, "index_skip_topk_offset", 2)
|
||||
layer_id = extract_layer_index(prefix)
|
||||
is_mtp_layer = False
|
||||
if self.is_v32:
|
||||
_index_topk_freq = getattr(config, "index_topk_freq", 1)
|
||||
_index_topk_pattern = getattr(config, "index_topk_pattern", None)
|
||||
_index_skip_topk_offset = getattr(config, "index_skip_topk_offset", 2)
|
||||
layer_id = extract_layer_index(prefix)
|
||||
|
||||
if _index_topk_pattern is None:
|
||||
_skip_topk = (
|
||||
max(layer_id - _index_skip_topk_offset + 1, 0) % _index_topk_freq != 0
|
||||
if _index_topk_pattern is None:
|
||||
_skip_topk = (
|
||||
max(layer_id - _index_skip_topk_offset + 1, 0) % _index_topk_freq
|
||||
!= 0
|
||||
)
|
||||
elif 0 <= layer_id < len(_index_topk_pattern):
|
||||
_skip_topk = _index_topk_pattern[layer_id] == "S"
|
||||
|
||||
# The skip pattern only governs backbone layers. MTP/nextn
|
||||
# layers (layer_id >= num_hidden_layers) always build a full
|
||||
# indexer: they compute indices at draft step 0 and toggle
|
||||
# at runtime via set_skip_topk
|
||||
# (index_share_for_mtp_iteration).
|
||||
_num_hidden_layers = getattr(config, "num_hidden_layers", None)
|
||||
is_mtp_layer = (
|
||||
_num_hidden_layers is not None and layer_id >= _num_hidden_layers
|
||||
)
|
||||
elif 0 <= layer_id < len(_index_topk_pattern):
|
||||
_skip_topk = _index_topk_pattern[layer_id] == "S"
|
||||
|
||||
# The skip pattern only governs backbone layers. MTP/nextn layers
|
||||
# (layer_id >= num_hidden_layers) always build a full indexer: they
|
||||
# compute indices at draft step 0 and toggle at runtime via
|
||||
# set_skip_topk (index_share_for_mtp_iteration).
|
||||
_num_hidden_layers = getattr(config, "num_hidden_layers", None)
|
||||
is_mtp_layer = _num_hidden_layers is not None and layer_id >= _num_hidden_layers
|
||||
|
||||
if self.is_v32 and (not _skip_topk or is_mtp_layer):
|
||||
self.indexer_rope_emb = get_rope(
|
||||
|
||||
@@ -66,6 +66,7 @@ from vllm.model_executor.models.utils import maybe_prefix
|
||||
from vllm.model_executor.models.vision import is_vit_use_data_parallel
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.transformers_utils.configs.moonvit import MoonViTConfig
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
|
||||
|
||||
def _apply_rope_input_validation(x, freqs_cis):
|
||||
@@ -758,7 +759,7 @@ class MoonVitPretrainedModel(PreTrainedModel):
|
||||
),
|
||||
]
|
||||
)
|
||||
metadata["cu_seqlens"] = torch.from_numpy(cu_seqlens_np).to(device)
|
||||
metadata["cu_seqlens"] = async_tensor_h2d(cu_seqlens_np, device=device)
|
||||
|
||||
if max_seqlen_override is not None:
|
||||
max_seqlen_val = int(max_seqlen_override)
|
||||
@@ -770,7 +771,7 @@ class MoonVitPretrainedModel(PreTrainedModel):
|
||||
metadata["max_seqlen"] = torch.tensor(max_seqlen_val, dtype=torch.int32)
|
||||
|
||||
gather_idx_np = _build_merge_gather_idx(grid_pairs, self.merge_kernel_size)
|
||||
metadata["merge_gather_idx"] = torch.from_numpy(gather_idx_np).to(device)
|
||||
metadata["merge_gather_idx"] = async_tensor_h2d(gather_idx_np, device=device)
|
||||
|
||||
return metadata
|
||||
|
||||
|
||||
@@ -83,9 +83,8 @@ from vllm.multimodal.parse import MultiModalDataItems
|
||||
from vllm.multimodal.processing import PromptReplacement, PromptUpdate
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.sequence import IntermediateTensors
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.tensor_schema import TensorSchema, TensorShape
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.utils.torch_utils import PIN_MEMORY, async_tensor_h2d
|
||||
from vllm.v1.attention.backends.registry import AttentionBackendEnum
|
||||
from vllm.v1.worker.encoder_cudagraph_defs import EncoderCudaGraphReplayBuffers
|
||||
|
||||
@@ -825,7 +824,7 @@ class Qwen2_5_VisionTransformer(nn.Module):
|
||||
@staticmethod
|
||||
def invert_permutation(perm: torch.Tensor) -> torch.Tensor:
|
||||
# building the inverse permutation in O(n) time
|
||||
inv = torch.empty_like(perm, pin_memory=is_pin_memory_available())
|
||||
inv = torch.empty_like(perm, pin_memory=PIN_MEMORY)
|
||||
inv[perm] = torch.arange(perm.numel(), device=perm.device, dtype=perm.dtype)
|
||||
return inv
|
||||
|
||||
|
||||
@@ -1202,6 +1202,49 @@ class Qwen3VLDummyInputsBuilder(BaseDummyInputsBuilder[Qwen3VLProcessingInfo]):
|
||||
return video_items
|
||||
|
||||
|
||||
def _replace_video_token_placeholders(
|
||||
prompt_ids: list[int],
|
||||
target: list[int],
|
||||
replacements: list[list[int]],
|
||||
) -> list[int]:
|
||||
"""Replace each 3-token video placeholder with its expanded sequence.
|
||||
|
||||
Args:
|
||||
prompt_ids: Token IDs of the original (unexpanded) prompt.
|
||||
target: 3-element list ``[vision_start_id, video_pad_id,
|
||||
vision_end_id]`` to search for.
|
||||
replacements: Per-video expanded token sequences, in prompt order.
|
||||
|
||||
Returns:
|
||||
Token IDs with every placeholder triplet replaced.
|
||||
"""
|
||||
result: list[int] = []
|
||||
repl_idx = 0
|
||||
i = 0
|
||||
n = len(prompt_ids)
|
||||
t0, t1, t2 = target
|
||||
num_repl = len(replacements)
|
||||
|
||||
while i < n:
|
||||
if (
|
||||
i + 2 < n
|
||||
and prompt_ids[i] == t0
|
||||
and prompt_ids[i + 1] == t1
|
||||
and prompt_ids[i + 2] == t2
|
||||
):
|
||||
result.extend(replacements[repl_idx])
|
||||
repl_idx += 1
|
||||
i += 3
|
||||
else:
|
||||
result.append(prompt_ids[i])
|
||||
i += 1
|
||||
|
||||
assert repl_idx == num_repl, (
|
||||
f"Found {repl_idx} video placeholders but expected {num_repl}"
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
class Qwen3VLMultiModalProcessor(BaseMultiModalProcessor[Qwen3VLProcessingInfo]):
|
||||
def _call_hf_processor(
|
||||
self,
|
||||
@@ -1211,15 +1254,23 @@ class Qwen3VLMultiModalProcessor(BaseMultiModalProcessor[Qwen3VLProcessingInfo])
|
||||
tok_kwargs: Mapping[str, object],
|
||||
) -> BatchFeature:
|
||||
mm_data = dict(mm_data)
|
||||
processor = self.info.get_hf_processor(**mm_kwargs)
|
||||
|
||||
# Separate video processing from image processing. Because the videos
|
||||
# are processed into several image patches
|
||||
video_input_ids_lst: list[list[int]] = []
|
||||
if videos := mm_data.pop("videos", []):
|
||||
video_grid_thw_lst = []
|
||||
pixel_values_videos_lst = []
|
||||
timestamps_per_video = []
|
||||
|
||||
hf_config = self.info.get_hf_config()
|
||||
tokenizer = self.info.get_tokenizer()
|
||||
merge_size = hf_config.vision_config.spatial_merge_size
|
||||
video_pruning_rate = self.info.ctx.get_mm_config().video_pruning_rate
|
||||
vision_start_token_id = hf_config.vision_start_token_id
|
||||
vision_end_token_id = hf_config.vision_end_token_id
|
||||
video_token_id = hf_config.video_token_id
|
||||
|
||||
for item in videos:
|
||||
video_array, metadata = item
|
||||
|
||||
@@ -1269,55 +1320,38 @@ class Qwen3VLMultiModalProcessor(BaseMultiModalProcessor[Qwen3VLProcessingInfo])
|
||||
tok_kwargs=tok_kwargs,
|
||||
)
|
||||
|
||||
merge_size = processor.video_processor.merge_size
|
||||
# Get video grid info for EVS calculation.
|
||||
# Discard HF output input_ids — we use get_video_repl below
|
||||
# to generate the correct (EVS-adjusted) token sequence.
|
||||
video_outputs.pop("input_ids", None)
|
||||
|
||||
video_grid_thw = video_outputs["video_grid_thw"]
|
||||
num_frames = int(video_grid_thw[0, 0])
|
||||
tokens_per_frame_base = int(video_grid_thw[0, 1:].prod()) // (
|
||||
merge_size**2
|
||||
)
|
||||
|
||||
# Apply EVS if enabled.
|
||||
video_pruning_rate = self.info.ctx.get_mm_config().video_pruning_rate
|
||||
if video_pruning_rate is not None and video_pruning_rate > 0.0:
|
||||
num_tokens = compute_retained_tokens_count(
|
||||
tokens_per_frame=tokens_per_frame_base,
|
||||
num_frames=num_frames,
|
||||
q=video_pruning_rate,
|
||||
)
|
||||
# Here we just need placeholders that won't actually be replaced -
|
||||
# we just need to make sure the total number of tokens is correct
|
||||
# assign all tokens to the first frame.
|
||||
tokens_per_frame = [num_tokens] + [0] * (num_frames - 1)
|
||||
select_token_id = False
|
||||
else:
|
||||
tokens_per_frame = [tokens_per_frame_base] * num_frames
|
||||
select_token_id = True
|
||||
|
||||
# Generate the video replacement with EVS-adjusted token counts
|
||||
tokenizer = self.info.get_tokenizer()
|
||||
hf_config = self.info.get_hf_config()
|
||||
video_repl = Qwen3VLMultiModalProcessor.get_video_repl(
|
||||
tokens_per_frame=tokens_per_frame,
|
||||
timestamps=timestamps,
|
||||
tokenizer=tokenizer,
|
||||
vision_start_token_id=hf_config.vision_start_token_id,
|
||||
vision_end_token_id=hf_config.vision_end_token_id,
|
||||
video_token_id=hf_config.video_token_id,
|
||||
vision_start_token_id=vision_start_token_id,
|
||||
vision_end_token_id=vision_end_token_id,
|
||||
video_token_id=video_token_id,
|
||||
select_token_id=select_token_id,
|
||||
)
|
||||
|
||||
# Convert token IDs to text for the HF processor flow
|
||||
video_placeholder = tokenizer.decode(
|
||||
video_repl.full, skip_special_tokens=False
|
||||
)
|
||||
input_ids = video_outputs.pop("input_ids")
|
||||
video_placeholder = processor.tokenizer.batch_decode(input_ids)[0]
|
||||
prompt = prompt.replace(
|
||||
"<|vision_start|><|video_pad|><|vision_end|>",
|
||||
video_placeholder,
|
||||
1,
|
||||
)
|
||||
video_input_ids_lst.append(list(video_repl.full))
|
||||
|
||||
video_grid_thw_lst.append(video_outputs["video_grid_thw"])
|
||||
pixel_values_videos_lst.append(video_outputs["pixel_values_videos"])
|
||||
@@ -1335,6 +1369,24 @@ class Qwen3VLMultiModalProcessor(BaseMultiModalProcessor[Qwen3VLProcessingInfo])
|
||||
mm_kwargs=mm_kwargs,
|
||||
tok_kwargs=tok_kwargs,
|
||||
)
|
||||
|
||||
# Replace each placeholder triplet with pre-computed video tokens.
|
||||
if video_input_ids_lst:
|
||||
hf_config = self.info.get_hf_config()
|
||||
video_target = [
|
||||
hf_config.vision_start_token_id,
|
||||
hf_config.video_token_id,
|
||||
hf_config.vision_end_token_id,
|
||||
]
|
||||
input_ids = processed_outputs.pop("input_ids")
|
||||
if not isinstance(input_ids, list):
|
||||
input_ids = input_ids.tolist()
|
||||
(prompt_ids,) = input_ids
|
||||
expanded_ids = _replace_video_token_placeholders(
|
||||
prompt_ids, video_target, video_input_ids_lst
|
||||
)
|
||||
processed_outputs["input_ids"] = [expanded_ids]
|
||||
|
||||
combined_outputs = dict(
|
||||
processed_outputs,
|
||||
**video_outputs,
|
||||
|
||||
@@ -13,6 +13,7 @@ from vllm.config.cache import CacheDType
|
||||
from vllm.platforms.interface import DeviceCapability
|
||||
from vllm.triton_utils import tl, triton
|
||||
from vllm.utils.math_utils import cdiv
|
||||
from vllm.utils.torch_utils import np_to_pinned_tensor
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
AttentionCGSupport,
|
||||
@@ -207,7 +208,7 @@ class DeepseekV4FlashMLAMetadataBuilder(
|
||||
# Zero-fill for cudagraphs
|
||||
self.req_id_per_token_buffer.fill_(0)
|
||||
self.req_id_per_token_buffer[: req_id_per_token.shape[0]].copy_(
|
||||
torch.from_numpy(req_id_per_token), non_blocking=True
|
||||
np_to_pinned_tensor(req_id_per_token), non_blocking=True
|
||||
)
|
||||
req_id_per_token = self.req_id_per_token_buffer[:num_tokens]
|
||||
|
||||
|
||||
@@ -488,7 +488,13 @@ class MultiModalBatchedField(BaseMultiModalField):
|
||||
# An optimization when `batch` contains only one tensor:
|
||||
# - produce exactly same result as `torch.stack(batch)`
|
||||
# - will achieve zero-copy if the tensor is contiguous
|
||||
return batch[0].unsqueeze(0).contiguous()
|
||||
out = batch[0].unsqueeze(0)
|
||||
if not pin_memory:
|
||||
return out.contiguous()
|
||||
# Avoid extra copy - pinning unpinned memory will make it contiguous
|
||||
if not out.is_contiguous() and out.is_pinned():
|
||||
out = out.contiguous()
|
||||
return out.pin_memory()
|
||||
first_shape = batch[0].shape
|
||||
if all(elem.shape == first_shape for elem in batch):
|
||||
out = torch.empty(
|
||||
@@ -538,7 +544,13 @@ class MultiModalFlatField(BaseMultiModalField):
|
||||
# An optimization when `batch` contains only one tensor:
|
||||
# - produce exactly same result as `torch.concat(batch)`
|
||||
# - will achieve zero-copy if the tensor is contiguous
|
||||
return batch[0].contiguous()
|
||||
out = batch[0]
|
||||
if not pin_memory:
|
||||
return out.contiguous()
|
||||
# Avoid extra copy - pinning unpinned memory will make it contiguous
|
||||
if not out.is_contiguous() and out.is_pinned():
|
||||
out = out.contiguous()
|
||||
return out.pin_memory()
|
||||
|
||||
dim = self.dim + (self.dim < 0) * len(batch[0].shape)
|
||||
|
||||
|
||||
@@ -0,0 +1,163 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
"""MiniMax M3 parser for reasoning markers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import functools
|
||||
from collections.abc import Iterable, Sequence
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from vllm.parser.engine.events import EventType
|
||||
from vllm.parser.engine.parser_engine import ParserEngine
|
||||
from vllm.parser.engine.parser_engine_config import (
|
||||
ParserEngineConfig,
|
||||
ParserState,
|
||||
Transition,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from vllm.tokenizers import TokenizerLike
|
||||
from vllm.tool_parsers.abstract_tool_parser import Tool
|
||||
|
||||
THINK_START = "<mm:think>"
|
||||
THINK_END = "</mm:think>"
|
||||
|
||||
|
||||
@functools.cache
|
||||
def minimax_m3_config(thinking: bool = False) -> ParserEngineConfig:
|
||||
return ParserEngineConfig(
|
||||
name="minimax_m3",
|
||||
initial_state=ParserState.REASONING if thinking else ParserState.CONTENT,
|
||||
terminals={
|
||||
"THINK_START": THINK_START,
|
||||
"THINK_END": THINK_END,
|
||||
},
|
||||
transitions={
|
||||
(ParserState.CONTENT, "THINK_START"): Transition(
|
||||
ParserState.REASONING,
|
||||
(EventType.REASONING_START,),
|
||||
),
|
||||
(ParserState.REASONING, "THINK_START"): Transition(
|
||||
ParserState.REASONING,
|
||||
(),
|
||||
),
|
||||
(ParserState.REASONING, "THINK_END"): Transition(
|
||||
ParserState.CONTENT,
|
||||
(EventType.REASONING_END,),
|
||||
),
|
||||
(ParserState.CONTENT, "THINK_END"): Transition(
|
||||
ParserState.CONTENT,
|
||||
(),
|
||||
),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
class MiniMaxM3Parser(ParserEngine):
|
||||
"""MiniMax M3 parser backed by the declarative parser engine."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
tokenizer: TokenizerLike,
|
||||
tools: list[Tool] | None = None,
|
||||
**kwargs,
|
||||
) -> None:
|
||||
chat_kwargs = kwargs.get("chat_template_kwargs", {}) or {}
|
||||
self._initial_in_reasoning = chat_kwargs.get("thinking_mode") == "enabled"
|
||||
kwargs.setdefault(
|
||||
"parser_engine_config",
|
||||
minimax_m3_config(thinking=self._initial_in_reasoning),
|
||||
)
|
||||
super().__init__(tokenizer, tools, **kwargs)
|
||||
self._start_token_ids = self._encode_marker(THINK_START)
|
||||
self._end_token_ids = self._encode_marker(THINK_END)
|
||||
|
||||
def _encode_marker(self, marker: str) -> tuple[int, ...]:
|
||||
try:
|
||||
token_ids = self.model_tokenizer.encode(marker, add_special_tokens=False)
|
||||
except TypeError:
|
||||
token_ids = self.model_tokenizer.encode(marker)
|
||||
return tuple(token_ids)
|
||||
|
||||
@staticmethod
|
||||
def _contains_token_sequence(
|
||||
token_ids: Sequence[int], marker_ids: Sequence[int]
|
||||
) -> bool:
|
||||
if not marker_ids or len(marker_ids) > len(token_ids):
|
||||
return False
|
||||
marker_len = len(marker_ids)
|
||||
return any(
|
||||
tuple(token_ids[i : i + marker_len]) == tuple(marker_ids)
|
||||
for i in range(len(token_ids) - marker_len + 1)
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _rfind_token_sequence(
|
||||
token_ids: Sequence[int], marker_ids: Sequence[int]
|
||||
) -> int:
|
||||
if not marker_ids or len(marker_ids) > len(token_ids):
|
||||
return -1
|
||||
marker_len = len(marker_ids)
|
||||
for i in range(len(token_ids) - marker_len, -1, -1):
|
||||
if tuple(token_ids[i : i + marker_len]) == tuple(marker_ids):
|
||||
return i
|
||||
return -1
|
||||
|
||||
def is_reasoning_end(self, input_ids: list[int]) -> bool:
|
||||
start_index = self._rfind_token_sequence(input_ids, self._start_token_ids)
|
||||
end_index = self._rfind_token_sequence(input_ids, self._end_token_ids)
|
||||
if end_index < 0:
|
||||
return False
|
||||
if start_index < 0:
|
||||
return True
|
||||
return end_index > start_index
|
||||
|
||||
def is_reasoning_end_streaming(
|
||||
self, input_ids: Sequence[int], delta_ids: Iterable[int]
|
||||
) -> bool:
|
||||
if self.reasoning_ended:
|
||||
return True
|
||||
if self._engine._lexer.buffer:
|
||||
return False
|
||||
if self._initial_in_reasoning:
|
||||
return False
|
||||
if self._engine.state == ParserState.CONTENT:
|
||||
return bool(input_ids)
|
||||
return False
|
||||
|
||||
def extract_content_ids(self, input_ids: list[int]) -> list[int]:
|
||||
end_index = self._rfind_token_sequence(input_ids, self._end_token_ids)
|
||||
if end_index >= 0:
|
||||
return input_ids[end_index + len(self._end_token_ids) :]
|
||||
|
||||
has_start = self._contains_token_sequence(input_ids, self._start_token_ids)
|
||||
if self._initial_in_reasoning and not has_start:
|
||||
return []
|
||||
|
||||
if not has_start:
|
||||
return input_ids
|
||||
return []
|
||||
|
||||
def count_reasoning_tokens(self, token_ids: Sequence[int]) -> int:
|
||||
count = 0
|
||||
depth = 1 if self._initial_in_reasoning else 0
|
||||
i = 0
|
||||
while i < len(token_ids):
|
||||
if tuple(token_ids[i : i + len(self._start_token_ids)]) == (
|
||||
self._start_token_ids
|
||||
):
|
||||
depth += 1
|
||||
i += len(self._start_token_ids)
|
||||
continue
|
||||
if tuple(token_ids[i : i + len(self._end_token_ids)]) == (
|
||||
self._end_token_ids
|
||||
):
|
||||
if depth > 0:
|
||||
depth -= 1
|
||||
i += len(self._end_token_ids)
|
||||
continue
|
||||
if depth > 0:
|
||||
count += 1
|
||||
i += 1
|
||||
return count
|
||||
@@ -9,7 +9,6 @@ from typing import TYPE_CHECKING
|
||||
from vllm import envs
|
||||
from vllm.plugins import PLATFORM_PLUGINS_GROUP, load_plugins_by_group
|
||||
from vllm.utils.import_utils import resolve_obj_by_qualname
|
||||
from vllm.utils.torch_utils import supports_xccl
|
||||
|
||||
from .interface import CpuArchEnum, Platform, PlatformEnum
|
||||
|
||||
@@ -135,7 +134,7 @@ def xpu_platform_plugin() -> str | None:
|
||||
try:
|
||||
import torch
|
||||
|
||||
if supports_xccl():
|
||||
if torch.distributed.is_xccl_available():
|
||||
dist_backend = "xccl"
|
||||
from vllm.platforms.xpu import XPUPlatform
|
||||
|
||||
|
||||
+11
-1
@@ -23,7 +23,6 @@ import vllm._C_stable_libtorch # noqa
|
||||
import vllm.envs as envs
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.import_utils import import_pynvml
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache
|
||||
from vllm.v1.attention.backends.registry import AttentionBackendEnum
|
||||
|
||||
from .interface import DeviceCapability, Platform, PlatformEnum, in_wsl
|
||||
@@ -88,6 +87,8 @@ def _get_backend_priorities(
|
||||
kv_cache_dtype: CacheDType | None = None,
|
||||
) -> list[AttentionBackendEnum]:
|
||||
"""Get backend priorities with lazy import to avoid circular dependency."""
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache
|
||||
|
||||
if use_mla:
|
||||
if device_capability.major == 10:
|
||||
# Sparse MLA backend priorities
|
||||
@@ -685,6 +686,15 @@ class CudaPlatformBase(Platform):
|
||||
# all the related functions work on real physical device ids.
|
||||
# the major benefit of using NVML is that it will not initialize CUDA
|
||||
class NvmlCudaPlatform(CudaPlatformBase):
|
||||
@classmethod
|
||||
@with_nvml_context
|
||||
def device_control_id_to_physical_device_id(cls, device_id: str) -> int:
|
||||
try:
|
||||
return int(device_id)
|
||||
except ValueError:
|
||||
handle = pynvml.nvmlDeviceGetHandleByUUID(device_id)
|
||||
return pynvml.nvmlDeviceGetIndex(handle)
|
||||
|
||||
@classmethod
|
||||
@cache
|
||||
@with_nvml_context
|
||||
|
||||
+102
-1
@@ -30,6 +30,33 @@ else:
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
_assigned_physical_gpu_ids: list[int] | None = None
|
||||
|
||||
|
||||
def set_assigned_physical_gpu_ids(ids: list[int]) -> None:
|
||||
"""Set the physical GPU IDs assigned to this worker process.
|
||||
Called during worker init so that device_id_to_physical_device_id()
|
||||
can map local_rank to the correct physical device without relying
|
||||
on CUDA_VISIBLE_DEVICES.
|
||||
|
||||
Idempotent: a second call with the same value is a no-op.
|
||||
Raises RuntimeError if called again with a different value.
|
||||
|
||||
This is expected to run during single-threaded worker initialization."""
|
||||
global _assigned_physical_gpu_ids
|
||||
if _assigned_physical_gpu_ids is not None:
|
||||
if _assigned_physical_gpu_ids != ids:
|
||||
raise RuntimeError(
|
||||
f"set_assigned_physical_gpu_ids called with conflicting values: "
|
||||
f"existing={_assigned_physical_gpu_ids}, new={ids}"
|
||||
)
|
||||
return
|
||||
_assigned_physical_gpu_ids = ids
|
||||
|
||||
|
||||
def get_assigned_physical_gpu_ids() -> list[int] | None:
|
||||
return _assigned_physical_gpu_ids
|
||||
|
||||
|
||||
@functools.cache
|
||||
def in_wsl() -> bool:
|
||||
@@ -233,8 +260,34 @@ class Platform:
|
||||
"""
|
||||
import vllm.kernels # noqa: F401
|
||||
|
||||
@classmethod
|
||||
def device_control_id_to_physical_device_id(cls, device_id: str) -> int:
|
||||
"""Map one device-control env entry to an integer physical device ID."""
|
||||
try:
|
||||
return int(device_id)
|
||||
except ValueError as e:
|
||||
raise ValueError(
|
||||
f"Non-integer device ID {device_id!r} is not supported by "
|
||||
f"{cls.device_name}."
|
||||
) from e
|
||||
|
||||
@classmethod
|
||||
def device_id_to_physical_device_id(cls, device_id: int):
|
||||
"""Map a vLLM-local logical device ID to a physical device ID.
|
||||
|
||||
The input is a logical local ID (e.g. a local rank), NOT a visible
|
||||
device ordinal; for the latter use
|
||||
visible_device_id_to_physical_device_id(). The two coincide only
|
||||
when no logical-to-physical mapping is in effect.
|
||||
"""
|
||||
if _assigned_physical_gpu_ids is not None:
|
||||
if device_id >= len(_assigned_physical_gpu_ids):
|
||||
raise IndexError(
|
||||
f"device_id {device_id} is out of range for "
|
||||
f"assigned_physical_gpu_ids {_assigned_physical_gpu_ids} "
|
||||
f"({len(_assigned_physical_gpu_ids)} devices assigned)"
|
||||
)
|
||||
return _assigned_physical_gpu_ids[device_id]
|
||||
# Treat empty device control env var as unset. This is a valid
|
||||
# configuration in Ray setups where the engine is launched in
|
||||
# a CPU-only placement group located on a GPU node.
|
||||
@@ -244,10 +297,58 @@ class Platform:
|
||||
):
|
||||
device_ids = os.environ[cls.device_control_env_var].split(",")
|
||||
physical_device_id = device_ids[device_id]
|
||||
return int(physical_device_id)
|
||||
return cls.device_control_id_to_physical_device_id(physical_device_id)
|
||||
else:
|
||||
return device_id
|
||||
|
||||
@classmethod
|
||||
def logical_device_id_to_visible_device_id(cls, device_id: int) -> int:
|
||||
"""Map a vLLM-local logical device ID to the current process's
|
||||
visible accelerator ordinal.
|
||||
|
||||
vLLM internals use logical local IDs. Physical IDs are used only
|
||||
at platform/topology boundaries. This helper performs the final
|
||||
translation needed by APIs such as ``torch.device("cuda:N")``.
|
||||
"""
|
||||
physical_device_id = cls.device_id_to_physical_device_id(device_id)
|
||||
device_control_env = os.environ.get(cls.device_control_env_var, "")
|
||||
if not device_control_env:
|
||||
return physical_device_id
|
||||
|
||||
visible_physical_device_ids = [
|
||||
cls.device_control_id_to_physical_device_id(physical_id)
|
||||
for physical_id in device_control_env.split(",")
|
||||
]
|
||||
if physical_device_id not in visible_physical_device_ids:
|
||||
raise RuntimeError(
|
||||
f"Physical device {physical_device_id} for logical device "
|
||||
f"{device_id} is not visible in {cls.device_control_env_var}="
|
||||
f"{device_control_env}"
|
||||
)
|
||||
return visible_physical_device_ids.index(physical_device_id)
|
||||
|
||||
@classmethod
|
||||
def visible_device_id_to_physical_device_id(cls, device_id: int) -> int:
|
||||
"""Map a visible accelerator ordinal (e.g. ``torch.device.index``)
|
||||
to a physical device ID.
|
||||
|
||||
This is the inverse of the env-var translation performed by
|
||||
logical_device_id_to_visible_device_id() and is independent of any
|
||||
logical-to-physical mapping set via set_assigned_physical_gpu_ids().
|
||||
"""
|
||||
device_control_env = os.environ.get(cls.device_control_env_var, "")
|
||||
if not device_control_env:
|
||||
return device_id
|
||||
visible_device_ids = device_control_env.split(",")
|
||||
if device_id >= len(visible_device_ids):
|
||||
raise IndexError(
|
||||
f"visible device ordinal {device_id} is out of range for "
|
||||
f"{cls.device_control_env_var}={device_control_env}"
|
||||
)
|
||||
return cls.device_control_id_to_physical_device_id(
|
||||
visible_device_ids[device_id]
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def import_kernels(cls) -> None:
|
||||
"""Import any platform-specific C kernels."""
|
||||
|
||||
@@ -14,7 +14,6 @@ import vllm_xpu_kernels._xpu_C # noqa
|
||||
|
||||
import vllm.envs as envs
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.torch_utils import supports_xpu_graph
|
||||
from vllm.v1.attention.backends.registry import AttentionBackendEnum
|
||||
|
||||
from .interface import DeviceCapability, Platform, PlatformEnum
|
||||
@@ -178,8 +177,6 @@ class XPUPlatform(Platform):
|
||||
|
||||
@classmethod
|
||||
def check_and_update_config(cls, vllm_config: VllmConfig) -> None:
|
||||
parallel_config = vllm_config.parallel_config
|
||||
|
||||
# lazy import to avoid circular import
|
||||
from vllm.config import CUDAGraphMode
|
||||
|
||||
@@ -190,6 +187,10 @@ class XPUPlatform(Platform):
|
||||
attention_config = vllm_config.attention_config
|
||||
if attention_config.backend is None:
|
||||
attention_config.backend = AttentionBackendEnum.FLASH_ATTN
|
||||
|
||||
# lazy import to avoid circular import
|
||||
from vllm.utils.torch_utils import supports_xpu_graph
|
||||
|
||||
if not supports_xpu_graph():
|
||||
compilation_config.cudagraph_mode = CUDAGraphMode.NONE
|
||||
logger.warning(
|
||||
@@ -324,9 +325,8 @@ class XPUPlatform(Platform):
|
||||
|
||||
@classmethod
|
||||
def get_device_communicator_cls(cls) -> str:
|
||||
from vllm.utils.torch_utils import supports_xccl
|
||||
|
||||
if not supports_xccl():
|
||||
if not torch.distributed.is_xccl_available():
|
||||
# Supports xccl with PyTorch versions >= 2.8.0.dev for XPU platform
|
||||
logger.warning(
|
||||
"xccl is not enabled in this torch build, communication"
|
||||
" is not available."
|
||||
|
||||
@@ -2,170 +2,19 @@
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
|
||||
from collections.abc import Iterable, Sequence
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from vllm.entrypoints.openai.engine.protocol import DeltaMessage
|
||||
from vllm.reasoning.basic_parsers import BaseThinkingReasoningParser
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest
|
||||
from vllm.entrypoints.openai.responses.protocol import ResponsesRequest
|
||||
from vllm.parser.engine.adapters import ParserEngineReasoningAdapter
|
||||
from vllm.parser.minimax_m3 import MiniMaxM3Parser
|
||||
|
||||
|
||||
class MiniMaxM3ReasoningParser(BaseThinkingReasoningParser):
|
||||
"""Reasoning parser for MiniMax M3 explicit thinking blocks.
|
||||
class MiniMaxM3ReasoningParser(ParserEngineReasoningAdapter):
|
||||
"""Reasoning parser adapter for MiniMax M3 explicit thinking blocks."""
|
||||
|
||||
MiniMax M3 emits reasoning as:
|
||||
|
||||
<mm:think>reasoning text</mm:think>assistant content
|
||||
|
||||
The M3 tokenizer exposes both markers as complete vocabulary tokens. The
|
||||
chat template may also prefill the start marker when
|
||||
``thinking_mode="enabled"``, so generated text can begin directly inside a
|
||||
reasoning block without emitting ``<mm:think>`` again.
|
||||
"""
|
||||
|
||||
@property
|
||||
def start_token(self) -> str:
|
||||
return "<mm:think>"
|
||||
|
||||
@property
|
||||
def end_token(self) -> str:
|
||||
return "</mm:think>"
|
||||
|
||||
def __init__(self, tokenizer, *args, **kwargs):
|
||||
super().__init__(tokenizer, *args, **kwargs)
|
||||
chat_kwargs = kwargs.get("chat_template_kwargs", {}) or {}
|
||||
self._initial_in_reasoning = chat_kwargs.get("thinking_mode") == "enabled"
|
||||
self._at_response_start = True
|
||||
|
||||
def extract_reasoning(
|
||||
self,
|
||||
model_output: str,
|
||||
request: "ChatCompletionRequest | ResponsesRequest",
|
||||
) -> tuple[str | None, str | None]:
|
||||
# MiniMax M3 can start a response with a stray closer. Drop that first
|
||||
# token only; later unmatched closers stay visible as content.
|
||||
if not self._initial_in_reasoning and model_output.startswith(self.end_token):
|
||||
content = model_output[len(self.end_token) :]
|
||||
return None, content or None
|
||||
|
||||
if self._initial_in_reasoning and self.start_token not in model_output:
|
||||
reasoning, end, content = model_output.partition(self.end_token)
|
||||
if not end:
|
||||
return model_output, None
|
||||
return reasoning, content or None
|
||||
|
||||
if self.start_token not in model_output:
|
||||
return None, model_output
|
||||
|
||||
content_before, _, after_start = model_output.partition(self.start_token)
|
||||
reasoning, end, content_after = after_start.partition(self.end_token)
|
||||
if not end:
|
||||
return reasoning, content_before or None
|
||||
|
||||
return reasoning, (content_before + content_after) or None
|
||||
_parser_engine_cls = MiniMaxM3Parser
|
||||
|
||||
def is_reasoning_end_streaming(
|
||||
self, input_ids: Sequence[int], delta_ids: Iterable[int]
|
||||
) -> bool:
|
||||
delta_ids = tuple(delta_ids)
|
||||
if self.end_token_id in delta_ids:
|
||||
return True
|
||||
if self.end_token_id in input_ids:
|
||||
return True
|
||||
if self._initial_in_reasoning:
|
||||
return False
|
||||
if self.start_token_id not in input_ids:
|
||||
return bool(input_ids)
|
||||
return False
|
||||
|
||||
def extract_content_ids(self, input_ids: list[int]) -> list[int]:
|
||||
if self.end_token_id in input_ids:
|
||||
end_index = len(input_ids) - 1 - input_ids[::-1].index(self.end_token_id)
|
||||
return input_ids[end_index + 1 :]
|
||||
|
||||
if self._initial_in_reasoning and self.start_token_id not in input_ids:
|
||||
return []
|
||||
|
||||
if self.start_token_id not in input_ids:
|
||||
return input_ids
|
||||
return []
|
||||
|
||||
def extract_reasoning_streaming(
|
||||
self,
|
||||
previous_text: str,
|
||||
current_text: str,
|
||||
delta_text: str,
|
||||
previous_token_ids: Sequence[int],
|
||||
current_token_ids: Sequence[int],
|
||||
delta_token_ids: Sequence[int],
|
||||
) -> DeltaMessage | None:
|
||||
if not delta_text:
|
||||
return None
|
||||
|
||||
if self._at_response_start and not self._initial_in_reasoning:
|
||||
# Apply the leading-closer tolerance once. Later unmatched closers
|
||||
# stay visible as content.
|
||||
self._at_response_start = False
|
||||
if delta_text.startswith(self.end_token):
|
||||
delta_text = delta_text[len(self.end_token) :]
|
||||
if not delta_text:
|
||||
return None
|
||||
if delta_token_ids and delta_token_ids[0] == self.end_token_id:
|
||||
delta_token_ids = delta_token_ids[1:]
|
||||
|
||||
if self.end_token_id in previous_token_ids:
|
||||
return DeltaMessage(content=delta_text)
|
||||
|
||||
if (
|
||||
self._initial_in_reasoning
|
||||
and self.start_token_id not in previous_token_ids
|
||||
and self.start_token_id not in delta_token_ids
|
||||
):
|
||||
if self.end_token_id in delta_token_ids:
|
||||
reasoning, _, content = delta_text.partition(self.end_token)
|
||||
return DeltaMessage(
|
||||
reasoning=reasoning or None,
|
||||
content=content or None,
|
||||
)
|
||||
return DeltaMessage(reasoning=delta_text)
|
||||
|
||||
if (
|
||||
self.start_token_id not in previous_token_ids
|
||||
and self.start_token_id not in delta_token_ids
|
||||
):
|
||||
return DeltaMessage(content=delta_text)
|
||||
|
||||
if self.end_token_id in delta_token_ids:
|
||||
reasoning_text, _, content = delta_text.partition(self.end_token)
|
||||
if self.start_token_id in delta_token_ids:
|
||||
_, _, reasoning_text = reasoning_text.partition(self.start_token)
|
||||
return DeltaMessage(
|
||||
reasoning=reasoning_text or None,
|
||||
content=content or None,
|
||||
)
|
||||
|
||||
if self.start_token_id in delta_token_ids:
|
||||
_, _, reasoning = delta_text.partition(self.start_token)
|
||||
return DeltaMessage(reasoning=reasoning) if reasoning else None
|
||||
|
||||
return DeltaMessage(reasoning=delta_text)
|
||||
|
||||
def count_reasoning_tokens(self, token_ids: Sequence[int]) -> int:
|
||||
if not self._initial_in_reasoning:
|
||||
return super().count_reasoning_tokens(token_ids)
|
||||
|
||||
count = 0
|
||||
depth = 1
|
||||
for token_id in token_ids:
|
||||
if token_id == self.start_token_id:
|
||||
depth += 1
|
||||
continue
|
||||
if token_id == self.end_token_id:
|
||||
if depth > 0:
|
||||
depth -= 1
|
||||
continue
|
||||
if depth > 0:
|
||||
count += 1
|
||||
return count
|
||||
return self._parser_engine.is_reasoning_end_streaming(
|
||||
list(input_ids), tuple(delta_ids)
|
||||
)
|
||||
|
||||
@@ -15,7 +15,7 @@ Register a lazy module mapping.
|
||||
Example:
|
||||
ToolParserManager.register_lazy_module(
|
||||
name="kimi_k2",
|
||||
module_path="vllm.tool_parsers.kimi_k2_parser",
|
||||
module_path="vllm.tool_parsers.kimi_k2_tool_parser",
|
||||
class_name="KimiK2ToolParser",
|
||||
)
|
||||
"""
|
||||
|
||||
+18
-15
@@ -3,7 +3,6 @@
|
||||
import contextlib
|
||||
import importlib.metadata
|
||||
import os
|
||||
import platform
|
||||
import random
|
||||
import threading
|
||||
from collections.abc import Callable, Collection
|
||||
@@ -18,6 +17,7 @@ from torch.library import Library, infer_schema
|
||||
|
||||
import vllm.envs as envs
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from vllm.config import ModelConfig
|
||||
@@ -68,9 +68,7 @@ MODELOPT_TO_VLLM_KV_CACHE_DTYPE_MAP = {
|
||||
T = TypeVar("T")
|
||||
|
||||
|
||||
# Pin memory in non-WSL case.
|
||||
# Logic duplicated here for now to avoid circular import.
|
||||
PIN_MEMORY = "microsoft" not in " ".join(platform.uname()).lower()
|
||||
PIN_MEMORY = is_pin_memory_available()
|
||||
|
||||
|
||||
def is_quantized_kv_cache(kv_cache_dtype: str) -> bool:
|
||||
@@ -606,14 +604,24 @@ def create_kv_caches_with_random(
|
||||
|
||||
|
||||
def async_tensor_h2d(
|
||||
data: list,
|
||||
dtype: torch.dtype,
|
||||
data: list | np.ndarray | torch.Tensor,
|
||||
device: str | torch.device,
|
||||
pin_memory: bool = PIN_MEMORY,
|
||||
dtype: torch.dtype | None = None,
|
||||
) -> torch.Tensor:
|
||||
"""Asynchronously create a tensor and copy it from host to device."""
|
||||
t = torch.tensor(data, dtype=dtype, pin_memory=pin_memory, device="cpu")
|
||||
return t.to(device=device, non_blocking=True)
|
||||
"""Copy list/numpy array/tensor async from host to device."""
|
||||
if isinstance(data, np.ndarray):
|
||||
data = torch.from_numpy(data)
|
||||
if isinstance(data, torch.Tensor):
|
||||
t = data.pin_memory() if PIN_MEMORY else data
|
||||
else:
|
||||
t = torch.tensor(data, dtype=dtype, pin_memory=PIN_MEMORY, device="cpu")
|
||||
assert t.is_cpu
|
||||
return t.to(device=device, dtype=dtype, non_blocking=True)
|
||||
|
||||
|
||||
def np_to_pinned_tensor(array: np.ndarray) -> torch.Tensor:
|
||||
t = torch.from_numpy(array)
|
||||
return t.pin_memory() if PIN_MEMORY else t
|
||||
|
||||
|
||||
def make_ndarray_with_pad(
|
||||
@@ -914,11 +922,6 @@ def _encode_layer_name(layer_name: str) -> str | LayerName:
|
||||
return LayerName(layer_name) if _USE_LAYERNAME else layer_name
|
||||
|
||||
|
||||
# Supports xccl with PyTorch versions >= 2.8.0.dev for XPU platform
|
||||
def supports_xccl() -> bool:
|
||||
return torch.distributed.is_xccl_available()
|
||||
|
||||
|
||||
# Supports XPU Graph with PyTorch versions >= 2.11.0.dev for XPU platform
|
||||
def supports_xpu_graph() -> bool:
|
||||
return is_torch_equal_or_newer("2.11.0.dev")
|
||||
|
||||
@@ -41,8 +41,8 @@ from vllm.utils.flashinfer import (
|
||||
use_trtllm_attention,
|
||||
)
|
||||
from vllm.utils.math_utils import cdiv
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import (
|
||||
PIN_MEMORY,
|
||||
canonicalize_singleton_dim_strides,
|
||||
is_quantized_kv_cache,
|
||||
is_strictly_contiguous,
|
||||
@@ -708,9 +708,7 @@ class FlashInferMetadataBuilder(AttentionMetadataBuilder[FlashInferMetadata]):
|
||||
# Since we do not have explicit synchronization in ModelRunnerV2, we do not pin
|
||||
# reused CPU buffers to avoid a race condition between step N async copies to
|
||||
# GPU and step N+1 buffer updates.
|
||||
self.pin_memory = (
|
||||
not vllm_config.use_v2_model_runner and is_pin_memory_available()
|
||||
)
|
||||
self.pin_memory = not vllm_config.use_v2_model_runner and PIN_MEMORY
|
||||
self.paged_kv_indptr = self._make_buffer(max_num_reqs + 1)
|
||||
self.paged_kv_indptr_cpu_buffer = torch.zeros_like(
|
||||
self.paged_kv_indptr.cpu, pin_memory=self.pin_memory
|
||||
|
||||
@@ -28,7 +28,11 @@ from vllm.logger import init_logger
|
||||
from vllm.model_executor.layers.attention import Attention
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.utils.math_utils import cdiv
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache, is_torch_equal_or_newer
|
||||
from vllm.utils.torch_utils import (
|
||||
async_tensor_h2d,
|
||||
is_quantized_kv_cache,
|
||||
is_torch_equal_or_newer,
|
||||
)
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
AttentionCGSupport,
|
||||
@@ -58,7 +62,7 @@ def _offsets_to_doc_ids_tensor(
|
||||
doc_ids = torch.repeat_interleave(
|
||||
torch.arange(len(counts), dtype=torch.int32), counts
|
||||
)
|
||||
return doc_ids.to(device, non_blocking=True)
|
||||
return async_tensor_h2d(doc_ids, device=device)
|
||||
|
||||
|
||||
def pad_to_multiple(x: torch.Tensor, multiple: int, dim: int):
|
||||
|
||||
@@ -8,6 +8,7 @@ from typing import Literal
|
||||
import torch
|
||||
|
||||
from vllm.config import VllmConfig
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
AttentionCGSupport,
|
||||
@@ -203,8 +204,8 @@ class GDNAttentionMetadataBuilder(AttentionMetadataBuilder[GDNAttentionMetadata]
|
||||
spec_sequence_masks = None
|
||||
spec_sequence_masks_cpu = None
|
||||
else:
|
||||
spec_sequence_masks = spec_sequence_masks_cpu.to(
|
||||
query_start_loc.device, non_blocking=True
|
||||
spec_sequence_masks = async_tensor_h2d(
|
||||
spec_sequence_masks_cpu, device=query_start_loc.device
|
||||
)
|
||||
|
||||
if spec_sequence_masks is None:
|
||||
@@ -376,12 +377,14 @@ class GDNAttentionMetadataBuilder(AttentionMetadataBuilder[GDNAttentionMetadata]
|
||||
)
|
||||
|
||||
assert prefill_query_start_loc_cpu is not None
|
||||
chunk_indices = prepare_chunk_indices(
|
||||
prefill_query_start_loc_cpu, FLA_CHUNK_SIZE
|
||||
).to(device=gpu_device, non_blocking=True)
|
||||
chunk_offsets = prepare_chunk_offsets(
|
||||
prefill_query_start_loc_cpu, FLA_CHUNK_SIZE
|
||||
).to(device=gpu_device, non_blocking=True)
|
||||
chunk_indices = async_tensor_h2d(
|
||||
prepare_chunk_indices(prefill_query_start_loc_cpu, FLA_CHUNK_SIZE),
|
||||
device=gpu_device,
|
||||
)
|
||||
chunk_offsets = async_tensor_h2d(
|
||||
prepare_chunk_offsets(prefill_query_start_loc_cpu, FLA_CHUNK_SIZE),
|
||||
device=gpu_device,
|
||||
)
|
||||
|
||||
if num_prefills > 0:
|
||||
has_initial_state = context_lens_tensor > 0
|
||||
|
||||
@@ -7,6 +7,7 @@ from typing import Any
|
||||
import torch
|
||||
|
||||
from vllm.config import VllmConfig
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
CommonAttentionMetadata,
|
||||
@@ -68,22 +69,22 @@ def compute_varlen_chunk_metadata(
|
||||
|
||||
# Exclusive prefix sum over logical-chunk lengths
|
||||
if chunk_lens:
|
||||
cu_chunk_seqlens = torch.tensor(
|
||||
[0] + list(itertools.accumulate(chunk_lens)),
|
||||
device=device,
|
||||
dtype=torch.int32,
|
||||
)
|
||||
# Final boundary must equal total tokens
|
||||
assert int(cu_chunk_seqlens[-1].item()) == total
|
||||
cu_chunk_seqlens_list = [0] + list(itertools.accumulate(chunk_lens))
|
||||
# Final boundary must equal total tokens (check on host to avoid a sync)
|
||||
assert cu_chunk_seqlens_list[-1] == total
|
||||
else:
|
||||
cu_chunk_seqlens = torch.tensor([0], device=device, dtype=torch.int32)
|
||||
cu_chunk_seqlens_list = [0]
|
||||
cu_chunk_seqlens = async_tensor_h2d(
|
||||
cu_chunk_seqlens_list, dtype=torch.int32, device=device
|
||||
)
|
||||
|
||||
last_chunk_indices_t = (
|
||||
torch.tensor(last_chunk_indices, device=device, dtype=torch.int32)
|
||||
if len(starts) > 0
|
||||
else torch.empty((0,), device=device, dtype=torch.int32)
|
||||
# last_chunk_indices is empty when there are no sequences (len(starts) == 0).
|
||||
last_chunk_indices_t = async_tensor_h2d(
|
||||
last_chunk_indices, dtype=torch.int32, device=device
|
||||
)
|
||||
seq_idx_chunks_t = async_tensor_h2d(
|
||||
seq_idx_chunks, dtype=torch.int32, device=device
|
||||
)
|
||||
seq_idx_chunks_t = torch.tensor(seq_idx_chunks, device=device, dtype=torch.int32)
|
||||
return cu_chunk_seqlens, last_chunk_indices_t, seq_idx_chunks_t
|
||||
|
||||
|
||||
|
||||
@@ -26,7 +26,7 @@ from vllm.model_executor.layers.attention.mla_attention import (
|
||||
get_mla_dims,
|
||||
)
|
||||
from vllm.platforms.interface import DeviceCapability
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache, np_to_pinned_tensor
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
AttentionCGSupport,
|
||||
@@ -217,7 +217,7 @@ class FlashInferMLASparseMetadataBuilder(
|
||||
# Zero-fill for cudagraphs
|
||||
self.req_id_per_token_buffer.fill_(0)
|
||||
self.req_id_per_token_buffer[: req_id_per_token.shape[0]].copy_(
|
||||
torch.from_numpy(req_id_per_token), non_blocking=True
|
||||
np_to_pinned_tensor(req_id_per_token), non_blocking=True
|
||||
)
|
||||
req_id_per_token_tensor = self.req_id_per_token_buffer[:num_tokens]
|
||||
|
||||
|
||||
@@ -16,7 +16,7 @@ from vllm.model_executor.layers.attention.mla_attention import (
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.platforms.interface import DeviceCapability
|
||||
from vllm.utils.platform_utils import num_compute_units
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache, np_to_pinned_tensor
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
AttentionCGSupport,
|
||||
@@ -503,7 +503,7 @@ class FlashMLASparseMetadataBuilder(AttentionMetadataBuilder[FlashMLASparseMetad
|
||||
# Zero-fill for cudagraphs
|
||||
self.req_id_per_token_buffer.fill_(0)
|
||||
self.req_id_per_token_buffer[: req_id_per_token.shape[0]].copy_(
|
||||
torch.from_numpy(req_id_per_token), non_blocking=True
|
||||
np_to_pinned_tensor(req_id_per_token), non_blocking=True
|
||||
)
|
||||
req_id_per_token = self.req_id_per_token_buffer[:num_tokens]
|
||||
|
||||
|
||||
@@ -17,7 +17,7 @@ from typing_extensions import runtime_checkable
|
||||
|
||||
from vllm.config import VllmConfig, get_layers_from_vllm_config
|
||||
from vllm.utils.math_utils import cdiv
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.utils.torch_utils import PIN_MEMORY, async_tensor_h2d, np_to_pinned_tensor
|
||||
from vllm.v1.kv_cache_interface import KVCacheSpec, MambaSpec
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -364,8 +364,8 @@ def make_local_attention_virtual_batches(
|
||||
# tensor first, which recovers perf.
|
||||
# Upload the index tensors to the block_table's device up-front so that the
|
||||
# fancy indexing below doesn't implicitly force a synchronous H2D copy.
|
||||
batch_indices_torch = torch.from_numpy(batch_indices).to(device, non_blocking=True)
|
||||
block_indices_torch = torch.from_numpy(block_indices).to(device, non_blocking=True)
|
||||
batch_indices_torch = async_tensor_h2d(batch_indices, device=device)
|
||||
block_indices_torch = async_tensor_h2d(block_indices, device=device)
|
||||
|
||||
# Save as a lambda so we can return this for update_block_table
|
||||
make_block_table = lambda block_table: block_table[
|
||||
@@ -379,8 +379,8 @@ def make_local_attention_virtual_batches(
|
||||
|
||||
return CommonAttentionMetadata(
|
||||
query_start_loc_cpu=query_start_loc_cpu,
|
||||
query_start_loc=query_start_loc_cpu.to(device=device, non_blocking=True),
|
||||
seq_lens=seq_lens_cpu.to(device=device, non_blocking=True),
|
||||
query_start_loc=async_tensor_h2d(query_start_loc_cpu, device=device),
|
||||
seq_lens=async_tensor_h2d(seq_lens_cpu, device=device),
|
||||
num_reqs=len(seq_lens_cpu),
|
||||
num_actual_tokens=common_attn_metadata.num_actual_tokens,
|
||||
max_query_len=seqlens_q_local.max(),
|
||||
@@ -808,14 +808,12 @@ def create_fast_prefill_custom_backend(
|
||||
|
||||
|
||||
def compute_causal_conv1d_metadata(
|
||||
query_start_loc_p_cpu: torch.Tensor,
|
||||
*,
|
||||
device: torch.device,
|
||||
):
|
||||
query_start_loc_p_cpu: torch.Tensor, *, device: torch.device
|
||||
) -> tuple[dict[int, dict[str, Any]], torch.Tensor, torch.Tensor]:
|
||||
# Needed for causal_conv1d. Use the CPU query_start_loc to avoid DtoH sync.
|
||||
assert query_start_loc_p_cpu.device.type == "cpu"
|
||||
seqlens = query_start_loc_p_cpu.diff()
|
||||
nums_dict = {} # type: ignore
|
||||
nums_dict: dict[int, dict[str, Any]] = {}
|
||||
batch_ptr = None
|
||||
token_chunk_offset_ptr = None
|
||||
for BLOCK_M in [8]: # cover all BLOCK_M values
|
||||
@@ -823,7 +821,7 @@ def compute_causal_conv1d_metadata(
|
||||
nums_dict[BLOCK_M] = {}
|
||||
nums_dict[BLOCK_M]["nums"] = nums
|
||||
nums_dict[BLOCK_M]["tot"] = nums.sum().item()
|
||||
mlist = torch.from_numpy(np.repeat(np.arange(len(nums)), nums))
|
||||
mlist = np_to_pinned_tensor(np.repeat(np.arange(len(nums)), nums))
|
||||
nums_dict[BLOCK_M]["mlist"] = mlist
|
||||
mlist_len = len(nums_dict[BLOCK_M]["mlist"])
|
||||
nums_dict[BLOCK_M]["mlist_len"] = mlist_len
|
||||
@@ -831,7 +829,7 @@ def compute_causal_conv1d_metadata(
|
||||
offsetlist = [] # type: ignore
|
||||
for idx, num in enumerate(nums):
|
||||
offsetlist.extend(range(num))
|
||||
offsetlist = torch.tensor(offsetlist, dtype=torch.int32)
|
||||
offsetlist = torch.tensor(offsetlist, dtype=torch.int32, pin_memory=PIN_MEMORY)
|
||||
nums_dict[BLOCK_M]["offsetlist"] = offsetlist
|
||||
|
||||
if batch_ptr is None:
|
||||
@@ -845,16 +843,15 @@ def compute_causal_conv1d_metadata(
|
||||
else:
|
||||
if batch_ptr.nelement() < MAX_NUM_PROGRAMS:
|
||||
batch_ptr.resize_(MAX_NUM_PROGRAMS).fill_(PAD_SLOT_ID)
|
||||
token_chunk_offset_ptr.resize_( # type: ignore
|
||||
MAX_NUM_PROGRAMS
|
||||
).fill_(PAD_SLOT_ID)
|
||||
assert token_chunk_offset_ptr is not None
|
||||
token_chunk_offset_ptr.resize_(MAX_NUM_PROGRAMS).fill_(PAD_SLOT_ID)
|
||||
|
||||
assert batch_ptr is not None
|
||||
batch_ptr[0:mlist_len].copy_(mlist, non_blocking=True)
|
||||
token_chunk_offset_ptr[ # type: ignore
|
||||
0:mlist_len
|
||||
].copy_(offsetlist, non_blocking=True)
|
||||
assert token_chunk_offset_ptr is not None
|
||||
token_chunk_offset_ptr[0:mlist_len].copy_(offsetlist, non_blocking=True)
|
||||
nums_dict[BLOCK_M]["batch_ptr"] = batch_ptr
|
||||
nums_dict[BLOCK_M]["token_chunk_offset_ptr"] = token_chunk_offset_ptr # type: ignore
|
||||
nums_dict[BLOCK_M]["token_chunk_offset_ptr"] = token_chunk_offset_ptr
|
||||
|
||||
return nums_dict, batch_ptr, token_chunk_offset_ptr
|
||||
|
||||
|
||||
@@ -58,7 +58,9 @@ def _indexer_k_quant_and_cache_kernel(
|
||||
slot_id = tl.load(slot_mapping_ptr + tid)
|
||||
if slot_id < 0:
|
||||
return
|
||||
block_id = slot_id // block_size
|
||||
# The packed KV layout makes per-block strides large
|
||||
# enough that block_id * stride can exceed 32-bit range.
|
||||
block_id = (slot_id // block_size).to(tl.int64)
|
||||
block_offset = slot_id % block_size
|
||||
tile_block_id = block_offset // BLOCK_TILE_SIZE
|
||||
tile_block_offset = block_offset % BLOCK_TILE_SIZE
|
||||
@@ -179,7 +181,9 @@ def _cp_gather_indexer_quant_cache_kernel(
|
||||
block_table_ptr + block_table_offset, mask=valid_block_table, other=-1
|
||||
)
|
||||
valid_block = valid_block_table & (block_id >= 0) & (block_id < NUM_BLOCKS)
|
||||
safe_block_id = tl.where(valid_block, block_id, 0)
|
||||
# The packed KV layout makes per-block strides large
|
||||
# enough that block_id * stride can exceed 32-bit range.
|
||||
safe_block_id = tl.where(valid_block, block_id, 0).to(tl.int64)
|
||||
safe_block_offset = tl.where(valid_block, block_offset, 0)
|
||||
tiled_block_offset = safe_block_offset % BLOCK_TILE_SIZE
|
||||
if LAYOUT == "SHUFFLE":
|
||||
|
||||
@@ -938,9 +938,7 @@ def _pool_bytes_per_block(kv_cache_groups: list[KVCacheGroupSpec]) -> int:
|
||||
kv_cache_groups[0].kv_cache_spec, UniformTypeKVCacheSpecs
|
||||
):
|
||||
return kv_cache_groups[0].kv_cache_spec.page_size_bytes
|
||||
if all(
|
||||
isinstance(g.kv_cache_spec, UniformTypeKVCacheSpecs) for g in kv_cache_groups
|
||||
):
|
||||
if _use_packed_kv_cache_groups(kv_cache_groups):
|
||||
# buckets = {page_size: [[layer_names], [layer_names], ...]}
|
||||
buckets = _bucket_layers_by_page_size(kv_cache_groups)
|
||||
return sum(ps * len(slots) for ps, slots in buckets.items())
|
||||
@@ -1218,16 +1216,29 @@ def _bucket_layers_by_page_size(
|
||||
return buckets
|
||||
|
||||
|
||||
def _get_kv_cache_config_deepseek_v4(
|
||||
def _use_packed_kv_cache_groups(
|
||||
kv_cache_groups: list[KVCacheGroupSpec],
|
||||
) -> bool:
|
||||
is_dsv4 = all(
|
||||
isinstance(group.kv_cache_spec, UniformTypeKVCacheSpecs)
|
||||
for group in kv_cache_groups
|
||||
)
|
||||
return is_dsv4 or (
|
||||
bool(envs.VLLM_USE_PACKED_HMA_KV_CACHE) and len(kv_cache_groups) > 1
|
||||
)
|
||||
|
||||
|
||||
def _get_kv_cache_config_packed(
|
||||
vllm_config: VllmConfig,
|
||||
kv_cache_groups: list[KVCacheGroupSpec],
|
||||
available_memory: int,
|
||||
) -> tuple[int, list[KVCacheTensor]]:
|
||||
"""DeepseekV4 KV cache tensor layout planning.
|
||||
"""Plan a packed per-block KV cache tensor layout.
|
||||
|
||||
Emit one KVCacheTensor per (slot_idx, page_size). Layers from different
|
||||
groups at the same slot share a tensor (they have independent block
|
||||
tables so block-id namespaces never collide).
|
||||
tables so block-id namespaces never collide). Each emitted tensor aliases
|
||||
one physical backing allocation, with per-block data laid out contiguously.
|
||||
"""
|
||||
# buckets = {page_size: [[layer_names], [layer_names], ...]}
|
||||
buckets = _bucket_layers_by_page_size(kv_cache_groups)
|
||||
@@ -1255,6 +1266,9 @@ def _get_kv_cache_config_deepseek_v4(
|
||||
return num_blocks, kv_cache_tensors
|
||||
|
||||
|
||||
_get_kv_cache_config_deepseek_v4 = _get_kv_cache_config_packed
|
||||
|
||||
|
||||
def get_kv_cache_config_from_groups(
|
||||
vllm_config: VllmConfig,
|
||||
kv_cache_groups: list[KVCacheGroupSpec],
|
||||
@@ -1299,13 +1313,11 @@ def get_kv_cache_config_from_groups(
|
||||
)
|
||||
for layer_name in kv_cache_groups[0].layer_names
|
||||
]
|
||||
elif all(
|
||||
isinstance(group.kv_cache_spec, UniformTypeKVCacheSpecs)
|
||||
for group in kv_cache_groups
|
||||
):
|
||||
# DeepseekV4: UniformTypeKVCacheSpecs but multiple groups.
|
||||
# Delegate to the DeepseekV4-specific allocator.
|
||||
num_blocks, kv_cache_tensors = _get_kv_cache_config_deepseek_v4(
|
||||
elif _use_packed_kv_cache_groups(kv_cache_groups):
|
||||
# DeepSeek V4 keeps the existing packed layout. Other multi-group
|
||||
# attention-only HMA layouts can opt in with
|
||||
# VLLM_USE_PACKED_HMA_KV_CACHE=1.
|
||||
num_blocks, kv_cache_tensors = _get_kv_cache_config_packed(
|
||||
vllm_config, kv_cache_groups, available_memory
|
||||
)
|
||||
else:
|
||||
|
||||
@@ -118,8 +118,8 @@ class CachedRequestData:
|
||||
# NOTE(woosuk): new_token_ids is only used for pipeline parallelism.
|
||||
# When PP is not used, new_token_ids will be empty.
|
||||
new_token_ids: list[list[int]]
|
||||
# For requests not scheduled in the last step, propagate the token ids to the
|
||||
# connector. Won't contain requests that were scheduled in the prior step.
|
||||
# MRV1-only: For requests not scheduled in the last step, propagate the token ids
|
||||
# to the connector. Won't contain requests scheduled in the prior step.
|
||||
all_token_ids: dict[str, list[int]]
|
||||
new_block_ids: list[tuple[list[int], ...] | None]
|
||||
num_computed_tokens: list[int]
|
||||
|
||||
@@ -101,6 +101,7 @@ class Scheduler(SchedulerInterface):
|
||||
self.finished_req_ids_dict: dict[int, set[str]] | None = (
|
||||
defaultdict(set) if include_finished_set else None
|
||||
)
|
||||
# Track requests scheduled in prior step (MRV1-only).
|
||||
self.prev_step_scheduled_req_ids: set[str] = set()
|
||||
|
||||
# Scheduling constraints.
|
||||
@@ -1010,8 +1011,8 @@ class Scheduler(SchedulerInterface):
|
||||
|
||||
# Construct the scheduler output.
|
||||
if self.use_v2_model_runner:
|
||||
scheduled_new_reqs = scheduled_new_reqs + scheduled_resumed_reqs
|
||||
scheduled_resumed_reqs = []
|
||||
scheduled_new_reqs.extend(scheduled_resumed_reqs)
|
||||
scheduled_resumed_reqs.clear()
|
||||
new_reqs_data = [
|
||||
NewRequestData.from_request(
|
||||
req,
|
||||
@@ -1037,9 +1038,10 @@ class Scheduler(SchedulerInterface):
|
||||
req_to_new_blocks,
|
||||
)
|
||||
|
||||
# Record the request ids that were scheduled in this step.
|
||||
self.prev_step_scheduled_req_ids.clear()
|
||||
self.prev_step_scheduled_req_ids.update(num_scheduled_tokens.keys())
|
||||
# Record the request ids that were scheduled in this step (MRV1-only).
|
||||
if not self.use_v2_model_runner:
|
||||
self.prev_step_scheduled_req_ids.clear()
|
||||
self.prev_step_scheduled_req_ids.update(num_scheduled_tokens.keys())
|
||||
|
||||
new_block_ids_to_zero = (
|
||||
(self.kv_cache_manager.take_new_block_ids() or None)
|
||||
@@ -1252,12 +1254,11 @@ class Scheduler(SchedulerInterface):
|
||||
req.num_computed_tokens : req.num_computed_tokens + num_tokens
|
||||
]
|
||||
new_token_ids.append(token_ids)
|
||||
scheduled_in_prev_step = req_id in self.prev_step_scheduled_req_ids
|
||||
if idx >= num_running_reqs:
|
||||
assert not scheduled_in_prev_step
|
||||
resumed_req_ids.add(req_id)
|
||||
if not scheduled_in_prev_step:
|
||||
all_token_ids[req_id] = req.all_token_ids.copy()
|
||||
if not self.use_v2_model_runner: # noqa: SIM102
|
||||
if req_id not in self.prev_step_scheduled_req_ids:
|
||||
all_token_ids[req_id] = req.all_token_ids.copy()
|
||||
new_block_ids.append(
|
||||
req_to_new_blocks[req_id].get_block_ids(allow_none=True)
|
||||
)
|
||||
|
||||
@@ -1109,9 +1109,3 @@ class AsyncLLM(EngineClient):
|
||||
async def finish_weight_update(self) -> None:
|
||||
"""Finish the current weight update."""
|
||||
await self.collective_rpc("finish_weight_update")
|
||||
# Invalidate cached state computed with the old weights so it isn't
|
||||
# reused for subsequent requests:
|
||||
# - prefix cache: KV blocks computed with the old weights
|
||||
# - encoder cache: multimodal embeddings keyed only by mm_hash
|
||||
await self.reset_prefix_cache()
|
||||
await self.reset_encoder_cache()
|
||||
|
||||
+16
-9
@@ -74,7 +74,7 @@ from vllm.v1.engine.utils import (
|
||||
EngineHandshakeMetadata,
|
||||
EngineZmqAddresses,
|
||||
SignalCallback,
|
||||
get_device_indices,
|
||||
get_physical_gpu_ids_for_local_dp_rank,
|
||||
)
|
||||
from vllm.v1.executor import Executor
|
||||
from vllm.v1.kv_cache_interface import KVCacheConfig, get_kv_cache_spec_kind
|
||||
@@ -2175,23 +2175,30 @@ class EngineCoreActorMixin:
|
||||
pass
|
||||
else:
|
||||
device_control_env_var = current_platform.device_control_env_var
|
||||
self._set_cuda_visible_devices(
|
||||
self._set_assigned_physical_gpu_ids(
|
||||
vllm_config, local_dp_rank, device_control_env_var
|
||||
)
|
||||
|
||||
def _set_cuda_visible_devices(
|
||||
self, vllm_config: VllmConfig, local_dp_rank: int, device_control_env_var: str
|
||||
def _set_assigned_physical_gpu_ids(
|
||||
self,
|
||||
vllm_config: VllmConfig,
|
||||
local_dp_rank: int,
|
||||
device_control_env_var: str,
|
||||
):
|
||||
world_size = vllm_config.parallel_config.world_size
|
||||
# Set CUDA_VISIBLE_DEVICES or equivalent.
|
||||
try:
|
||||
value = get_device_indices(
|
||||
device_control_env_var, local_dp_rank, world_size
|
||||
physical_gpu_ids = get_physical_gpu_ids_for_local_dp_rank(
|
||||
device_control_env_var,
|
||||
local_dp_rank,
|
||||
world_size,
|
||||
user_assigned_gpu_ids=(
|
||||
vllm_config.parallel_config.assigned_physical_gpu_ids
|
||||
),
|
||||
)
|
||||
os.environ[device_control_env_var] = value
|
||||
vllm_config.parallel_config.assigned_physical_gpu_ids = physical_gpu_ids
|
||||
except IndexError as e:
|
||||
raise Exception(
|
||||
f"Error setting {device_control_env_var}: "
|
||||
f"Error computing assigned_physical_gpu_ids: "
|
||||
f"local range: [{local_dp_rank * world_size}, "
|
||||
f"{(local_dp_rank + 1) * world_size}) "
|
||||
f'base value: "{os.getenv(device_control_env_var)}"'
|
||||
|
||||
+66
-43
@@ -12,7 +12,6 @@ from multiprocessing import Process, connection
|
||||
from multiprocessing.process import BaseProcess
|
||||
from multiprocessing.queues import Queue
|
||||
from typing import TYPE_CHECKING, cast
|
||||
from unittest.mock import patch
|
||||
|
||||
import msgspec
|
||||
import zmq
|
||||
@@ -175,38 +174,38 @@ class CoreEngineProcManager:
|
||||
self.manager_stopped = threading.Event()
|
||||
self.failed_proc_name: str | None = None
|
||||
|
||||
# All ranks share this config object: capture the user-provided
|
||||
# --device-ids list before the per-rank shard overwrites it. Mutating
|
||||
# the config before each proc.start() works because the spawn method
|
||||
# pickles process args at start() time, sequentially per rank.
|
||||
user_assigned_gpu_ids = vllm_config.parallel_config.assigned_physical_gpu_ids
|
||||
try:
|
||||
for proc, local_dp_rank in zip(self.processes, local_dp_ranks):
|
||||
# Adjust device control in DP for platforms that cannot rely
|
||||
# on torch.accelerator.set_device_index(), and for Ray launchers.
|
||||
device_control_context: contextlib.AbstractContextManager[None] = (
|
||||
contextlib.nullcontext()
|
||||
)
|
||||
# Populate the logical-to-physical GPU mapping in DP for
|
||||
# platforms that cannot rely on
|
||||
# torch.accelerator.set_device_index(), and for Ray.
|
||||
needs_device_env_isolation = not (
|
||||
current_platform.is_cuda_alike() or current_platform.is_xpu()
|
||||
)
|
||||
if is_dp and (
|
||||
needs_device_env_isolation or vllm_config.parallel_config.use_ray
|
||||
):
|
||||
device_control_context = set_device_control_env_var(
|
||||
vllm_config, local_dp_rank
|
||||
set_assigned_physical_gpu_ids_for_dp_rank(
|
||||
vllm_config, local_dp_rank, user_assigned_gpu_ids
|
||||
)
|
||||
|
||||
with (
|
||||
device_control_context,
|
||||
numa_utils.configure_subprocess(
|
||||
# EngineCore itself does not have a TP/PP-local rank.
|
||||
# When DP is enabled, set_device_control_env_var()
|
||||
# narrows visible devices to this DP shard first, so
|
||||
# local_rank=0 means "the first local GPU in this
|
||||
# shard". The actual TP/PP worker processes spawned by
|
||||
# the executor are bound separately with their own
|
||||
# local_rank values.
|
||||
vllm_config,
|
||||
local_rank=0,
|
||||
dp_local_rank=local_dp_rank,
|
||||
process_kind="EngineCore",
|
||||
),
|
||||
with numa_utils.configure_subprocess(
|
||||
# EngineCore itself does not have a TP/PP-local rank.
|
||||
# When DP is enabled, set_assigned_physical_gpu_ids_for_dp_rank()
|
||||
# populates the logical-to-physical mapping for this DP
|
||||
# shard, so local_rank=0 means "the first local GPU in
|
||||
# this shard". The actual TP/PP worker processes spawned
|
||||
# by the executor are bound separately with their own
|
||||
# local_rank values.
|
||||
vllm_config,
|
||||
local_rank=0,
|
||||
dp_local_rank=local_dp_rank,
|
||||
process_kind="EngineCore",
|
||||
):
|
||||
proc.start()
|
||||
finally:
|
||||
@@ -281,55 +280,79 @@ class SignalCallback:
|
||||
self._event.set()
|
||||
|
||||
|
||||
@contextlib.contextmanager
|
||||
def set_device_control_env_var(
|
||||
vllm_config: VllmConfig, local_dp_rank: int
|
||||
) -> Iterator[None]:
|
||||
def set_assigned_physical_gpu_ids_for_dp_rank(
|
||||
vllm_config: VllmConfig,
|
||||
local_dp_rank: int,
|
||||
user_assigned_gpu_ids: list[int] | None = None,
|
||||
) -> None:
|
||||
"""
|
||||
Temporarily set CUDA_VISIBLE_DEVICES or equivalent
|
||||
for engine subprocess.
|
||||
Populate assigned_physical_gpu_ids on the config for the given DP rank.
|
||||
|
||||
user_assigned_gpu_ids is the full (un-sharded) --device-ids list, if the
|
||||
user provided one; this DP rank's shard is sliced from it. It is passed
|
||||
explicitly rather than read from the config because callers may reuse
|
||||
one config object across DP ranks, overwriting the field each time.
|
||||
"""
|
||||
world_size = vllm_config.parallel_config.world_size
|
||||
local_world_size = vllm_config.parallel_config.local_world_size
|
||||
evar = current_platform.device_control_env_var
|
||||
|
||||
value = get_device_indices(evar, local_dp_rank, world_size, local_world_size)
|
||||
with patch.dict(os.environ, values=((evar, value),)):
|
||||
yield
|
||||
physical_gpu_ids = get_physical_gpu_ids_for_local_dp_rank(
|
||||
evar,
|
||||
local_dp_rank,
|
||||
world_size,
|
||||
local_world_size,
|
||||
user_assigned_gpu_ids=user_assigned_gpu_ids,
|
||||
)
|
||||
vllm_config.parallel_config.assigned_physical_gpu_ids = physical_gpu_ids
|
||||
|
||||
|
||||
def get_device_indices(
|
||||
def get_physical_gpu_ids_for_local_dp_rank(
|
||||
device_control_env_var: str,
|
||||
local_dp_rank: int,
|
||||
world_size: int,
|
||||
local_world_size: int | None = None,
|
||||
):
|
||||
user_assigned_gpu_ids: list[int] | None = None,
|
||||
) -> list[int]:
|
||||
"""
|
||||
Returns a comma-separated string of device indices for the specified
|
||||
Returns list of physical GPU IDs for the specified
|
||||
data parallel rank.
|
||||
|
||||
For example, if world_size=2 and local_dp_rank=1, and there are 4 devices,
|
||||
this will select devices 2 and 3 for local_dp_rank=1.
|
||||
this will return [2, 3] for local_dp_rank=1.
|
||||
|
||||
If user_assigned_gpu_ids is provided (e.g. from --device-ids), this DP
|
||||
rank's shard is sliced from it instead of being derived from the
|
||||
device-control env var.
|
||||
"""
|
||||
if local_world_size is None:
|
||||
local_world_size = world_size
|
||||
if user_assigned_gpu_ids is not None:
|
||||
start = local_dp_rank * world_size
|
||||
stop = start + local_world_size
|
||||
if stop > len(user_assigned_gpu_ids):
|
||||
raise ValueError(
|
||||
f"--device-ids provides {len(user_assigned_gpu_ids)} devices, "
|
||||
f"but DP rank {local_dp_rank} needs devices [{start}, {stop})"
|
||||
)
|
||||
return user_assigned_gpu_ids[start:stop]
|
||||
try:
|
||||
value = ",".join(
|
||||
str(current_platform.device_id_to_physical_device_id(i))
|
||||
return [
|
||||
current_platform.device_id_to_physical_device_id(i)
|
||||
for i in range(
|
||||
local_dp_rank * world_size,
|
||||
local_dp_rank * world_size + local_world_size,
|
||||
)
|
||||
)
|
||||
]
|
||||
except IndexError as e:
|
||||
raise Exception(
|
||||
f"Error setting {device_control_env_var}: "
|
||||
f"Error computing device indices for "
|
||||
f"{device_control_env_var}: "
|
||||
f"local range: [{local_dp_rank * world_size}, "
|
||||
f"{(local_dp_rank + 1) * world_size}) "
|
||||
"base value: "
|
||||
f'"{os.getenv(device_control_env_var)}"'
|
||||
) from e
|
||||
return value
|
||||
|
||||
|
||||
def _apply_dp_identity_suffix(dp_vllm_config, dp_rank: int) -> None:
|
||||
@@ -453,11 +476,11 @@ class CoreEngineActorManager:
|
||||
# https://github.com/ray-project/ray/blob/master/python/ray/_private/accelerators/intel_gpu.py#L56 # noqa: E501
|
||||
if current_platform.is_xpu():
|
||||
device_evar = current_platform.device_control_env_var
|
||||
device_indices = get_device_indices(
|
||||
physical_gpu_ids = get_physical_gpu_ids_for_local_dp_rank(
|
||||
device_evar, local_index, world_size
|
||||
)
|
||||
actor_env_vars = self.env_vars_dict.copy()
|
||||
actor_env_vars[device_evar] = device_indices
|
||||
actor_env_vars[device_evar] = ",".join(str(d) for d in physical_gpu_ids)
|
||||
runtime_env = RuntimeEnv(env_vars=actor_env_vars)
|
||||
|
||||
actor = (
|
||||
|
||||
@@ -826,6 +826,16 @@ class WorkerProc:
|
||||
signal.signal(signal.SIGTERM, signal_handler)
|
||||
signal.signal(signal.SIGINT, signal_handler)
|
||||
|
||||
# Publish the logical-to-physical mapping early so topology helpers
|
||||
# work before init_device (needed by set_worker_net_device below).
|
||||
assigned_physical_gpu_ids = kwargs[
|
||||
"vllm_config"
|
||||
].parallel_config.assigned_physical_gpu_ids
|
||||
if assigned_physical_gpu_ids is not None:
|
||||
from vllm.platforms.interface import set_assigned_physical_gpu_ids
|
||||
|
||||
set_assigned_physical_gpu_ids(assigned_physical_gpu_ids)
|
||||
|
||||
# Set net device env vars for the worker if VLLM_GPU_NIC_PCIE_MAPPING is set
|
||||
set_worker_net_device(kwargs.get("local_rank", 0), kwargs["vllm_config"])
|
||||
|
||||
|
||||
@@ -258,30 +258,35 @@ class RayDistributedExecutor(Executor):
|
||||
}
|
||||
self.collective_rpc("adjust_rank", args=(rerank_mapping,))
|
||||
|
||||
# Get the set of GPU IDs used on each node.
|
||||
worker_node_and_gpu_ids = []
|
||||
# Get the set of physical GPU IDs used on each node.
|
||||
worker_node_and_physical_gpu_ids = []
|
||||
for worker in [self.driver_dummy_worker] + self.workers:
|
||||
if worker is None:
|
||||
# driver_dummy_worker can be None when using ray spmd worker.
|
||||
continue
|
||||
worker_node_and_gpu_ids.append(
|
||||
ray.get(worker.get_node_and_gpu_ids.remote()) # type: ignore[attr-defined]
|
||||
worker_node_and_physical_gpu_ids.append(
|
||||
ray.get(worker.get_node_and_physical_gpu_ids.remote()) # type: ignore[attr-defined]
|
||||
)
|
||||
|
||||
node_workers = defaultdict(list) # node id -> list of worker ranks
|
||||
node_gpus = defaultdict(list) # node id -> list of gpu ids
|
||||
node_physical_gpu_ids = defaultdict(list) # node id -> physical GPU IDs
|
||||
|
||||
for i, (node_id, gpu_ids) in enumerate(worker_node_and_gpu_ids):
|
||||
for i, (node_id, physical_gpu_ids) in enumerate(
|
||||
worker_node_and_physical_gpu_ids
|
||||
):
|
||||
node_workers[node_id].append(i)
|
||||
# `gpu_ids` can be a list of strings or integers.
|
||||
# `physical_gpu_ids` can be a list of strings or integers.
|
||||
# convert them to integers for consistency.
|
||||
# NOTE: gpu_ids can be larger than 9 (e.g. 16 GPUs),
|
||||
# NOTE: physical GPU IDs can be larger than 9 (e.g. 16 GPUs),
|
||||
# string sorting is not sufficient.
|
||||
# see https://github.com/vllm-project/vllm/issues/5590
|
||||
gpu_ids = [int(x) for x in gpu_ids]
|
||||
node_gpus[node_id].extend(gpu_ids)
|
||||
for node_id, gpu_ids in node_gpus.items():
|
||||
node_gpus[node_id] = sorted(gpu_ids)
|
||||
physical_gpu_ids = [
|
||||
current_platform.device_control_id_to_physical_device_id(str(x))
|
||||
for x in physical_gpu_ids
|
||||
]
|
||||
node_physical_gpu_ids[node_id].extend(physical_gpu_ids)
|
||||
for node_id, physical_gpu_ids in node_physical_gpu_ids.items():
|
||||
node_physical_gpu_ids[node_id] = sorted(physical_gpu_ids)
|
||||
|
||||
all_ips = set(worker_ips + [driver_ip])
|
||||
n_ips = len(all_ips)
|
||||
@@ -297,23 +302,8 @@ class RayDistributedExecutor(Executor):
|
||||
" each node."
|
||||
)
|
||||
|
||||
# Set environment variables for the driver and workers.
|
||||
# We set CUDA_VISIBLE_DEVICES to ALL GPUs on the node for each worker.
|
||||
# This is needed because:
|
||||
# 1. Ray's compiled DAG needs to find the allocated GPU in
|
||||
# CUDA_VISIBLE_DEVICES.
|
||||
# 2. vLLM's communication layer (NCCL, CustomAllreduce) needs to see
|
||||
# all GPUs for P2P checks and communication setup. Though if it was
|
||||
# just this reason, we could have also just kept the visible devices
|
||||
# unset.
|
||||
# Each worker will use local_rank to index into the visible devices.
|
||||
all_args_to_update_environment_variables = [
|
||||
{
|
||||
current_platform.device_control_env_var: ",".join(
|
||||
map(str, node_gpus[node_id])
|
||||
),
|
||||
}
|
||||
for (node_id, _) in worker_node_and_gpu_ids
|
||||
all_args_to_update_environment_variables: list[dict[str, str]] = [
|
||||
{} for _ in worker_node_and_physical_gpu_ids
|
||||
]
|
||||
|
||||
# Environment variables to copy from driver to workers
|
||||
@@ -336,7 +326,7 @@ class RayDistributedExecutor(Executor):
|
||||
"update_environment_variables", args=(self._get_env_vars_to_be_updated(),)
|
||||
)
|
||||
|
||||
if len(node_gpus) == 1:
|
||||
if len(node_physical_gpu_ids) == 1:
|
||||
# in single node case, we don't need to get the IP address.
|
||||
# the loopback address is sufficient
|
||||
# NOTE: a node may have several IP addresses, one for each
|
||||
@@ -352,10 +342,11 @@ class RayDistributedExecutor(Executor):
|
||||
|
||||
# Initialize the actual workers inside worker wrapper.
|
||||
all_kwargs = []
|
||||
for rank, (node_id, _) in enumerate(worker_node_and_gpu_ids):
|
||||
for rank, (node_id, _) in enumerate(worker_node_and_physical_gpu_ids):
|
||||
local_rank = node_workers[node_id].index(rank)
|
||||
kwargs = dict(
|
||||
vllm_config=self.vllm_config,
|
||||
assigned_physical_gpu_ids=sorted(node_physical_gpu_ids[node_id]),
|
||||
local_rank=local_rank,
|
||||
rank=rank,
|
||||
distributed_init_method=distributed_init_method,
|
||||
|
||||
@@ -79,24 +79,25 @@ class RayWorkerProc(WorkerProc):
|
||||
1. __init__: lightweight setup, stores init args (no device/model init)
|
||||
2. initialize_worker: called after GPU IDs are discovered, completes
|
||||
the full WorkerProc initialization with the correct local_rank and
|
||||
CUDA_VISIBLE_DEVICES.
|
||||
logical-to-physical GPU mapping.
|
||||
|
||||
CUDA_VISIBLE_DEVICES setup flow:
|
||||
GPU assignment flow:
|
||||
|
||||
1. RayExecutorV2 enables RAY_EXPERIMENTAL_NOSET_CUDA_VISIBLE_DEVICES so Ray does
|
||||
not set CUDA_VISIBLE_DEVICES on RayWorkerProc actors at creation time.
|
||||
2. Each actor is scheduled with a placement group and bundle index; Ray resolves
|
||||
the physical GPU ID for that bundle at placement time.
|
||||
3. After placement, the worker discovers that GPU ID and sets
|
||||
CUDA_VISIBLE_DEVICES before finishing WorkerProc initialization.
|
||||
3. After placement, the executor discovers each worker's GPU ID and passes the
|
||||
node's logical-to-physical mapping (assigned_physical_gpu_ids) to
|
||||
initialize_worker(); CUDA_VISIBLE_DEVICES is never modified.
|
||||
|
||||
There is no workaround for this unset-and-reset sequence when the placement group
|
||||
is externally managed: scheduling must complete before CUDA_VISIBLE_DEVICES can
|
||||
match the GPU tied to the worker's bundle.
|
||||
Scheduling must complete before the mapping is known when the placement
|
||||
group is externally managed: only then is the GPU tied to the worker's
|
||||
bundle resolved.
|
||||
|
||||
This sequence allows multiple vLLM instances to coexist on the same node:
|
||||
each instance is unaware which physical devices others hold, and the
|
||||
externally managed placement group avoids CUDA_VISIBLE_DEVICES conflicts
|
||||
externally managed placement group avoids device assignment conflicts
|
||||
by binding workers to specific placement group bundles.
|
||||
"""
|
||||
|
||||
@@ -120,28 +121,33 @@ class RayWorkerProc(WorkerProc):
|
||||
is_driver_worker=is_driver_worker,
|
||||
)
|
||||
|
||||
def get_node_and_gpu_ids(self) -> tuple[str, list[int]]:
|
||||
"""Return (node_id, gpu_ids) assigned to this actor by Ray."""
|
||||
def get_node_and_physical_gpu_ids(self) -> tuple[str, list[int]]:
|
||||
"""Return (node_id, physical_gpu_ids) assigned to this actor by Ray."""
|
||||
node_id = ray.get_runtime_context().get_node_id()
|
||||
device_key = current_platform.ray_device_key
|
||||
if not device_key:
|
||||
raise RuntimeError(
|
||||
f"current platform {current_platform.device_name} does not support ray."
|
||||
)
|
||||
gpu_ids = ray.get_runtime_context().get_accelerator_ids()[device_key]
|
||||
return node_id, [int(x) for x in gpu_ids]
|
||||
physical_gpu_ids = ray.get_runtime_context().get_accelerator_ids()[device_key]
|
||||
return node_id, [
|
||||
current_platform.device_control_id_to_physical_device_id(str(x))
|
||||
for x in physical_gpu_ids
|
||||
]
|
||||
|
||||
def initialize_worker(
|
||||
self,
|
||||
local_rank: int,
|
||||
env_vars: dict[str, str],
|
||||
driver_env_vars: dict[str, str] | None = None,
|
||||
assigned_physical_gpu_ids: list[int] | None = None,
|
||||
) -> None:
|
||||
"""Complete initialization after GPU assignment is known.
|
||||
|
||||
*driver_env_vars* are applied with ``setdefault`` — they fill
|
||||
in missing vars but never overwrite node-local values.
|
||||
*env_vars* (e.g. CUDA_VISIBLE_DEVICES) always overwrite.
|
||||
*env_vars* always overwrite.
|
||||
*assigned_physical_gpu_ids* maps local_rank to physical CUDA device ID.
|
||||
"""
|
||||
if driver_env_vars:
|
||||
for key, value in driver_env_vars.items():
|
||||
@@ -149,6 +155,13 @@ class RayWorkerProc(WorkerProc):
|
||||
for key, value in env_vars.items():
|
||||
os.environ[key] = value
|
||||
|
||||
if assigned_physical_gpu_ids is not None:
|
||||
vllm_config = self._init_kwargs["vllm_config"]
|
||||
assert isinstance(vllm_config, VllmConfig)
|
||||
vllm_config.parallel_config.assigned_physical_gpu_ids = (
|
||||
assigned_physical_gpu_ids
|
||||
)
|
||||
|
||||
self.local_rank = local_rank
|
||||
super().__init__(
|
||||
local_rank=local_rank,
|
||||
@@ -365,36 +378,48 @@ class RayExecutorV2(MultiprocExecutor):
|
||||
)
|
||||
self.ray_worker_handles.append(handle)
|
||||
|
||||
# Step 6: Discover GPU IDs assigned to each worker via Ray runtime context.
|
||||
worker_node_and_gpu_ids = ray.get(
|
||||
[h.actor.get_node_and_gpu_ids.remote() for h in self.ray_worker_handles]
|
||||
# Step 6: Discover physical GPU IDs assigned to each worker via Ray
|
||||
# runtime context.
|
||||
worker_node_and_physical_gpu_ids = ray.get(
|
||||
[
|
||||
h.actor.get_node_and_physical_gpu_ids.remote()
|
||||
for h in self.ray_worker_handles
|
||||
]
|
||||
)
|
||||
|
||||
node_workers: dict[str, list[int]] = defaultdict(list)
|
||||
node_gpus: dict[str, list[int]] = defaultdict(list)
|
||||
for i, (node_id, gpu_ids) in enumerate(worker_node_and_gpu_ids):
|
||||
node_physical_gpu_ids: dict[str, list[int]] = defaultdict(list)
|
||||
for i, (node_id, physical_gpu_ids) in enumerate(
|
||||
worker_node_and_physical_gpu_ids
|
||||
):
|
||||
node_workers[node_id].append(i)
|
||||
node_gpus[node_id].extend(gpu_ids)
|
||||
for node_id, gpu_ids in node_gpus.items():
|
||||
node_gpus[node_id] = sorted(gpu_ids)
|
||||
node_physical_gpu_ids[node_id].extend(physical_gpu_ids)
|
||||
for node_id, physical_gpu_ids in node_physical_gpu_ids.items():
|
||||
node_physical_gpu_ids[node_id] = sorted(physical_gpu_ids)
|
||||
|
||||
# Step 7: Initialize workers with correct local_rank and
|
||||
# CUDA_VISIBLE_DEVICES. Each worker sees all GPUs assigned to
|
||||
# this executor on its node; local_rank indexes into that set.
|
||||
# Step 7: Initialize workers with local logical ranks and the
|
||||
# logical-to-physical GPU mapping discovered from Ray placement.
|
||||
init_worker_refs = []
|
||||
for i, (node_id, _) in enumerate(worker_node_and_gpu_ids):
|
||||
for i, (node_id, _) in enumerate(worker_node_and_physical_gpu_ids):
|
||||
local_rank = node_workers[node_id].index(i)
|
||||
worker_env_vars = {
|
||||
current_platform.device_control_env_var: ",".join(
|
||||
map(str, node_gpus[node_id])
|
||||
),
|
||||
}
|
||||
assigned_physical_gpu_ids = sorted(node_physical_gpu_ids[node_id])
|
||||
worker_env_vars: dict[str, str] = {}
|
||||
self.ray_worker_handles[i].local_rank = local_rank
|
||||
init_worker_refs.append(
|
||||
self.ray_worker_handles[i].actor.initialize_worker.remote(
|
||||
local_rank, worker_env_vars, self.driver_env_vars
|
||||
local_rank,
|
||||
worker_env_vars,
|
||||
self.driver_env_vars,
|
||||
assigned_physical_gpu_ids=assigned_physical_gpu_ids,
|
||||
)
|
||||
)
|
||||
# Also set on the executor-side config for consistency. The mapping
|
||||
# is per-node, so only do this when all workers share one node.
|
||||
if len(node_physical_gpu_ids) == 1:
|
||||
node_id_0 = worker_node_and_physical_gpu_ids[0][0]
|
||||
self.vllm_config.parallel_config.assigned_physical_gpu_ids = sorted(
|
||||
node_physical_gpu_ids[node_id_0]
|
||||
)
|
||||
ray.get(init_worker_refs)
|
||||
|
||||
# Step 8: Collect response MQ handles
|
||||
|
||||
@@ -93,7 +93,7 @@ try:
|
||||
def get_node_ip(self) -> str:
|
||||
return get_ip()
|
||||
|
||||
def get_node_and_gpu_ids(self) -> tuple[str, list[int]]:
|
||||
def get_node_and_physical_gpu_ids(self) -> tuple[str, list[int]]:
|
||||
node_id = ray.get_runtime_context().get_node_id()
|
||||
device_key = vllm.platforms.current_platform.ray_device_key
|
||||
if not device_key:
|
||||
@@ -101,8 +101,10 @@ try:
|
||||
"current platform %s does not support ray.",
|
||||
vllm.platforms.current_platform.device_name,
|
||||
)
|
||||
gpu_ids = ray.get_runtime_context().get_accelerator_ids()[device_key]
|
||||
return node_id, gpu_ids
|
||||
physical_gpu_ids = ray.get_runtime_context().get_accelerator_ids()[
|
||||
device_key
|
||||
]
|
||||
return node_id, physical_gpu_ids
|
||||
|
||||
def setup_device_if_necessary(self):
|
||||
# TODO(swang): This is needed right now because Ray CG executes
|
||||
|
||||
@@ -4,7 +4,10 @@ from typing_extensions import override
|
||||
|
||||
from vllm.v1.kv_offload.base import BlockIDsLoadStoreSpec
|
||||
|
||||
METRIC_STORES_SKIPPED = "vllm:kv_offload_stores_skipped"
|
||||
|
||||
class CPUOffloadingMetrics:
|
||||
STORES_SKIPPED = "vllm:kv_offload_stores_skipped"
|
||||
CPU_CACHE_USAGE_PERC = "vllm:kv_offload_cpu_cache_usage_perc"
|
||||
|
||||
|
||||
class CPULoadStoreSpec(BlockIDsLoadStoreSpec):
|
||||
|
||||
@@ -14,7 +14,7 @@ from vllm.logger import init_logger
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.triton_utils import HAS_TRITON, triton
|
||||
from vllm.utils.math_utils import cdiv
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
from vllm.v1.kv_offload.base import (
|
||||
BlockIDsLoadStoreSpec,
|
||||
CanonicalKVCacheRef,
|
||||
@@ -156,7 +156,7 @@ def pin_mmap_region(region: SharedOffloadRegion) -> None:
|
||||
def _new_descriptor_buffers(
|
||||
num_copy_ops: int,
|
||||
) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
pin = is_pin_memory_available()
|
||||
pin = PIN_MEMORY
|
||||
# CUDA cache_kernels.cu requires int64; XPU DMA engine requires uint64.
|
||||
ptr_dtype = torch.uint64 if current_platform.is_xpu() else torch.int64
|
||||
return (
|
||||
@@ -482,7 +482,7 @@ class CpuGpuOffloadingHandlers:
|
||||
num_cpu_blocks: int,
|
||||
mmap_region: SharedOffloadRegion | None = None,
|
||||
):
|
||||
pin_memory = is_pin_memory_available()
|
||||
pin_memory = PIN_MEMORY
|
||||
logger.info("Allocating %d CPU tensors...", len(kv_caches.tensors))
|
||||
self._mmap_region = mmap_region
|
||||
if mmap_region is not None and pin_memory:
|
||||
|
||||
@@ -18,7 +18,10 @@ from vllm.v1.kv_offload.base import (
|
||||
ReqContext,
|
||||
RequestOffloadingContext,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.common import METRIC_STORES_SKIPPED, CPULoadStoreSpec
|
||||
from vllm.v1.kv_offload.cpu.common import (
|
||||
CPULoadStoreSpec,
|
||||
CPUOffloadingMetrics,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.policies.arc import ARCCachePolicy
|
||||
from vllm.v1.kv_offload.cpu.policies.base import BlockStatus, CachePolicy
|
||||
from vllm.v1.kv_offload.cpu.policies.lru import LRUCachePolicy
|
||||
@@ -282,13 +285,21 @@ class CPUOffloadingManager(OffloadingManager):
|
||||
self.events.clear()
|
||||
|
||||
def get_stats(self) -> OffloadingConnectorStats | None:
|
||||
if self.store_threshold < 2:
|
||||
return None
|
||||
|
||||
stats = OffloadingConnectorStats()
|
||||
stats.increase_counter(
|
||||
METRIC_STORES_SKIPPED,
|
||||
self.stores_skipped_in_current_batch,
|
||||
|
||||
# Compute cache usage.
|
||||
num_used = (
|
||||
self._num_allocated_blocks
|
||||
- len(self._free_list)
|
||||
- self._num_evictable_cache_blocks
|
||||
)
|
||||
self.stores_skipped_in_current_batch = 0
|
||||
usage = num_used / self._num_blocks if self._num_blocks > 0 else 0.0
|
||||
stats.set_gauge(CPUOffloadingMetrics.CPU_CACHE_USAGE_PERC, usage)
|
||||
|
||||
if self.store_threshold >= 2:
|
||||
stats.increase_counter(
|
||||
CPUOffloadingMetrics.STORES_SKIPPED,
|
||||
self.stores_skipped_in_current_batch,
|
||||
)
|
||||
self.stores_skipped_in_current_batch = 0
|
||||
return stats
|
||||
|
||||
@@ -14,11 +14,15 @@ from vllm.v1.kv_offload.base import (
|
||||
GPULoadStoreSpec,
|
||||
LoadStoreSpec,
|
||||
OffloadingCounterMetadata,
|
||||
OffloadingGaugeMetadata,
|
||||
OffloadingManager,
|
||||
OffloadingMetricMetadata,
|
||||
OffloadingSpec,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.common import METRIC_STORES_SKIPPED, CPULoadStoreSpec
|
||||
from vllm.v1.kv_offload.cpu.common import (
|
||||
CPULoadStoreSpec,
|
||||
CPUOffloadingMetrics,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.gpu_worker import CpuGpuOffloadingHandlers
|
||||
from vllm.v1.kv_offload.cpu.manager import CPUOffloadingManager
|
||||
from vllm.v1.kv_offload.worker.worker import OffloadingHandler
|
||||
@@ -31,17 +35,27 @@ class CPUOffloadingSpec(OffloadingSpec):
|
||||
def build_metric_definitions(
|
||||
cls, extra_config: dict[str, Any]
|
||||
) -> dict[str, OffloadingMetricMetadata]:
|
||||
store_threshold = int(extra_config.get("store_threshold", 0))
|
||||
if store_threshold < 2:
|
||||
return {}
|
||||
return {
|
||||
METRIC_STORES_SKIPPED: OffloadingCounterMetadata(
|
||||
definitions: dict[str, OffloadingMetricMetadata] = {
|
||||
CPUOffloadingMetrics.CPU_CACHE_USAGE_PERC: OffloadingGaugeMetadata(
|
||||
documentation=(
|
||||
"Number of KV offload stores skipped because the reuse "
|
||||
"threshold was not reached."
|
||||
"Fraction of CPU KV-cache space currently pinned by active "
|
||||
"transfers (0.0 = idle, 1.0 = saturated). Sustained high "
|
||||
"values indicate transfers (stores or promotions) may be "
|
||||
"dropped due to insufficient capacity."
|
||||
),
|
||||
)
|
||||
}
|
||||
store_threshold = int(extra_config.get("store_threshold", 0))
|
||||
if store_threshold >= 2:
|
||||
definitions[CPUOffloadingMetrics.STORES_SKIPPED] = (
|
||||
OffloadingCounterMetadata(
|
||||
documentation=(
|
||||
"Number of KV offload stores skipped because the reuse "
|
||||
"threshold was not reached."
|
||||
),
|
||||
)
|
||||
)
|
||||
return definitions
|
||||
|
||||
def __init__(self, vllm_config: VllmConfig, kv_cache_config: KVCacheConfig):
|
||||
super().__init__(vllm_config, kv_cache_config)
|
||||
@@ -58,7 +72,15 @@ class CPUOffloadingSpec(OffloadingSpec):
|
||||
self.cpu_page_size_per_worker = 0
|
||||
assert kv_cache_config is not None
|
||||
if kv_cache_config.num_blocks > 0 and world_size > 0:
|
||||
total_gpu_kv_bytes = sum(t.size for t in kv_cache_config.kv_cache_tensors)
|
||||
is_packed = any(t.block_stride for t in kv_cache_config.kv_cache_tensors)
|
||||
assert not is_packed or all(
|
||||
t.block_stride for t in kv_cache_config.kv_cache_tensors
|
||||
)
|
||||
total_gpu_kv_bytes = (
|
||||
kv_cache_config.kv_cache_tensors[0].size
|
||||
if is_packed
|
||||
else sum(t.size for t in kv_cache_config.kv_cache_tensors)
|
||||
)
|
||||
kv_bytes_per_block = (
|
||||
total_gpu_kv_bytes // kv_cache_config.num_blocks
|
||||
) * world_size
|
||||
|
||||
@@ -7,9 +7,7 @@ import torch
|
||||
|
||||
from vllm.pooling_params import PoolingParams
|
||||
from vllm.tasks import PoolingTask
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
|
||||
pin_memory = is_pin_memory_available()
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -134,7 +132,7 @@ class PoolingMetadata:
|
||||
num_scheduled_tokens_cpu = torch.from_numpy(num_scheduled_tokens_np)
|
||||
if query_start_loc_gpu is None:
|
||||
cumsum = torch.zeros(
|
||||
n_seq + 1, dtype=torch.int64, pin_memory=pin_memory, device="cpu"
|
||||
n_seq + 1, dtype=torch.int64, pin_memory=PIN_MEMORY, device="cpu"
|
||||
)
|
||||
torch.cumsum(num_scheduled_tokens_cpu, dim=0, out=cumsum[1:])
|
||||
cumsum = cumsum.to(device, non_blocking=True)
|
||||
|
||||
@@ -7,6 +7,7 @@ import numpy as np
|
||||
import torch
|
||||
|
||||
from vllm import SamplingParams
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.v1.sample.logits_processor.interface import (
|
||||
BatchUpdate,
|
||||
LogitsProcessor,
|
||||
@@ -118,7 +119,6 @@ class MinPLogitsProcessor(LogitsProcessor):
|
||||
class LogitBiasLogitsProcessor(LogitsProcessor):
|
||||
def __init__(self, _, device: torch.device, is_pin_memory: bool):
|
||||
self.device = device
|
||||
self.pin_memory = is_pin_memory
|
||||
self.biases: dict[int, dict[int, float]] = {}
|
||||
|
||||
self.bias_tensor: torch.Tensor = torch.tensor(())
|
||||
@@ -154,9 +154,7 @@ class LogitBiasLogitsProcessor(LogitsProcessor):
|
||||
)
|
||||
|
||||
def _device_tensor(self, data: list, dtype: torch.dtype) -> torch.Tensor:
|
||||
return torch.tensor(
|
||||
data, device="cpu", dtype=dtype, pin_memory=self.pin_memory
|
||||
).to(device=self.device, non_blocking=True)
|
||||
return async_tensor_h2d(data, device=self.device, dtype=dtype)
|
||||
|
||||
def apply(self, logits: torch.Tensor) -> torch.Tensor:
|
||||
if self.biases:
|
||||
@@ -170,7 +168,6 @@ class MinTokensLogitsProcessor(LogitsProcessor):
|
||||
):
|
||||
# index -> (min_toks, output_token_ids, stop_token_ids)
|
||||
self.device = device
|
||||
self.pin_memory = is_pin_memory
|
||||
self.min_toks: dict[int, tuple[int, Sequence[int], set[int]]] = {}
|
||||
|
||||
# (req_idx_tensor,eos_tok_id_tensor)
|
||||
@@ -227,9 +224,7 @@ class MinTokensLogitsProcessor(LogitsProcessor):
|
||||
)
|
||||
|
||||
def _device_tensor(self, data: list, dtype: torch.dtype) -> torch.Tensor:
|
||||
return torch.tensor(
|
||||
data, device="cpu", dtype=dtype, pin_memory=self.pin_memory
|
||||
).to(device=self.device, non_blocking=True)
|
||||
return async_tensor_h2d(data, device=self.device, dtype=dtype)
|
||||
|
||||
def apply(self, logits: torch.Tensor) -> torch.Tensor:
|
||||
if self.min_toks:
|
||||
@@ -283,8 +278,8 @@ class MinTokensLogitsProcessor(LogitsProcessor):
|
||||
toks_arr = np.concatenate(all_toks)
|
||||
# (row_indices, token_indices) for index_put_ to set -inf.
|
||||
logits_slice = (
|
||||
torch.from_numpy(rows_arr).to(self.device, non_blocking=True),
|
||||
torch.from_numpy(toks_arr).to(self.device, non_blocking=True),
|
||||
async_tensor_h2d(rows_arr, device=self.device),
|
||||
async_tensor_h2d(toks_arr, device=self.device),
|
||||
)
|
||||
logits.index_put_(logits_slice, self.neg_inf_tensor)
|
||||
|
||||
|
||||
@@ -4,8 +4,7 @@
|
||||
import torch
|
||||
|
||||
from vllm.model_executor.layers.utils import apply_penalties
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import make_tensor_with_pad
|
||||
from vllm.utils.torch_utils import PIN_MEMORY, make_tensor_with_pad
|
||||
|
||||
|
||||
def apply_all_penalties(
|
||||
@@ -52,6 +51,6 @@ def _convert_to_tensors(
|
||||
pad=vocab_size,
|
||||
device="cpu",
|
||||
dtype=torch.int64,
|
||||
pin_memory=is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
return output_tokens_tensor.to(device, non_blocking=True)
|
||||
|
||||
@@ -6,7 +6,7 @@ import torch
|
||||
import torch.nn as nn
|
||||
|
||||
from vllm.config.model import LogprobsMode
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
from vllm.v1.outputs import LogprobsTensors, SamplerOutput
|
||||
from vllm.v1.sample.metadata import SamplingMetadata
|
||||
from vllm.v1.sample.ops.bad_words import apply_bad_words
|
||||
@@ -65,7 +65,7 @@ class Sampler(nn.Module):
|
||||
):
|
||||
super().__init__()
|
||||
self.topk_topp_sampler = TopKTopPSampler(logprobs_mode, use_fp64_gumbel)
|
||||
self.pin_memory = is_pin_memory_available()
|
||||
self.pin_memory = PIN_MEMORY
|
||||
self.logprobs_mode = logprobs_mode
|
||||
self.use_fp64_gumbel = use_fp64_gumbel
|
||||
|
||||
|
||||
@@ -7,7 +7,7 @@ from typing import TYPE_CHECKING, Any
|
||||
import torch
|
||||
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.utils.torch_utils import PIN_MEMORY, async_tensor_h2d
|
||||
from vllm.v1.sample.logits_processor.interface import (
|
||||
BatchUpdate,
|
||||
MoveDirectionality,
|
||||
@@ -22,12 +22,11 @@ def maybe_create_thinking_budget_state_holder(
|
||||
max_num_seqs: int,
|
||||
num_spec_tokens: int,
|
||||
device: torch.device,
|
||||
is_pin_memory: bool,
|
||||
) -> "ThinkingBudgetStateHolder | None":
|
||||
if reasoning_config is None:
|
||||
return None
|
||||
return ThinkingBudgetStateHolder(
|
||||
reasoning_config, max_num_seqs, num_spec_tokens, device, is_pin_memory
|
||||
reasoning_config, max_num_seqs, num_spec_tokens, device, PIN_MEMORY
|
||||
)
|
||||
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user