forked from Karylab-cklius/vllm
Compare commits
17
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7fc60bb26d | ||
|
|
1f2c614c27 | ||
|
|
183a430c13 | ||
|
|
a346d589f5 | ||
|
|
7df3d7dada | ||
|
|
8dd1b702f2 | ||
|
|
f57ac274b2 | ||
|
|
6e919960af | ||
|
|
c88d3d4775 | ||
|
|
ab7fcbdd5d | ||
|
|
3b4a76b63f | ||
|
|
cc22621b51 | ||
|
|
77148992cf | ||
|
|
891cc4b9c5 | ||
|
|
1bdf9810aa | ||
|
|
f24d8d5bb4 | ||
|
|
928e13af5f |
@@ -109,6 +109,7 @@ steps:
|
||||
- image-build-amd
|
||||
commands:
|
||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- export PYTORCH_ROCM_ARCH=gfx942 # Limit Quark compilation to save time
|
||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-mi3xx.txt
|
||||
|
||||
- label: MoE Refactor Integration Test (H100 - TEMPORARY)
|
||||
|
||||
@@ -8,6 +8,8 @@ AnthropicServingMessages._convert_anthropic_to_openai_request().
|
||||
Also covers extended-thinking edge cases such as ``redacted_thinking``
|
||||
blocks echoed back by Anthropic clients, and streaming conversion in
|
||||
``message_stream_converter``.
|
||||
|
||||
Also covers cache usage computation in ``_build_anthropic_usage``.
|
||||
"""
|
||||
|
||||
import json
|
||||
@@ -18,7 +20,11 @@ import pytest
|
||||
from vllm.entrypoints.anthropic.protocol import (
|
||||
AnthropicMessagesRequest,
|
||||
)
|
||||
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
|
||||
from vllm.entrypoints.anthropic.serving import (
|
||||
AnthropicServingMessages,
|
||||
_build_anthropic_usage,
|
||||
_get_cached_tokens,
|
||||
)
|
||||
from vllm.entrypoints.openai.chat_completion.protocol import (
|
||||
ChatCompletionResponseStreamChoice,
|
||||
ChatCompletionStreamResponse,
|
||||
@@ -27,6 +33,7 @@ from vllm.entrypoints.openai.engine.protocol import (
|
||||
DeltaFunctionCall,
|
||||
DeltaMessage,
|
||||
DeltaToolCall,
|
||||
PromptTokenUsageInfo,
|
||||
UsageInfo,
|
||||
)
|
||||
|
||||
@@ -653,6 +660,108 @@ class TestThinkingBlockConversion:
|
||||
assert asst.get("content") == "Hi!"
|
||||
|
||||
|
||||
# ======================================================================
|
||||
# Cache usage computation
|
||||
# ======================================================================
|
||||
|
||||
|
||||
class TestGetCachedTokens:
|
||||
"""Tests for _get_cached_tokens helper."""
|
||||
|
||||
def test_none_usage(self):
|
||||
assert _get_cached_tokens(None) is None
|
||||
|
||||
def test_no_prompt_tokens_details(self):
|
||||
usage = UsageInfo(prompt_tokens=100, completion_tokens=10)
|
||||
assert _get_cached_tokens(usage) is None
|
||||
|
||||
def test_cached_tokens_present(self):
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=80),
|
||||
)
|
||||
assert _get_cached_tokens(usage) == 80
|
||||
|
||||
def test_cached_tokens_zero(self):
|
||||
"""Zero cached tokens should return 0, not None."""
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=0),
|
||||
)
|
||||
assert _get_cached_tokens(usage) == 0
|
||||
|
||||
def test_cached_tokens_none_in_details(self):
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=None),
|
||||
)
|
||||
assert _get_cached_tokens(usage) is None
|
||||
|
||||
|
||||
class TestBuildAnthropicUsage:
|
||||
"""Tests for _build_anthropic_usage helper.
|
||||
|
||||
Anthropic defines: total_input = input_tokens + cache_read + cache_creation
|
||||
vLLM's prompt_tokens is the total.
|
||||
"""
|
||||
|
||||
def test_no_cache_info(self):
|
||||
"""When cache info is unavailable, return raw prompt_tokens."""
|
||||
result = _build_anthropic_usage(100, 10, None)
|
||||
assert result.input_tokens == 100
|
||||
assert result.output_tokens == 10
|
||||
assert result.cache_read_input_tokens is None
|
||||
assert result.cache_creation_input_tokens is None
|
||||
|
||||
def test_cache_hit(self):
|
||||
"""When cache is hit, input_tokens excludes cached tokens."""
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=80),
|
||||
)
|
||||
result = _build_anthropic_usage(100, 10, usage)
|
||||
assert result.input_tokens == 20 # 100 - 80
|
||||
assert result.output_tokens == 10
|
||||
assert result.cache_read_input_tokens == 80
|
||||
assert result.cache_creation_input_tokens == 0
|
||||
|
||||
def test_zero_cached_tokens(self):
|
||||
"""Zero cached tokens should still set cache_creation to 0."""
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=0),
|
||||
)
|
||||
result = _build_anthropic_usage(100, 10, usage)
|
||||
assert result.input_tokens == 100 # 100 - 0
|
||||
assert result.cache_read_input_tokens == 0
|
||||
assert result.cache_creation_input_tokens == 0
|
||||
|
||||
def test_all_tokens_cached(self):
|
||||
"""When all tokens are cached, input_tokens should be 0."""
|
||||
usage = UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=10,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=100),
|
||||
)
|
||||
result = _build_anthropic_usage(100, 10, usage)
|
||||
assert result.input_tokens == 0
|
||||
assert result.cache_read_input_tokens == 100
|
||||
assert result.cache_creation_input_tokens == 0
|
||||
|
||||
def test_no_prompt_tokens_details(self):
|
||||
"""UsageInfo without prompt_tokens_details returns no cache info."""
|
||||
usage = UsageInfo(prompt_tokens=100, completion_tokens=10)
|
||||
result = _build_anthropic_usage(100, 10, usage)
|
||||
assert result.input_tokens == 100
|
||||
assert result.cache_read_input_tokens is None
|
||||
assert result.cache_creation_input_tokens is None
|
||||
|
||||
|
||||
class TestInlineSystemMessageInMessagesArray:
|
||||
"""Verify that ``role: system`` messages embedded inside the ``messages``
|
||||
array are preserved in their original position.
|
||||
@@ -1098,6 +1207,135 @@ class TestMessageStartIncludesTypeAndRole:
|
||||
assert message["role"] == "assistant"
|
||||
|
||||
|
||||
class TestStreamingCacheUsageSemantics:
|
||||
"""Locks in the documented streaming behavior of cache usage fields.
|
||||
|
||||
vLLM's OpenAI chat completion streaming only attaches
|
||||
``prompt_tokens_details`` to the terminal usage chunk. The Anthropic layer
|
||||
mirrors that contract: cache fields are omitted on ``message_start`` (key
|
||||
absence signals "unknown") and populated on ``message_delta`` (the final
|
||||
cumulative count). This is intentionally consistent with vLLM's OpenAI
|
||||
behavior, even though Anthropic's upstream API populates cache fields on
|
||||
``message_start``; closing that gap requires plumbing cache info into the
|
||||
first chunk at the OpenAI layer, which is out of scope here.
|
||||
"""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_streaming_cache_fields_absent_then_populated(self):
|
||||
"""First chunk lacks prompt_tokens_details (vLLM contract);
|
||||
message_start omits cache fields. The final chunk carries
|
||||
prompt_tokens_details, so message_delta carries resolved values."""
|
||||
|
||||
async def sse_input():
|
||||
yield _make_stream_chunk(
|
||||
delta=DeltaMessage(role="assistant", content="hi"),
|
||||
usage=UsageInfo(prompt_tokens=100, total_tokens=100),
|
||||
)
|
||||
yield _make_stream_chunk(finish_reason="stop")
|
||||
yield _make_stream_chunk(
|
||||
choices=[],
|
||||
usage=UsageInfo(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=5,
|
||||
total_tokens=105,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=80),
|
||||
),
|
||||
)
|
||||
yield "data: [DONE]"
|
||||
|
||||
converter = _make_stream_converter()
|
||||
output = []
|
||||
async for event in converter.message_stream_converter(sse_input()):
|
||||
output.append(event)
|
||||
events = _parse_sse_events(output)
|
||||
|
||||
# message_start: cache fields unknown → omitted from JSON entirely.
|
||||
start_usage = events[0][1]["message"]["usage"]
|
||||
assert events[0][0] == "message_start"
|
||||
assert start_usage["input_tokens"] == 100
|
||||
assert "cache_read_input_tokens" not in start_usage
|
||||
assert "cache_creation_input_tokens" not in start_usage
|
||||
|
||||
# message_delta: authoritative usage with cache fields populated.
|
||||
delta_usage = next(
|
||||
data["usage"] for ev, data in events if ev == "message_delta"
|
||||
)
|
||||
assert delta_usage["input_tokens"] == 20 # 100 - 80
|
||||
assert delta_usage["cache_read_input_tokens"] == 80
|
||||
assert delta_usage["cache_creation_input_tokens"] == 0
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_streaming_no_cache_hit(self):
|
||||
"""When the final chunk reports cached_tokens=0, message_delta carries
|
||||
cache fields = 0 (cache miss); message_start still omits them."""
|
||||
|
||||
async def sse_input():
|
||||
yield _make_stream_chunk(
|
||||
delta=DeltaMessage(role="assistant"),
|
||||
usage=UsageInfo(prompt_tokens=50, total_tokens=50),
|
||||
)
|
||||
yield _make_stream_chunk(finish_reason="stop")
|
||||
yield _make_stream_chunk(
|
||||
choices=[],
|
||||
usage=UsageInfo(
|
||||
prompt_tokens=50,
|
||||
completion_tokens=5,
|
||||
total_tokens=55,
|
||||
prompt_tokens_details=PromptTokenUsageInfo(cached_tokens=0),
|
||||
),
|
||||
)
|
||||
yield "data: [DONE]"
|
||||
|
||||
converter = _make_stream_converter()
|
||||
output = []
|
||||
async for event in converter.message_stream_converter(sse_input()):
|
||||
output.append(event)
|
||||
events = _parse_sse_events(output)
|
||||
|
||||
start_usage = events[0][1]["message"]["usage"]
|
||||
delta_usage = next(
|
||||
data["usage"] for ev, data in events if ev == "message_delta"
|
||||
)
|
||||
assert start_usage["input_tokens"] == 50
|
||||
assert "cache_read_input_tokens" not in start_usage
|
||||
assert "cache_creation_input_tokens" not in start_usage
|
||||
assert delta_usage["input_tokens"] == 50 # 50 - 0
|
||||
assert delta_usage["cache_read_input_tokens"] == 0
|
||||
assert delta_usage["cache_creation_input_tokens"] == 0
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_streaming_no_prompt_tokens_details_at_all(self):
|
||||
"""If --enable-prompt-tokens-details is off, no chunk carries cache
|
||||
info; both message_start and message_delta omit cache fields."""
|
||||
|
||||
async def sse_input():
|
||||
yield _make_stream_chunk(
|
||||
delta=DeltaMessage(role="assistant"),
|
||||
usage=UsageInfo(prompt_tokens=30, total_tokens=30),
|
||||
)
|
||||
yield _make_stream_chunk(finish_reason="stop")
|
||||
yield _make_stream_chunk(
|
||||
choices=[],
|
||||
usage=UsageInfo(prompt_tokens=30, completion_tokens=2, total_tokens=32),
|
||||
)
|
||||
yield "data: [DONE]"
|
||||
|
||||
converter = _make_stream_converter()
|
||||
output = []
|
||||
async for event in converter.message_stream_converter(sse_input()):
|
||||
output.append(event)
|
||||
events = _parse_sse_events(output)
|
||||
|
||||
start_usage = events[0][1]["message"]["usage"]
|
||||
delta_usage = next(
|
||||
data["usage"] for ev, data in events if ev == "message_delta"
|
||||
)
|
||||
assert "cache_read_input_tokens" not in start_usage
|
||||
assert "cache_creation_input_tokens" not in start_usage
|
||||
assert "cache_read_input_tokens" not in delta_usage
|
||||
assert "cache_creation_input_tokens" not in delta_usage
|
||||
|
||||
|
||||
# ======================================================================
|
||||
# Auto-detection of system-first template requirement
|
||||
# ======================================================================
|
||||
|
||||
@@ -8,6 +8,7 @@ import pytest
|
||||
import pytest_asyncio
|
||||
|
||||
from tests.utils import RemoteLaunchRenderServer
|
||||
from vllm.tokenizers import get_tokenizer
|
||||
|
||||
MODEL_NAME = "hmellor/tiny-random-LlamaForCausalLM"
|
||||
|
||||
@@ -486,3 +487,438 @@ async def test_derender_completion_kv_transfer_params_passthrough(client):
|
||||
)
|
||||
assert response.status_code == 200
|
||||
assert response.json()["kv_transfer_params"] == kv
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# E2E: render -> derender roundtrip with parser (reasoning + tool calls)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
PARSER_MODEL = "deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B"
|
||||
|
||||
_E2E_TOOLS = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"description": "Get weather for a city",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"city": {"type": "string"}},
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def parser_server():
|
||||
args = [
|
||||
"--enable-auto-tool-choice",
|
||||
"--tool-call-parser",
|
||||
"hermes",
|
||||
"--reasoning-parser",
|
||||
"deepseek_r1",
|
||||
]
|
||||
with RemoteLaunchRenderServer(PARSER_MODEL, args) as remote_server:
|
||||
yield remote_server
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def parser_client(parser_server):
|
||||
async with httpx.AsyncClient(
|
||||
base_url=parser_server.url_for(""), timeout=60.0
|
||||
) as http_client:
|
||||
yield http_client
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def parser_tokenizer():
|
||||
return get_tokenizer(PARSER_MODEL)
|
||||
|
||||
|
||||
def _encode(tokenizer, text: str) -> list[int]:
|
||||
return tokenizer.encode(text, add_special_tokens=False)
|
||||
|
||||
|
||||
def _decoded(tokenizer, token_ids: list[int]) -> str:
|
||||
return tokenizer.decode(token_ids, skip_special_tokens=True)
|
||||
|
||||
|
||||
def _require_markers_survive(tokenizer, text: str, *markers: str) -> list[int]:
|
||||
"""Encode text and skip the test if any marker is lost in roundtrip."""
|
||||
ids = _encode(tokenizer, text)
|
||||
decoded = tokenizer.decode(ids, skip_special_tokens=False)
|
||||
for m in markers:
|
||||
if m not in decoded:
|
||||
pytest.skip(f"Marker {m!r} lost in encode->decode roundtrip")
|
||||
return ids
|
||||
|
||||
|
||||
async def _e2e_render_chat(
|
||||
client: httpx.AsyncClient,
|
||||
model: str,
|
||||
messages: list[dict],
|
||||
) -> dict:
|
||||
resp = await client.post(
|
||||
"/v1/chat/completions/render",
|
||||
json={"model": model, "messages": messages},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
return resp.json()
|
||||
|
||||
|
||||
def _e2e_generate_response(
|
||||
token_ids: list[int],
|
||||
request_id: str = "chatcmpl-e2e-test",
|
||||
) -> dict:
|
||||
return {
|
||||
"request_id": request_id,
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"token_ids": token_ids,
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_plain_roundtrip(parser_client, parser_tokenizer):
|
||||
"""Plain text without reasoning markers roundtrips correctly."""
|
||||
messages = [{"role": "user", "content": "What is 2+2?"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
answer = "The answer is four."
|
||||
output_ids = _encode(parser_tokenizer, answer)
|
||||
expected = _decoded(parser_tokenizer, output_ids)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
content = resp.json()["choices"][0]["message"]["content"]
|
||||
assert content == expected
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_token_identity(parser_client, parser_tokenizer):
|
||||
"""encode(derender(token_ids)) == token_ids (RL invariant)."""
|
||||
messages = [{"role": "user", "content": "Hi"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
answer = "Hello! How can I help?"
|
||||
output_ids = _encode(parser_tokenizer, answer)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
content = resp.json()["choices"][0]["message"]["content"]
|
||||
re_encoded = _encode(parser_tokenizer, content)
|
||||
assert output_ids == re_encoded
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_non_ascii_roundtrip(parser_client, parser_tokenizer):
|
||||
"""CJK + emoji roundtrip without U+FFFD."""
|
||||
messages = [{"role": "user", "content": "Reply in Chinese"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
answer = "你好世界 😀"
|
||||
output_ids = _encode(parser_tokenizer, answer)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
content = resp.json()["choices"][0]["message"]["content"]
|
||||
assert "�" not in content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_parsed_reasoning(parser_client, parser_tokenizer):
|
||||
"""<think>...</think> splits into reasoning + content."""
|
||||
messages = [{"role": "user", "content": "What is 2+3?"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
reasoning_text = "The user wants 2 plus 3. That is 5."
|
||||
answer_text = "The answer is 5."
|
||||
output_text = f"<think>{reasoning_text}</think>{answer_text}"
|
||||
output_ids = _require_markers_survive(parser_tokenizer, output_text, "</think>")
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
"chat_request": {
|
||||
"model": PARSER_MODEL,
|
||||
"messages": messages,
|
||||
"include_reasoning": True,
|
||||
},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
msg = resp.json()["choices"][0]["message"]
|
||||
assert msg["reasoning"] is not None
|
||||
assert reasoning_text in msg["reasoning"]
|
||||
assert answer_text in msg["content"]
|
||||
assert "<think>" not in msg["content"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_parsed_tool_call(parser_client, parser_tokenizer):
|
||||
"""<tool_call> extracted into tool_calls field."""
|
||||
messages = [{"role": "user", "content": "Weather in Paris?"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
output_text = (
|
||||
"<think>Let me check the weather.</think>"
|
||||
'<tool_call>\n{"name": "get_weather", '
|
||||
'"arguments": {"city": "Paris"}}\n</tool_call>'
|
||||
)
|
||||
output_ids = _require_markers_survive(
|
||||
parser_tokenizer,
|
||||
output_text,
|
||||
"</think>",
|
||||
"<tool_call>",
|
||||
"</tool_call>",
|
||||
)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
"chat_request": {
|
||||
"model": PARSER_MODEL,
|
||||
"messages": messages,
|
||||
"tools": _E2E_TOOLS,
|
||||
"tool_choice": "auto",
|
||||
},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
choice = resp.json()["choices"][0]
|
||||
assert choice["message"]["tool_calls"]
|
||||
assert choice["message"]["tool_calls"][0]["function"]["name"] == "get_weather"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_parsed_reasoning_and_tool_call(parser_client, parser_tokenizer):
|
||||
"""Reasoning + tool call in the same output."""
|
||||
messages = [{"role": "user", "content": "Weather in Paris?"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
reasoning_text = "I should look up the weather."
|
||||
tool_text = (
|
||||
'<tool_call>\n{"name": "get_weather", '
|
||||
'"arguments": {"city": "Paris"}}\n</tool_call>'
|
||||
)
|
||||
output_text = f"<think>{reasoning_text}</think>{tool_text}"
|
||||
output_ids = _require_markers_survive(
|
||||
parser_tokenizer, output_text, "</think>", "<tool_call>"
|
||||
)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
"chat_request": {
|
||||
"model": PARSER_MODEL,
|
||||
"messages": messages,
|
||||
"tools": _E2E_TOOLS,
|
||||
"tool_choice": "auto",
|
||||
"include_reasoning": True,
|
||||
},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
choice = resp.json()["choices"][0]
|
||||
assert choice["message"]["reasoning"] is not None
|
||||
assert reasoning_text in choice["message"]["reasoning"]
|
||||
assert choice["message"]["tool_calls"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_no_chat_request_fallback(parser_client, parser_tokenizer):
|
||||
"""Without chat_request, derender falls back to plain detokenization."""
|
||||
messages = [{"role": "user", "content": "Hello"}]
|
||||
gen_req = await _e2e_render_chat(parser_client, PARSER_MODEL, messages)
|
||||
|
||||
answer = "Hi there!"
|
||||
output_ids = _encode(parser_tokenizer, answer)
|
||||
|
||||
resp = await parser_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": PARSER_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
content = resp.json()["choices"][0]["message"]["content"]
|
||||
assert "Hi" in content
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# E2E: HarmonyParser + GPT-OSS
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
HARMONY_MODEL = "openai/gpt-oss-20b"
|
||||
|
||||
|
||||
def _ensure_harmony_vocab():
|
||||
"""Pre-cache the o200k_base BPE file needed by openai-harmony.
|
||||
|
||||
The Rust tiktoken-rs backend downloads from Azure Blob Storage, which
|
||||
may be unreachable in some environments. When the cache is cold we
|
||||
fetch the file ourselves and place it in ``/tmp/tiktoken-rs-cache/``
|
||||
using the SHA-1(URL) filename that tiktoken-rs expects.
|
||||
"""
|
||||
import hashlib
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
url = "https://openaipublic.blob.core.windows.net/encodings/o200k_base.tiktoken"
|
||||
cache_dir = Path("/tmp/tiktoken-rs-cache")
|
||||
cache_key = hashlib.sha1(url.encode()).hexdigest()
|
||||
cache_file = cache_dir / cache_key
|
||||
if not cache_file.exists():
|
||||
cache_dir.mkdir(parents=True, exist_ok=True)
|
||||
urllib.request.urlretrieve(url, cache_file)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def harmony_server():
|
||||
_ensure_harmony_vocab()
|
||||
args = [
|
||||
"--trust-remote-code",
|
||||
"--enable-auto-tool-choice",
|
||||
"--tool-call-parser",
|
||||
"openai",
|
||||
"--reasoning-parser",
|
||||
"openai_gptoss",
|
||||
]
|
||||
with RemoteLaunchRenderServer(HARMONY_MODEL, args) as remote_server:
|
||||
yield remote_server
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def harmony_client(harmony_server):
|
||||
async with httpx.AsyncClient(
|
||||
base_url=harmony_server.url_for(""), timeout=60.0
|
||||
) as http_client:
|
||||
yield http_client
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def harmony_tokenizer():
|
||||
return get_tokenizer(HARMONY_MODEL, trust_remote_code=True)
|
||||
|
||||
|
||||
def _harmony_extract_assistant_ids(
|
||||
tokenizer, assistant_msg: dict, user_content: str = "test"
|
||||
) -> list[int]:
|
||||
"""Extract assistant token IDs via apply_chat_template diff."""
|
||||
prompt = [{"role": "user", "content": user_content}]
|
||||
full = prompt + [assistant_msg]
|
||||
text_prompt = tokenizer.apply_chat_template(
|
||||
prompt, add_generation_prompt=True, tokenize=False
|
||||
)
|
||||
text_full = tokenizer.apply_chat_template(
|
||||
full, add_generation_prompt=False, tokenize=False
|
||||
)
|
||||
prompt_ids = tokenizer.encode(text_prompt)
|
||||
full_ids = tokenizer.encode(text_full)
|
||||
assistant_ids = list(full_ids[len(prompt_ids) :])
|
||||
if not assistant_ids:
|
||||
pytest.skip("Could not extract assistant tokens for Harmony")
|
||||
return assistant_ids
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_harmony_plain_roundtrip(harmony_client, harmony_tokenizer):
|
||||
"""GPT-OSS content-only roundtrip."""
|
||||
messages = [{"role": "user", "content": "What is 2+2?"}]
|
||||
gen_req = await _e2e_render_chat(harmony_client, HARMONY_MODEL, messages)
|
||||
|
||||
assistant_msg = {"role": "assistant", "content": "Four."}
|
||||
output_ids = _harmony_extract_assistant_ids(harmony_tokenizer, assistant_msg)
|
||||
|
||||
resp = await harmony_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": HARMONY_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
"chat_request": {
|
||||
"model": HARMONY_MODEL,
|
||||
"messages": messages,
|
||||
},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
content = resp.json()["choices"][0]["message"]["content"]
|
||||
assert content is not None and len(content) > 0
|
||||
assert "Four" in content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_harmony_reasoning(harmony_client, harmony_tokenizer):
|
||||
"""GPT-OSS reasoning: analysis channel extracted."""
|
||||
messages = [{"role": "user", "content": "Add 2 and 3."}]
|
||||
gen_req = await _e2e_render_chat(harmony_client, HARMONY_MODEL, messages)
|
||||
|
||||
reasoning_text = "The user wants 2 plus 3."
|
||||
answer_text = "The answer is 5."
|
||||
assistant_msg = {
|
||||
"role": "assistant",
|
||||
"thinking": reasoning_text,
|
||||
"content": answer_text,
|
||||
}
|
||||
output_ids = _harmony_extract_assistant_ids(harmony_tokenizer, assistant_msg)
|
||||
|
||||
decoded = harmony_tokenizer.decode(output_ids)
|
||||
if reasoning_text not in decoded:
|
||||
pytest.skip("Harmony template did not render thinking")
|
||||
|
||||
resp = await harmony_client.post(
|
||||
"/v1/chat/completions/derender",
|
||||
json={
|
||||
"model": HARMONY_MODEL,
|
||||
"generate_response": _e2e_generate_response(output_ids),
|
||||
"prompt_tokens": len(gen_req["token_ids"]),
|
||||
"chat_request": {
|
||||
"model": HARMONY_MODEL,
|
||||
"messages": messages,
|
||||
"include_reasoning": True,
|
||||
},
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200, resp.text
|
||||
msg = resp.json()["choices"][0]["message"]
|
||||
assert msg["reasoning"] is not None
|
||||
assert reasoning_text in msg["reasoning"]
|
||||
assert answer_text in (msg["content"] or "")
|
||||
|
||||
@@ -83,6 +83,20 @@ class MiniMaxM3Tokenizer:
|
||||
return "".join(tokens)
|
||||
|
||||
|
||||
class SplitMiniMaxM3Tokenizer(MiniMaxM3Tokenizer):
|
||||
"""Tokenizer that exposes marker vocab entries but encodes them as text."""
|
||||
|
||||
def tokenize(self, text: str) -> list[str]:
|
||||
return list(text)
|
||||
|
||||
|
||||
class RuntimeSplitMiniMaxM3Tokenizer(MiniMaxM3Tokenizer):
|
||||
"""Tokenizer whose runtime output splits markers despite atomic encodes."""
|
||||
|
||||
def encode_runtime(self, text: str) -> list[int]:
|
||||
return [self._add_token(token) for token in list(text)]
|
||||
|
||||
|
||||
def make_parser(
|
||||
chat_template_kwargs: dict[str, str] | None = None,
|
||||
) -> tuple[MiniMaxM3ReasoningParser, MiniMaxM3Tokenizer]:
|
||||
@@ -105,7 +119,8 @@ def run_streaming(
|
||||
reasoning_end_states: list[bool] = []
|
||||
|
||||
for chunk in chunks:
|
||||
delta_token_ids = tokenizer.encode(chunk, add_special_tokens=False)
|
||||
encode_runtime = getattr(tokenizer, "encode_runtime", tokenizer.encode)
|
||||
delta_token_ids = encode_runtime(chunk)
|
||||
current_text = previous_text + chunk
|
||||
current_token_ids = previous_token_ids + delta_token_ids
|
||||
delta = parser.extract_reasoning_streaming(
|
||||
@@ -174,14 +189,14 @@ def test_nonstreaming_drops_leading_end_tag():
|
||||
assert content == "answer"
|
||||
|
||||
|
||||
def test_nonstreaming_non_leading_end_tag_is_content():
|
||||
def test_nonstreaming_end_tag_in_content_state_is_dropped():
|
||||
parser, _ = make_parser()
|
||||
request = ChatCompletionRequest(messages=[], model="test-model")
|
||||
|
||||
reasoning, content = parser.extract_reasoning("XXX</mm:think>YYY", request)
|
||||
|
||||
assert reasoning is None
|
||||
assert content == "XXX</mm:think>YYY"
|
||||
assert content == "XXXYYY"
|
||||
|
||||
|
||||
def test_nonstreaming_enabled_mode_starts_in_reasoning():
|
||||
@@ -246,7 +261,7 @@ def test_streaming_drops_leading_end_tag():
|
||||
assert end_states == [True, True]
|
||||
|
||||
|
||||
def test_streaming_non_leading_end_tag_is_content():
|
||||
def test_streaming_end_tag_in_content_state_is_dropped():
|
||||
parser, tokenizer = make_parser()
|
||||
|
||||
reasoning, content, end_states = run_streaming(
|
||||
@@ -256,7 +271,7 @@ def test_streaming_non_leading_end_tag_is_content():
|
||||
)
|
||||
|
||||
assert reasoning is None
|
||||
assert content == "XXX</mm:think>YYY"
|
||||
assert content == "XXXYYY"
|
||||
assert end_states == [True]
|
||||
|
||||
|
||||
@@ -288,6 +303,110 @@ def test_streaming_plain_content_ends_reasoning_phase():
|
||||
assert end_states == [True, True]
|
||||
|
||||
|
||||
def test_streaming_split_marker_tokens_are_not_returned():
|
||||
tokenizer = RuntimeSplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(tokenizer)
|
||||
|
||||
reasoning, content, end_states = run_streaming(
|
||||
parser,
|
||||
tokenizer,
|
||||
["<mm:think>", "Reasoning", " content", "</mm:think>", "content"],
|
||||
)
|
||||
|
||||
assert reasoning == "Reasoning content"
|
||||
assert content == "content"
|
||||
assert end_states == [False, False, False, True, True]
|
||||
|
||||
|
||||
def test_streaming_split_marker_text_drives_end_state():
|
||||
tokenizer = RuntimeSplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(tokenizer)
|
||||
previous_text = ""
|
||||
previous_token_ids: list[int] = []
|
||||
|
||||
for chunk in ["<mm:think>", "Reasoning", " content", "</mm:think>"]:
|
||||
delta_token_ids = tokenizer.encode_runtime(chunk)
|
||||
current_text = previous_text + chunk
|
||||
current_token_ids = previous_token_ids + delta_token_ids
|
||||
parser.extract_reasoning_streaming(
|
||||
previous_text=previous_text,
|
||||
current_text=current_text,
|
||||
delta_text=chunk,
|
||||
previous_token_ids=previous_token_ids,
|
||||
current_token_ids=current_token_ids,
|
||||
delta_token_ids=delta_token_ids,
|
||||
)
|
||||
previous_text = current_text
|
||||
previous_token_ids = current_token_ids
|
||||
|
||||
assert parser.is_reasoning_end_streaming(previous_token_ids, []) is True
|
||||
|
||||
|
||||
def test_streaming_split_marker_tokens_enabled_mode():
|
||||
tokenizer = RuntimeSplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(
|
||||
tokenizer, chat_template_kwargs={"thinking_mode": "enabled"}
|
||||
)
|
||||
|
||||
reasoning, content, end_states = run_streaming(
|
||||
parser,
|
||||
tokenizer,
|
||||
["Reasoning", " content", "</mm:think>", "content"],
|
||||
)
|
||||
|
||||
assert reasoning == "Reasoning content"
|
||||
assert content == "content"
|
||||
assert end_states == [False, False, True, True]
|
||||
|
||||
|
||||
def test_streaming_split_marker_text_across_deltas():
|
||||
tokenizer = RuntimeSplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(tokenizer)
|
||||
|
||||
reasoning, content, end_states = run_streaming(
|
||||
parser,
|
||||
tokenizer,
|
||||
["<mm:", "think>", "Reasoning", " content", "</mm:", "think>", "content"],
|
||||
)
|
||||
|
||||
assert reasoning == "Reasoning content"
|
||||
assert content == "content"
|
||||
assert end_states == [False, False, False, False, False, True, True]
|
||||
|
||||
|
||||
def test_streaming_split_leading_end_marker_text_across_deltas():
|
||||
tokenizer = RuntimeSplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(tokenizer)
|
||||
|
||||
reasoning, content, end_states = run_streaming(
|
||||
parser,
|
||||
tokenizer,
|
||||
["</mm:", "think>", "content"],
|
||||
)
|
||||
|
||||
assert reasoning is None
|
||||
assert content == "content"
|
||||
assert end_states == [False, True, True]
|
||||
|
||||
|
||||
def test_token_id_helpers_with_split_marker_tokens():
|
||||
tokenizer = SplitMiniMaxM3Tokenizer()
|
||||
parser = MiniMaxM3ReasoningParser(tokenizer)
|
||||
output_ids = tokenizer.encode(
|
||||
"<mm:think>abc</mm:think>def", add_special_tokens=False
|
||||
)
|
||||
open_reasoning_ids = tokenizer.encode("<mm:think>abc", add_special_tokens=False)
|
||||
content_ids = tokenizer.encode("plain", add_special_tokens=False)
|
||||
|
||||
assert parser.is_reasoning_end(output_ids)
|
||||
assert not parser.is_reasoning_end(open_reasoning_ids)
|
||||
assert not parser.is_reasoning_end(content_ids)
|
||||
assert tokenizer.decode(parser.extract_content_ids(output_ids)) == "def"
|
||||
assert parser.extract_content_ids(open_reasoning_ids) == []
|
||||
assert parser.extract_content_ids(content_ids) == content_ids
|
||||
assert parser.count_reasoning_tokens(output_ids) == len(tokenizer.encode("abc"))
|
||||
|
||||
|
||||
def test_token_id_helpers():
|
||||
parser, tokenizer = make_parser()
|
||||
output_ids = tokenizer.encode(
|
||||
|
||||
@@ -1,16 +1,23 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
"""Tests for contiguous KV cache packing in _get_kv_cache_config_deepseek_v4."""
|
||||
"""Tests for contiguous KV cache packing."""
|
||||
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from vllm.v1.core.kv_cache_utils import _get_kv_cache_config_deepseek_v4
|
||||
from vllm import envs
|
||||
from vllm.v1.core.kv_cache_utils import (
|
||||
_get_kv_cache_config_deepseek_v4,
|
||||
get_kv_cache_config_from_groups,
|
||||
)
|
||||
from vllm.v1.kv_cache_interface import (
|
||||
FullAttentionSpec,
|
||||
KVCacheGroupSpec,
|
||||
KVCacheTensor,
|
||||
MLAAttentionSpec,
|
||||
SlidingWindowSpec,
|
||||
UniformTypeKVCacheSpecs,
|
||||
)
|
||||
|
||||
@@ -28,6 +35,25 @@ def _make_mla_spec(page_size: int, block_size: int = 256) -> MLAAttentionSpec:
|
||||
)
|
||||
|
||||
|
||||
def _make_full_spec() -> FullAttentionSpec:
|
||||
return FullAttentionSpec(
|
||||
block_size=16,
|
||||
num_kv_heads=2,
|
||||
head_size=64,
|
||||
dtype=torch.float16,
|
||||
)
|
||||
|
||||
|
||||
def _make_sw_spec() -> SlidingWindowSpec:
|
||||
return SlidingWindowSpec(
|
||||
block_size=16,
|
||||
num_kv_heads=2,
|
||||
head_size=64,
|
||||
dtype=torch.float16,
|
||||
sliding_window=128,
|
||||
)
|
||||
|
||||
|
||||
def _make_groups(n_c4, n_c128, n_swa):
|
||||
PS_C4_MLA = 37440
|
||||
PS_C4_IDX = 8640
|
||||
@@ -130,6 +156,73 @@ class TestInterleavedPacking:
|
||||
for i, v in enumerate(views):
|
||||
assert (v == i + 1).all(), f"View {i} was corrupted"
|
||||
|
||||
def test_hma_attention_groups_keep_default_backing(self, monkeypatch):
|
||||
monkeypatch.setattr(envs, "VLLM_USE_PACKED_HMA_KV_CACHE", False, raising=False)
|
||||
full = _make_full_spec()
|
||||
sw = _make_sw_spec()
|
||||
page_size = full.page_size_bytes
|
||||
groups = [
|
||||
KVCacheGroupSpec(["full.0", "full.1"], full),
|
||||
KVCacheGroupSpec(["sw.0", "sw.2"], sw),
|
||||
KVCacheGroupSpec(["sw.1", "sw.3"], sw),
|
||||
]
|
||||
|
||||
config = get_kv_cache_config_from_groups(
|
||||
_mock_vllm_config(), groups, available_memory=page_size * 2 * 32
|
||||
)
|
||||
|
||||
assert config.num_blocks == 32
|
||||
assert sum(t.size for t in config.kv_cache_tensors) == page_size * 2 * 32
|
||||
assert config.kv_cache_tensors == [
|
||||
KVCacheTensor(size=page_size * 32, shared_by=["full.0", "sw.0", "sw.1"]),
|
||||
KVCacheTensor(size=page_size * 32, shared_by=["full.1", "sw.2", "sw.3"]),
|
||||
]
|
||||
|
||||
def test_hma_attention_groups_use_packed_backing_with_flag(self, monkeypatch):
|
||||
monkeypatch.setattr(envs, "VLLM_USE_PACKED_HMA_KV_CACHE", True, raising=False)
|
||||
full = _make_full_spec()
|
||||
sw = _make_sw_spec()
|
||||
page_size = full.page_size_bytes
|
||||
groups = [
|
||||
KVCacheGroupSpec(["full.0", "full.1"], full),
|
||||
KVCacheGroupSpec(["sw.0", "sw.2"], sw),
|
||||
KVCacheGroupSpec(["sw.1", "sw.3"], sw),
|
||||
]
|
||||
|
||||
config = get_kv_cache_config_from_groups(
|
||||
_mock_vllm_config(), groups, available_memory=page_size * 2 * 32
|
||||
)
|
||||
|
||||
assert config.num_blocks == 32
|
||||
assert {t.size for t in config.kv_cache_tensors} == {page_size * 2 * 32}
|
||||
assert config.kv_cache_tensors == [
|
||||
KVCacheTensor(
|
||||
size=page_size * 2 * 32,
|
||||
shared_by=["full.0", "sw.0", "sw.1"],
|
||||
offset=0,
|
||||
block_stride=page_size * 2,
|
||||
),
|
||||
KVCacheTensor(
|
||||
size=page_size * 2 * 32,
|
||||
shared_by=["full.1", "sw.2", "sw.3"],
|
||||
offset=page_size,
|
||||
block_stride=page_size * 2,
|
||||
),
|
||||
]
|
||||
|
||||
def test_single_group_attention_keeps_unpacked_layout(self):
|
||||
spec = _make_full_spec()
|
||||
groups = [KVCacheGroupSpec(["full.0", "full.1"], spec)]
|
||||
|
||||
config = get_kv_cache_config_from_groups(
|
||||
_mock_vllm_config(), groups, available_memory=spec.page_size_bytes * 2 * 32
|
||||
)
|
||||
|
||||
assert sum(t.size for t in config.kv_cache_tensors) == (
|
||||
spec.page_size_bytes * 2 * 32
|
||||
)
|
||||
assert [t.block_stride for t in config.kv_cache_tensors] == [0, 0]
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
pytest.main([__file__, "-v"])
|
||||
|
||||
@@ -144,6 +144,43 @@ def test_async_scheduling_pp_allows_rescheduling_with_output_placeholders():
|
||||
assert req.request_id in output.num_scheduled_tokens
|
||||
|
||||
|
||||
def test_cached_request_data_resumed_all_token_ids_mrv1_only():
|
||||
"""all_token_ids carries a resumed request's token ids to the connector
|
||||
for the V1 model runner, but is skipped entirely for the V2 model runner.
|
||||
"""
|
||||
from vllm.v1.core.kv_cache_manager import KVCacheBlocks
|
||||
|
||||
scheduler = create_scheduler()
|
||||
(req,) = create_requests(num_requests=1, num_tokens=8)
|
||||
req.append_output_token_ids([101, 102, 103])
|
||||
|
||||
# A resumed request was not scheduled in the previous step.
|
||||
assert req.request_id not in scheduler.prev_step_scheduled_req_ids
|
||||
|
||||
empty_blocks = KVCacheBlocks(blocks=((),))
|
||||
|
||||
def make_cached():
|
||||
return scheduler._make_cached_request_data(
|
||||
running_reqs=[],
|
||||
resumed_reqs=[req],
|
||||
num_scheduled_tokens={req.request_id: 1},
|
||||
spec_decode_tokens={},
|
||||
req_to_new_blocks={req.request_id: empty_blocks},
|
||||
)
|
||||
|
||||
# V1 model runner: the full token id list is propagated.
|
||||
assert not scheduler.use_v2_model_runner
|
||||
cached = make_cached()
|
||||
assert req.request_id in cached.resumed_req_ids
|
||||
assert cached.all_token_ids[req.request_id] == list(req.all_token_ids)
|
||||
|
||||
# V2 model runner: all_token_ids is skipped entirely.
|
||||
scheduler.use_v2_model_runner = True
|
||||
cached = make_cached()
|
||||
assert req.request_id in cached.resumed_req_ids
|
||||
assert cached.all_token_ids == {}
|
||||
|
||||
|
||||
def test_schedule_partial_requests():
|
||||
"""Test scheduling behavior with partial requests.
|
||||
|
||||
|
||||
@@ -7,7 +7,10 @@ from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.coordinator imp
|
||||
ExternalCachedBlockPool,
|
||||
MooncakeStoreCoordinator,
|
||||
)
|
||||
from vllm.v1.core.kv_cache_utils import BlockHash, BlockHashListWithBlockSize
|
||||
from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.data import (
|
||||
chunk_hashes_for_block_size,
|
||||
)
|
||||
from vllm.v1.core.kv_cache_utils import BlockHash
|
||||
from vllm.v1.kv_cache_interface import (
|
||||
FullAttentionSpec,
|
||||
KVCacheGroupSpec,
|
||||
@@ -182,7 +185,7 @@ def test_coordinator_group_block_size_double_hash():
|
||||
]
|
||||
coord = _make_coord(groups, hash_block_size=16)
|
||||
hs = _hashes(4)
|
||||
big_hashes = list(BlockHashListWithBlockSize(hs, 16, 32))
|
||||
big_hashes = list(chunk_hashes_for_block_size(hs, 16, 32))
|
||||
exists = {(0, bytes(h)) for h in hs}
|
||||
exists |= {(1, bytes(bh)) for bh in big_hashes}
|
||||
cmap = ExternalCachedBlockPool(exists)
|
||||
|
||||
@@ -323,8 +323,8 @@ def test_recv_skips_swa_blocks_before_window():
|
||||
|
||||
def test_chunked_token_database_hash_block_size_smaller_than_block_size():
|
||||
"""DSv4-style: hash_block_size=4, group block_size=16 — process_tokens
|
||||
must merge every 4 fine hashes into one chunk hash via
|
||||
BlockHashListWithBlockSize."""
|
||||
keys each 16-token chunk by its last fine hash, keeping the Mooncake key
|
||||
at one digest instead of concatenating all 4 fine hashes."""
|
||||
md = KeyMetadata("m", 0, 0, 0, 0, group_id=3)
|
||||
db = ChunkedTokenDatabase(md, block_size=16, hash_block_size=4)
|
||||
db.set_kv_caches_base_addr([0])
|
||||
@@ -335,8 +335,7 @@ def test_chunked_token_database_hash_block_size_smaller_than_block_size():
|
||||
assert len(out) == 2
|
||||
assert out[0][0] == 0 and out[0][1] == 16
|
||||
assert out[1][0] == 16 and out[1][1] == 32
|
||||
# Each chunk's hash is the concatenation of 4 fine hashes.
|
||||
expected0 = b"".join(fine_hashes[0:4]).hex()
|
||||
expected1 = b"".join(fine_hashes[4:8]).hex()
|
||||
assert out[0][2].chunk_hash == expected0
|
||||
assert out[1][2].chunk_hash == expected1
|
||||
# Each chunk's hash is its last (4th) fine hash, which already chains the
|
||||
# prior three.
|
||||
assert out[0][2].chunk_hash == fine_hashes[3].hex()
|
||||
assert out[1][2].chunk_hash == fine_hashes[7].hex()
|
||||
|
||||
@@ -23,6 +23,7 @@ from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store import (
|
||||
worker as mooncake_store_worker,
|
||||
)
|
||||
from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.data import (
|
||||
BlobBlockHashes,
|
||||
ChunkedTokenDatabase,
|
||||
KeyMetadata,
|
||||
LoadSpec,
|
||||
@@ -32,6 +33,7 @@ from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.data import (
|
||||
from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.metrics import (
|
||||
MooncakeStoreConnectorStats,
|
||||
)
|
||||
from vllm.v1.core.kv_cache_utils import BlockHash
|
||||
|
||||
|
||||
def _default_send_coord() -> mooncake_store_worker.MooncakeStoreCoordinator:
|
||||
@@ -1179,9 +1181,9 @@ def test_store_sending_thread_kv_events_use_group_chunk_metadata():
|
||||
assert full_event.group_idx == 0
|
||||
assert full_event.block_size == 32
|
||||
assert full_event.token_ids == list(range(32))
|
||||
assert full_event.block_hashes == [
|
||||
maybe_convert_block_hash(BlockHash(b"".join(hs)))
|
||||
]
|
||||
# block_size=32 over hash_block_size=8 (scale 4): the chunk is keyed by its
|
||||
# last sub-hash, not the concatenation of all four.
|
||||
assert full_event.block_hashes == [maybe_convert_block_hash(BlockHash(hs[3]))]
|
||||
|
||||
assert swa_event.group_idx == 1
|
||||
assert swa_event.block_size == 8
|
||||
@@ -1749,3 +1751,33 @@ def test_store_worker_close_swallows_store_errors():
|
||||
worker.close()
|
||||
|
||||
assert worker.store is None
|
||||
|
||||
|
||||
def test_blob_block_hashes_wire_roundtrip():
|
||||
"""The lookup wire format sends a ``hash_len`` frame plus the raw hashes
|
||||
concatenated back-to-back; the server rebuilds them through a zero-copy
|
||||
``BlobBlockHashes`` view over the frame buffer."""
|
||||
hashes = [BlockHash(bytes([i]) * 16) for i in range(5)]
|
||||
hash_len = len(hashes[0])
|
||||
|
||||
# Client side (LookupKeyClient._lookup): flat payload frame.
|
||||
blob = b"".join(hashes)
|
||||
|
||||
# Server side (LookupKeyServer): view over the frame buffer (a memoryview),
|
||||
# never materializing the full hash list upfront.
|
||||
view = BlobBlockHashes(memoryview(blob), hash_len)
|
||||
|
||||
assert len(view) == 5
|
||||
assert list(view) == hashes # default Sequence iter terminates via IndexError
|
||||
assert [bytes(h) for h in view] == hashes
|
||||
assert bytes(view[-1]) == hashes[-1]
|
||||
assert [bytes(h) for h in view[1:3]] == hashes[1:3]
|
||||
with pytest.raises(IndexError):
|
||||
_ = view[5]
|
||||
|
||||
|
||||
def test_blob_block_hashes_empty():
|
||||
"""Empty lookups send hash_len=0 and an empty payload."""
|
||||
view = BlobBlockHashes(memoryview(b""), 0)
|
||||
assert len(view) == 0
|
||||
assert list(view) == []
|
||||
|
||||
@@ -14,12 +14,13 @@ from vllm.v1.kv_offload.base import (
|
||||
ReqContext,
|
||||
make_offload_key,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.common import CPULoadStoreSpec
|
||||
from vllm.v1.kv_offload.cpu.common import (
|
||||
CPULoadStoreSpec,
|
||||
CPUOffloadingMetrics,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.manager import CPUOffloadingManager
|
||||
from vllm.v1.kv_offload.cpu.policies.arc import ARCCachePolicy
|
||||
|
||||
STORES_SKIPPED = "vllm:kv_offload_stores_skipped"
|
||||
|
||||
|
||||
def make_req_context(
|
||||
req_id: str = "", kv_transfer_params: dict | None = None
|
||||
@@ -181,10 +182,45 @@ def test_filter_reused_manager_reports_stores_skipped_counter():
|
||||
)
|
||||
stats = manager.get_stats()
|
||||
assert stats is not None
|
||||
assert stats.reduce()[STORES_SKIPPED] == 3
|
||||
assert stats.reduce()[CPUOffloadingMetrics.STORES_SKIPPED] == 3
|
||||
stats = manager.get_stats()
|
||||
assert stats is not None
|
||||
assert stats.reduce()[STORES_SKIPPED] == 0
|
||||
assert stats.reduce()[CPUOffloadingMetrics.STORES_SKIPPED] == 0
|
||||
|
||||
|
||||
def test_cpu_manager_reports_cache_usage_gauge():
|
||||
def check_usage_stats(manager: CPUOffloadingManager, value: float):
|
||||
stats = manager.get_stats()
|
||||
assert stats is not None
|
||||
assert stats.reduce()[
|
||||
CPUOffloadingMetrics.CPU_CACHE_USAGE_PERC
|
||||
] == pytest.approx(value)
|
||||
|
||||
# Zero-capacity manager always reports 0.0
|
||||
manager = make_cpu_manager(num_blocks=0)
|
||||
check_usage_stats(manager, 0.0)
|
||||
|
||||
# Empty manager (4 blocks, none allocated): usage = 0.0
|
||||
manager = make_cpu_manager(num_blocks=4)
|
||||
check_usage_stats(manager, 0.0)
|
||||
|
||||
# After allocating 2 of 4 blocks: usage = 0.5
|
||||
manager.prepare_store(to_keys([1, 2]), _EMPTY_REQ_CTX)
|
||||
check_usage_stats(manager, 0.5)
|
||||
|
||||
# After filling all 4 blocks: usage = 1.0
|
||||
manager.prepare_store(to_keys([3, 4]), _EMPTY_REQ_CTX)
|
||||
check_usage_stats(manager, 1.0)
|
||||
|
||||
# After completing store, the blocks becomes evictable as it is not actively used
|
||||
# and usage drops.
|
||||
manager.complete_store(to_keys([1, 2]), _EMPTY_REQ_CTX)
|
||||
check_usage_stats(manager, 0.5)
|
||||
|
||||
# After completing store, the blocks becomes evictable as it is not actively used
|
||||
# and usage drops.
|
||||
manager.complete_store(to_keys([3, 4]), _EMPTY_REQ_CTX)
|
||||
check_usage_stats(manager, 0.0)
|
||||
|
||||
|
||||
def test_cpu_manager():
|
||||
|
||||
@@ -145,7 +145,6 @@ def _generate_fake_sampling_metadata(
|
||||
vllm_config.scheduler_config.max_num_seqs,
|
||||
num_spec,
|
||||
device,
|
||||
PIN_MEMORY_AVAILABLE,
|
||||
)
|
||||
fake_sampling_metadata = SamplingMetadata(
|
||||
temperature=torch.full((batch_size,), 0.0),
|
||||
@@ -880,7 +879,6 @@ def test_maybe_create_thinking_budget_holder_without_reasoning():
|
||||
cfg.scheduler_config.max_num_seqs,
|
||||
0,
|
||||
torch.device("cpu"),
|
||||
False,
|
||||
)
|
||||
is None
|
||||
)
|
||||
|
||||
@@ -6,6 +6,7 @@ from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from vllm import SamplingParams
|
||||
@@ -1528,3 +1529,232 @@ def test_reset_pending_loads() -> None:
|
||||
# All GPU blocks free
|
||||
num_used = gpu_pool.num_gpu_blocks - gpu_pool.get_num_free_blocks()
|
||||
assert num_used == 1, f"Expected only null block in use, got {num_used}"
|
||||
|
||||
|
||||
def _make_cp_vllm_config(
|
||||
dcp_world_size: int = 1,
|
||||
pcp_world_size: int = 1,
|
||||
) -> VllmConfig:
|
||||
"""VllmConfig with context-parallel sizes set for scheduler-only tests."""
|
||||
cfg = _make_vllm_config()
|
||||
|
||||
cfg.parallel_config.decode_context_parallel_size = dcp_world_size
|
||||
cfg.parallel_config.prefill_context_parallel_size = pcp_world_size
|
||||
return cfg
|
||||
|
||||
|
||||
def _make_cp_scheduler(
|
||||
*,
|
||||
dcp_world_size: int = 1,
|
||||
pcp_world_size: int = 1,
|
||||
num_cpu_blocks: int = 8,
|
||||
num_gpu_blocks: int = 16,
|
||||
lazy: bool = False,
|
||||
) -> SchedulerFixture:
|
||||
"""Build a SimpleCPUOffloadScheduler with CP-scaled virtual block size."""
|
||||
cp_world_size = dcp_world_size * pcp_world_size
|
||||
virtual_block_size = BLOCK_SIZE * cp_world_size
|
||||
|
||||
kv_cache_config = _make_kv_cache_config(num_gpu_blocks)
|
||||
vllm_config = _make_cp_vllm_config(dcp_world_size, pcp_world_size)
|
||||
cpu_capacity_bytes = _BYTES_PER_BLOCK * num_cpu_blocks
|
||||
|
||||
sched = SimpleCPUOffloadScheduler(
|
||||
vllm_config=vllm_config,
|
||||
kv_cache_config=kv_cache_config,
|
||||
cpu_capacity_bytes=cpu_capacity_bytes,
|
||||
scheduler_block_size=virtual_block_size,
|
||||
hash_block_size=virtual_block_size,
|
||||
lazy_offload=lazy,
|
||||
)
|
||||
|
||||
gpu_block_pool = BlockPool(
|
||||
num_gpu_blocks=num_gpu_blocks,
|
||||
enable_caching=True,
|
||||
hash_block_size=virtual_block_size,
|
||||
)
|
||||
sched.bind_gpu_block_pool(gpu_block_pool)
|
||||
|
||||
return SchedulerFixture(
|
||||
scheduler=sched,
|
||||
gpu_block_pool=gpu_block_pool,
|
||||
vllm_config=vllm_config,
|
||||
kv_cache_config=kv_cache_config,
|
||||
)
|
||||
|
||||
|
||||
def _make_cp_request(
|
||||
num_blocks: int,
|
||||
virtual_block_size: int,
|
||||
request_id: str | None = None,
|
||||
) -> Request:
|
||||
"""Create a request whose block hashes are computed at the virtual
|
||||
(CP-scaled) block size, matching what the real scheduler does.
|
||||
"""
|
||||
global _req_counter
|
||||
_req_counter += 1
|
||||
if request_id is None:
|
||||
request_id = f"req-cp-{_req_counter}"
|
||||
|
||||
num_tokens = num_blocks * virtual_block_size + 1
|
||||
start = _req_counter * 10000
|
||||
prompt_token_ids = list(range(start, start + num_tokens))
|
||||
sampling_params = SamplingParams(max_tokens=1)
|
||||
|
||||
return Request(
|
||||
request_id=request_id,
|
||||
prompt_token_ids=prompt_token_ids,
|
||||
sampling_params=sampling_params,
|
||||
pooling_params=None,
|
||||
mm_features=None,
|
||||
block_hasher=get_request_block_hasher(virtual_block_size, sha256),
|
||||
)
|
||||
|
||||
|
||||
def _allocate_cp_gpu_blocks(
|
||||
gpu_block_pool: BlockPool,
|
||||
request: Request,
|
||||
num_blocks: int,
|
||||
virtual_block_size: int,
|
||||
group_id: int = 0,
|
||||
) -> list:
|
||||
"""Allocate GPU blocks and cache them using the CP-scaled block size."""
|
||||
blocks = gpu_block_pool.get_new_blocks(num_blocks)
|
||||
num_full = min(num_blocks, len(request.block_hashes))
|
||||
if num_full > 0:
|
||||
gpu_block_pool.cache_full_blocks(
|
||||
request=request,
|
||||
blocks=blocks,
|
||||
num_cached_blocks=0,
|
||||
num_full_blocks=num_full,
|
||||
block_size=virtual_block_size,
|
||||
kv_cache_group_id=group_id,
|
||||
)
|
||||
return blocks
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Test 15: CP block size scaling is correct
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize(
|
||||
"dcp_world_size, pcp_world_size",
|
||||
[
|
||||
(2, 1), # DCP only
|
||||
(1, 2), # PCP only
|
||||
(2, 2), # DCP + PCP
|
||||
],
|
||||
)
|
||||
def test_cp_block_size_scaling(dcp_world_size: int, pcp_world_size: int) -> None:
|
||||
"""Verify that the scheduler's block_size and cp_world_size are correctly
|
||||
scaled when context parallelism is enabled."""
|
||||
fix = _make_cp_scheduler(
|
||||
dcp_world_size=dcp_world_size, pcp_world_size=pcp_world_size
|
||||
)
|
||||
sched = fix.scheduler
|
||||
|
||||
expected_cp = dcp_world_size * pcp_world_size
|
||||
assert sched.cp_world_size == expected_cp
|
||||
assert sched.block_size == BLOCK_SIZE * expected_cp
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Test 16: CP eager store-and-load roundtrip
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize(
|
||||
"dcp_world_size, pcp_world_size",
|
||||
[
|
||||
(2, 1),
|
||||
(1, 2),
|
||||
],
|
||||
)
|
||||
def test_cp_eager_store_and_load_roundtrip(
|
||||
dcp_world_size: int, pcp_world_size: int
|
||||
) -> None:
|
||||
"""With CP enabled, store blocks to CPU and reload them for a new request
|
||||
with matching tokens. Verifies that hash matching and transfer-pair
|
||||
construction work with the virtual block size."""
|
||||
fix = _make_cp_scheduler(
|
||||
dcp_world_size=dcp_world_size,
|
||||
pcp_world_size=pcp_world_size,
|
||||
num_cpu_blocks=8,
|
||||
num_gpu_blocks=16,
|
||||
lazy=False,
|
||||
)
|
||||
sched = fix.scheduler
|
||||
cp = dcp_world_size * pcp_world_size
|
||||
vbs = BLOCK_SIZE * cp
|
||||
|
||||
num_blocks = 2
|
||||
req = _make_cp_request(num_blocks, vbs)
|
||||
|
||||
# Allocate GPU blocks and register hashes
|
||||
gpu_blocks = _allocate_cp_gpu_blocks(fix.gpu_block_pool, req, num_blocks, vbs)
|
||||
kv_blocks = KVCacheBlocks(blocks=(gpu_blocks,))
|
||||
req.num_computed_tokens = num_blocks * vbs
|
||||
sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
|
||||
|
||||
block_ids = kv_blocks.get_block_ids()
|
||||
sched_out = make_scheduler_output(
|
||||
{req.request_id: num_blocks * vbs},
|
||||
new_reqs={req.request_id: block_ids},
|
||||
)
|
||||
|
||||
meta = sched.build_connector_meta(sched_out)
|
||||
assert meta.store_event >= 0, "Expected a store event"
|
||||
assert len(meta.store_gpu_blocks) == num_blocks
|
||||
assert len(meta.store_cpu_blocks) == num_blocks
|
||||
simulate_store_completion(sched, meta.store_event)
|
||||
|
||||
# New request with same tokens — should get a full CPU cache hit.
|
||||
req2 = Request(
|
||||
request_id="req-cp-load",
|
||||
prompt_token_ids=req.prompt_token_ids,
|
||||
sampling_params=req.sampling_params,
|
||||
pooling_params=None,
|
||||
mm_features=None,
|
||||
block_hasher=req._block_hasher,
|
||||
)
|
||||
|
||||
hit_tokens, is_async = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
|
||||
assert hit_tokens == num_blocks * vbs
|
||||
assert is_async is True
|
||||
|
||||
# Allocate fresh GPU blocks for the load.
|
||||
gpu_blocks2 = fix.gpu_block_pool.get_new_blocks(num_blocks)
|
||||
kv_blocks2 = KVCacheBlocks(blocks=(gpu_blocks2,))
|
||||
sched.update_state_after_alloc(req2, kv_blocks2, num_external_tokens=hit_tokens)
|
||||
|
||||
sched_out2 = make_scheduler_output(
|
||||
{req2.request_id: 1},
|
||||
new_reqs={req2.request_id: kv_blocks2.get_block_ids()},
|
||||
)
|
||||
meta2 = sched.build_connector_meta(sched_out2)
|
||||
assert meta2.load_event >= 0, "Expected a load event"
|
||||
assert len(meta2.load_gpu_blocks) == num_blocks
|
||||
assert len(meta2.load_cpu_blocks) == num_blocks
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Test 17: CP lazy target blocks are scaled correctly
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize("cp_world_size", [1, 2, 4])
|
||||
def test_cp_lazy_target_blocks_scaling(cp_world_size: int) -> None:
|
||||
"""_estimate_lazy_target_blocks returns fewer blocks when cp_world_size > 1
|
||||
because each virtual block covers more tokens."""
|
||||
kv_cache_config = _make_kv_cache_config(num_blocks=16)
|
||||
max_batched = 64
|
||||
|
||||
target_base = SimpleCPUOffloadScheduler._estimate_lazy_target_blocks(
|
||||
kv_cache_config, max_batched, cp_world_size=1
|
||||
)
|
||||
target_cp = SimpleCPUOffloadScheduler._estimate_lazy_target_blocks(
|
||||
kv_cache_config, max_batched, cp_world_size=cp_world_size
|
||||
)
|
||||
|
||||
if cp_world_size == 1:
|
||||
assert target_cp == target_base
|
||||
else:
|
||||
assert target_cp < target_base, (
|
||||
f"cp_world_size={cp_world_size}: target_cp={target_cp} should be "
|
||||
f"less than target_base={target_base}"
|
||||
)
|
||||
|
||||
@@ -35,7 +35,6 @@ def mock_model_runner_with_input_batch():
|
||||
max_model_len=1024,
|
||||
max_num_batched_tokens=1024,
|
||||
device="cpu",
|
||||
pin_memory=False,
|
||||
vocab_size=32000,
|
||||
block_sizes=[16],
|
||||
kernel_block_sizes=[16],
|
||||
|
||||
@@ -10,7 +10,6 @@ import torch
|
||||
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.sampling_params import SamplingParams
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import make_tensor_with_pad
|
||||
from vllm.v1.pool.metadata import PoolingMetadata
|
||||
from vllm.v1.sample.logits_processor import LogitsProcessors
|
||||
@@ -236,7 +235,6 @@ def test_sampling_metadata_in_input_batch(device: str, batch_size: int):
|
||||
max_model_len=1024,
|
||||
max_num_batched_tokens=1024,
|
||||
device=torch.device(device),
|
||||
pin_memory=is_pin_memory_available(),
|
||||
vocab_size=1024,
|
||||
block_sizes=[1],
|
||||
kernel_block_sizes=[1],
|
||||
@@ -331,7 +329,6 @@ def test_swap_states_in_input_batch(device: str, batch_size: int, swap_list: lis
|
||||
max_model_len=1024,
|
||||
max_num_batched_tokens=1024,
|
||||
device=torch.device(device),
|
||||
pin_memory=is_pin_memory_available(),
|
||||
vocab_size=1024,
|
||||
block_sizes=[1],
|
||||
kernel_block_sizes=[1],
|
||||
@@ -341,7 +338,6 @@ def test_swap_states_in_input_batch(device: str, batch_size: int, swap_list: lis
|
||||
max_model_len=1024,
|
||||
max_num_batched_tokens=1024,
|
||||
device=torch.device(device),
|
||||
pin_memory=is_pin_memory_available(),
|
||||
vocab_size=1024,
|
||||
block_sizes=[1],
|
||||
kernel_block_sizes=[1],
|
||||
@@ -410,7 +406,6 @@ def test_pooling_prompt_lens_not_aliased(device: str):
|
||||
max_model_len=MAX_PROMPT_SIZE + NUM_OUTPUT_TOKENS,
|
||||
max_num_batched_tokens=batch_size * (MAX_PROMPT_SIZE + NUM_OUTPUT_TOKENS),
|
||||
device=torch.device(device),
|
||||
pin_memory=is_pin_memory_available(),
|
||||
vocab_size=VOCAB_SIZE,
|
||||
block_sizes=[16],
|
||||
kernel_block_sizes=[16],
|
||||
@@ -459,7 +454,6 @@ def test_pooling_metadata_token_id_buffers(
|
||||
max_model_len=MAX_PROMPT_SIZE + NUM_OUTPUT_TOKENS,
|
||||
max_num_batched_tokens=MAX_PROMPT_SIZE + NUM_OUTPUT_TOKENS,
|
||||
device=torch.device("cpu"),
|
||||
pin_memory=False,
|
||||
vocab_size=VOCAB_SIZE,
|
||||
block_sizes=[16],
|
||||
kernel_block_sizes=[16],
|
||||
|
||||
@@ -85,7 +85,6 @@ def initialize_kv_cache(runner: GPUModelRunner):
|
||||
max_model_len=runner.max_model_len,
|
||||
max_num_batched_tokens=runner.max_num_tokens,
|
||||
device=runner.device,
|
||||
pin_memory=runner.pin_memory,
|
||||
vocab_size=runner.model_config.get_vocab_size(),
|
||||
block_sizes=[kv_cache_config.kv_cache_groups[0].kv_cache_spec.block_size],
|
||||
kernel_block_sizes=[
|
||||
@@ -1405,7 +1404,6 @@ def test_input_batch_with_kernel_block_sizes():
|
||||
max_model_len = 512
|
||||
max_num_batched_tokens = 512
|
||||
device = torch.device(DEVICE_TYPE)
|
||||
pin_memory = False
|
||||
vocab_size = 50272
|
||||
|
||||
# Test with different kernel block sizes
|
||||
@@ -1417,7 +1415,6 @@ def test_input_batch_with_kernel_block_sizes():
|
||||
max_model_len=max_model_len,
|
||||
max_num_batched_tokens=max_num_batched_tokens,
|
||||
device=device,
|
||||
pin_memory=pin_memory,
|
||||
vocab_size=vocab_size,
|
||||
block_sizes=block_sizes,
|
||||
kernel_block_sizes=kernel_block_sizes,
|
||||
@@ -1478,7 +1475,6 @@ def test_hybrid_cache_integration(default_vllm_config, dist_init):
|
||||
max_model_len=runner.max_model_len,
|
||||
max_num_batched_tokens=runner.max_num_tokens,
|
||||
device=runner.device,
|
||||
pin_memory=runner.pin_memory,
|
||||
vocab_size=runner.model_config.get_vocab_size(),
|
||||
block_sizes=[kv_cache_config.kv_cache_groups[0].kv_cache_spec.block_size],
|
||||
kernel_block_sizes=[16],
|
||||
|
||||
@@ -18,8 +18,8 @@ import torch
|
||||
|
||||
from vllm.device_allocator import AllocationData, HandleType
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.system_utils import find_loaded_library
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
@@ -196,7 +196,7 @@ class CuMemAllocator:
|
||||
size_in_bytes,
|
||||
dtype=torch.uint8,
|
||||
device="cpu",
|
||||
pin_memory=is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
cpu_ptr = cpu_backup_tensor.data_ptr()
|
||||
libcudart.cudaMemcpy(cpu_ptr, ptr, size_in_bytes)
|
||||
|
||||
@@ -11,7 +11,7 @@ import torch
|
||||
|
||||
from vllm.device_allocator import AllocationData, HandleType
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
@@ -188,7 +188,7 @@ class XpuMemAllocator:
|
||||
size_in_bytes,
|
||||
dtype=torch.uint8,
|
||||
device="cpu",
|
||||
pin_memory=is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
cpu_ptr = cpu_backup_tensor.data_ptr()
|
||||
_xpu_memcpy_sync(
|
||||
|
||||
@@ -2,13 +2,15 @@
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
"""External-store cache-hit coordinator for MooncakeStoreConnector."""
|
||||
|
||||
from collections.abc import Sequence
|
||||
from typing import cast
|
||||
|
||||
from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.data import (
|
||||
chunk_hashes_for_block_size,
|
||||
)
|
||||
from vllm.v1.core.block_pool import BlockPool
|
||||
from vllm.v1.core.kv_cache_utils import (
|
||||
BlockHash,
|
||||
BlockHashList,
|
||||
BlockHashListWithBlockSize,
|
||||
KVCacheBlock,
|
||||
)
|
||||
from vllm.v1.core.single_type_kv_cache_manager import (
|
||||
@@ -120,7 +122,7 @@ class MooncakeStoreCoordinator:
|
||||
|
||||
def find_longest_cache_hit(
|
||||
self,
|
||||
block_hashes: list[BlockHash],
|
||||
block_hashes: Sequence[BlockHash],
|
||||
max_length: int,
|
||||
cached_block_pool: ExternalCachedBlockPool,
|
||||
*,
|
||||
@@ -147,7 +149,7 @@ class MooncakeStoreCoordinator:
|
||||
|
||||
def load_mask(
|
||||
self,
|
||||
block_hashes: list[BlockHash],
|
||||
block_hashes: Sequence[BlockHash],
|
||||
token_len: int,
|
||||
) -> tuple[list[bool], ...]:
|
||||
"""Per-group load masks: ``mask[g][i]`` is True iff group ``g``'s
|
||||
@@ -236,17 +238,15 @@ class MooncakeStoreCoordinator:
|
||||
return tuple(masks)
|
||||
|
||||
def block_hashes_for_spec(
|
||||
self, block_hashes: list[BlockHash], spec: KVCacheSpec
|
||||
) -> BlockHashList:
|
||||
if spec.block_size == self.hash_block_size:
|
||||
return block_hashes
|
||||
return BlockHashListWithBlockSize(
|
||||
self, block_hashes: Sequence[BlockHash], spec: KVCacheSpec
|
||||
) -> Sequence[BlockHash]:
|
||||
return chunk_hashes_for_block_size(
|
||||
block_hashes, self.hash_block_size, spec.block_size
|
||||
)
|
||||
|
||||
def _find_hit_blocks(
|
||||
self,
|
||||
block_hashes: list[BlockHash],
|
||||
block_hashes: Sequence[BlockHash],
|
||||
max_length: int,
|
||||
cached_block_pool: ExternalCachedBlockPool,
|
||||
*,
|
||||
@@ -264,7 +264,7 @@ class MooncakeStoreCoordinator:
|
||||
spec, group_ids, manager_cls = self.attention_groups[0]
|
||||
hashes = self.block_hashes_for_spec(block_hashes, spec)
|
||||
hit_blocks = manager_cls.find_longest_cache_hit(
|
||||
block_hashes=hashes,
|
||||
block_hashes=hashes, # type: ignore[arg-type]
|
||||
max_length=max_length,
|
||||
kv_cache_group_ids=group_ids,
|
||||
block_pool=cast(BlockPool, cached_block_pool),
|
||||
@@ -304,7 +304,7 @@ class MooncakeStoreCoordinator:
|
||||
_max_length = min(curr_hit_length + spec.block_size, max_length)
|
||||
hashes = self.block_hashes_for_spec(block_hashes, spec)
|
||||
hit_blocks = manager_cls.find_longest_cache_hit(
|
||||
block_hashes=hashes,
|
||||
block_hashes=hashes, # type: ignore[arg-type]
|
||||
max_length=_max_length,
|
||||
kv_cache_group_ids=group_ids,
|
||||
block_pool=cast(BlockPool, cached_block_pool),
|
||||
|
||||
@@ -5,8 +5,9 @@
|
||||
# (vllm_ascend/distributed/kv_transfer/kv_pool/ascend_store/).
|
||||
"""Data classes for MooncakeStoreConnector."""
|
||||
|
||||
from collections.abc import Iterable
|
||||
from collections.abc import Iterable, Sequence
|
||||
from dataclasses import dataclass
|
||||
from typing import cast
|
||||
|
||||
import torch
|
||||
|
||||
@@ -23,6 +24,77 @@ from vllm.v1.core.kv_cache_utils import (
|
||||
logger = init_logger(__name__)
|
||||
|
||||
|
||||
class BlobBlockHashes(Sequence[BlockHash]):
|
||||
"""Lazy view over a flat buffer of fixed-size block hashes to avoid the overhead
|
||||
of materializing all hashes upfront.
|
||||
"""
|
||||
|
||||
def __init__(self, blob: memoryview, hash_len: int):
|
||||
self._blob = blob
|
||||
self._hash_len = hash_len
|
||||
self._n = len(blob) // hash_len if hash_len else 0
|
||||
|
||||
def __len__(self) -> int:
|
||||
return self._n
|
||||
|
||||
def __getitem__(self, idx):
|
||||
if isinstance(idx, slice):
|
||||
return [self[i] for i in range(*idx.indices(self._n))]
|
||||
if idx < 0:
|
||||
idx += self._n
|
||||
if not 0 <= idx < self._n:
|
||||
raise IndexError(idx)
|
||||
off = idx * self._hash_len
|
||||
return BlockHash(self._blob[off : off + self._hash_len])
|
||||
|
||||
|
||||
class _CompactChunkHashList(BlockHashListWithBlockSize):
|
||||
"""View that keys each ``block_size`` chunk by the last constituent
|
||||
``hash_block_size`` hash instead of concatenating all of them.
|
||||
|
||||
The engine chains block hashes (each hash folds in the previous one), so the
|
||||
final sub-block hash of a chunk already uniquely identifies the whole chunk
|
||||
and its prefix. Using it keeps a Mooncake key at a single hash digest
|
||||
regardless of the ``block_size`` / ``hash_block_size`` ratio, instead of
|
||||
growing the key linearly with it (e.g. 64x for ``block_size=256``,
|
||||
``hash_block_size=4``).
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
block_hashes: Sequence[BlockHash],
|
||||
hash_block_size: int,
|
||||
target_block_size: int,
|
||||
):
|
||||
# Accept any indexable sequence (e.g. the lazy ``BlobBlockHashes``), not
|
||||
# just ``list``; the base only indexes/sizes it.
|
||||
assert target_block_size % hash_block_size == 0
|
||||
self.block_hashes = block_hashes # type: ignore[assignment]
|
||||
self.scale_factor = target_block_size // hash_block_size
|
||||
|
||||
def _get_value_at(self, idx: int) -> BlockHash:
|
||||
return self.block_hashes[idx * self.scale_factor + self.scale_factor - 1]
|
||||
|
||||
|
||||
def chunk_hashes_for_block_size(
|
||||
block_hashes: Sequence[BlockHash],
|
||||
hash_block_size: int,
|
||||
block_size: int,
|
||||
) -> Sequence[BlockHash]:
|
||||
"""Map ``hash_block_size``-granular block hashes to one compact hash per
|
||||
``block_size`` chunk (the chunk's last sub-hash). Returns ``block_hashes``
|
||||
unchanged when the two sizes are equal.
|
||||
"""
|
||||
if block_size == hash_block_size:
|
||||
return block_hashes
|
||||
# Structurally a Sequence[BlockHash] (indexable + sized); the base class
|
||||
# just isn't declared as one.
|
||||
return cast(
|
||||
"Sequence[BlockHash]",
|
||||
_CompactChunkHashList(block_hashes, hash_block_size, block_size),
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class KeyMetadata:
|
||||
"""Metadata for constructing pool keys."""
|
||||
@@ -138,18 +210,15 @@ class ChunkedTokenDatabase:
|
||||
Args:
|
||||
token_len: Total number of tokens.
|
||||
block_hashes: Block hashes computed at ``hash_block_size`` granularity.
|
||||
When ``block_size > hash_block_size`` consecutive hashes are merged
|
||||
up to the group's ``block_size`` via ``BlockHashListWithBlockSize``.
|
||||
When ``block_size > hash_block_size`` each group's ``block_size`` chunk
|
||||
is keyed by its last sub-hash via ``chunk_hashes_for_block_size``.
|
||||
mask_num: Number of tokens to skip from the beginning.
|
||||
"""
|
||||
if not block_hashes:
|
||||
return
|
||||
if self.block_size == self.hash_block_size:
|
||||
chunk_hashes: Iterable[BlockHash] = block_hashes
|
||||
else:
|
||||
chunk_hashes = BlockHashListWithBlockSize(
|
||||
block_hashes, self.hash_block_size, self.block_size
|
||||
)
|
||||
chunk_hashes: Iterable[BlockHash] = chunk_hashes_for_block_size(
|
||||
block_hashes, self.hash_block_size, self.block_size
|
||||
)
|
||||
for chunk_id, h in enumerate(chunk_hashes):
|
||||
start_idx = chunk_id * self.block_size
|
||||
if start_idx >= token_len:
|
||||
|
||||
@@ -11,7 +11,10 @@ Wire format (REQ/REP over IPC):
|
||||
|
||||
msg_type == LOOKUP_MSG:
|
||||
frame 1: token_len (u32 big-endian, 4 bytes)
|
||||
frame 2..n: msgpack-encoded list[str] of block-hash hex digests
|
||||
frame 2: hash_len (u16 big-endian, 2 bytes) — byte length of each
|
||||
fixed-size block hash (0 when there are no hashes)
|
||||
frame 3: raw block hashes concatenated back-to-back (each hash_len
|
||||
bytes); the server splits on hash_len
|
||||
Response: [hit_count: u32 big-endian, 4 bytes]
|
||||
|
||||
msg_type == RESET_MSG:
|
||||
|
||||
@@ -18,7 +18,7 @@ import socket
|
||||
import threading
|
||||
import time
|
||||
from collections import defaultdict
|
||||
from collections.abc import Callable
|
||||
from collections.abc import Callable, Sequence
|
||||
from concurrent.futures import Future, ThreadPoolExecutor
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Literal, TypeVar
|
||||
@@ -45,6 +45,7 @@ from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.coordinator imp
|
||||
MooncakeStoreCoordinator,
|
||||
)
|
||||
from vllm.distributed.kv_transfer.kv_connector.v1.mooncake.store.data import ( # noqa: E501
|
||||
BlobBlockHashes,
|
||||
ChunkedTokenDatabase,
|
||||
KeyMetadata,
|
||||
MooncakeStoreConnectorMetadata,
|
||||
@@ -65,7 +66,6 @@ from vllm.v1.core.kv_cache_utils import (
|
||||
resolve_kv_cache_block_sizes,
|
||||
)
|
||||
from vllm.v1.kv_cache_interface import KVCacheConfig, KVCacheGroupSpec
|
||||
from vllm.v1.serial_utils import MsgpackDecoder, MsgpackEncoder
|
||||
|
||||
from .metrics import MooncakeStoreConnectorStats
|
||||
|
||||
@@ -1372,7 +1372,7 @@ class MooncakeStoreWorker:
|
||||
|
||||
return finished_sending
|
||||
|
||||
def lookup(self, token_len: int, block_hashes: list[BlockHash]) -> int:
|
||||
def lookup(self, token_len: int, block_hashes: Sequence[BlockHash]) -> int:
|
||||
"""Check how many prefix tokens exist in the store.
|
||||
|
||||
Checks across all TP ranks and PP ranks.
|
||||
@@ -1392,6 +1392,11 @@ class MooncakeStoreWorker:
|
||||
group_hashes = self.coord.block_hashes_for_spec(
|
||||
block_hashes, self._kv_cache_groups[g_idx].kv_cache_spec
|
||||
)
|
||||
metadata_templates = [
|
||||
dataclasses.replace(db.metadata, tp_rank=tp, pp_rank=pp)
|
||||
for tp in range(tp_count)
|
||||
for pp in range(self.pp_size)
|
||||
]
|
||||
for chunk_id, h in enumerate(group_hashes):
|
||||
start_idx = chunk_id * spec_block_size
|
||||
if start_idx >= token_len:
|
||||
@@ -1400,11 +1405,11 @@ class MooncakeStoreWorker:
|
||||
chunk_id >= len(lookup_mask) or not lookup_mask[chunk_id]
|
||||
):
|
||||
continue
|
||||
for tp in range(tp_count):
|
||||
for pp in range(self.pp_size):
|
||||
md = dataclasses.replace(db.metadata, tp_rank=tp, pp_rank=pp)
|
||||
candidate_keys.append(PoolKey(md, h.hex()).to_string())
|
||||
candidate_meta.append((g_idx, bytes(h)))
|
||||
h_hex = h.hex()
|
||||
h_bytes = bytes(h)
|
||||
for md in metadata_templates:
|
||||
candidate_keys.append(PoolKey(md, h_hex).to_string())
|
||||
candidate_meta.append((g_idx, h_bytes))
|
||||
|
||||
if not candidate_keys:
|
||||
return 0
|
||||
@@ -1483,7 +1488,6 @@ class LookupKeyServer:
|
||||
store_worker: MooncakeStoreWorker,
|
||||
vllm_config: VllmConfig,
|
||||
):
|
||||
self.decoder = MsgpackDecoder()
|
||||
self.ctx = zmq.Context() # type: ignore[attr-defined]
|
||||
socket_path = get_zmq_rpc_path_lookup(vllm_config)
|
||||
self._ipc_path = socket_path.removeprefix("ipc://")
|
||||
@@ -1506,9 +1510,9 @@ class LookupKeyServer:
|
||||
|
||||
if msg_type == LOOKUP_MSG:
|
||||
token_len = int.from_bytes(all_frames[1], byteorder="big")
|
||||
hash_frames = all_frames[2:]
|
||||
hashes_str = self.decoder.decode(hash_frames)
|
||||
block_hashes = [BlockHash(bytes.fromhex(s)) for s in hashes_str]
|
||||
hash_len = int.from_bytes(all_frames[2], byteorder="big")
|
||||
blob = all_frames[3].buffer
|
||||
block_hashes = BlobBlockHashes(blob, hash_len)
|
||||
result = self.store_worker.lookup(token_len, block_hashes)
|
||||
self.socket.send(result.to_bytes(4, "big"))
|
||||
|
||||
@@ -1557,7 +1561,6 @@ class LookupKeyClient:
|
||||
"""
|
||||
|
||||
def __init__(self, vllm_config: VllmConfig):
|
||||
self.encoder = MsgpackEncoder()
|
||||
self.ctx = zmq.Context() # type: ignore[attr-defined]
|
||||
socket_path = get_zmq_rpc_path_lookup(vllm_config)
|
||||
self.socket = make_zmq_socket(
|
||||
@@ -1574,14 +1577,16 @@ class LookupKeyClient:
|
||||
self.futures: dict[str, Future[int]] = {}
|
||||
|
||||
def _lookup(self, token_len: int, block_hashes: list[BlockHash]) -> int:
|
||||
hash_strs = [h.hex() for h in block_hashes]
|
||||
hash_frames = self.encoder.encode(hash_strs)
|
||||
token_len_bytes = token_len.to_bytes(4, byteorder="big")
|
||||
all_frames = [LOOKUP_MSG, token_len_bytes] + list(hash_frames)
|
||||
hash_len = len(block_hashes[0]) if block_hashes else 0
|
||||
all_frames = (
|
||||
LOOKUP_MSG,
|
||||
token_len.to_bytes(4, byteorder="big"),
|
||||
hash_len.to_bytes(2, byteorder="big"),
|
||||
b"".join(block_hashes),
|
||||
)
|
||||
self.socket.send_multipart(all_frames, copy=False)
|
||||
resp = self.socket.recv()
|
||||
result = int.from_bytes(resp, "big")
|
||||
return result
|
||||
return int.from_bytes(resp, "big")
|
||||
|
||||
def lookup(
|
||||
self,
|
||||
|
||||
@@ -50,7 +50,8 @@ class OffloadingConnectorWorker:
|
||||
def register_kv_caches(
|
||||
self, kv_caches: dict[str, torch.Tensor | list[torch.Tensor]]
|
||||
):
|
||||
num_blocks = self.spec.kv_cache_config.num_blocks
|
||||
kv_cache_config = self.spec.kv_cache_config
|
||||
num_blocks = kv_cache_config.num_blocks
|
||||
|
||||
# layer_name -> (num_blocks, page_size_bytes) tensor
|
||||
tensors_per_block: dict[str, tuple[torch.Tensor, ...]] = {}
|
||||
@@ -58,7 +59,7 @@ class OffloadingConnectorWorker:
|
||||
unpadded_page_size_bytes: dict[str, int] = {}
|
||||
# layer_name -> size of page in bytes
|
||||
page_size_bytes: dict[str, int] = {}
|
||||
for kv_cache_group in self.spec.kv_cache_config.kv_cache_groups:
|
||||
for kv_cache_group in kv_cache_config.kv_cache_groups:
|
||||
group_layer_names = kv_cache_group.layer_names
|
||||
group_kv_cache_spec = kv_cache_group.kv_cache_spec
|
||||
if isinstance(group_kv_cache_spec, UniformTypeKVCacheSpecs):
|
||||
@@ -122,9 +123,35 @@ class OffloadingConnectorWorker:
|
||||
else:
|
||||
raise NotImplementedError
|
||||
|
||||
packed_kv_cache_tensor = next(
|
||||
(t for t in kv_cache_config.kv_cache_tensors if t.block_stride), None
|
||||
)
|
||||
is_dsv4 = all(
|
||||
isinstance(group.kv_cache_spec, UniformTypeKVCacheSpecs)
|
||||
for group in kv_cache_config.kv_cache_groups
|
||||
)
|
||||
if packed_kv_cache_tensor is not None and not is_dsv4:
|
||||
(tensor,) = tensors_per_block[packed_kv_cache_tensor.shared_by[0]]
|
||||
block_stride = tensor.stride(0)
|
||||
packed_tensor = tensor.as_strided(
|
||||
(num_blocks, block_stride),
|
||||
(block_stride, 1),
|
||||
storage_offset=0,
|
||||
)
|
||||
self._register_handlers(
|
||||
CanonicalKVCaches(
|
||||
[CanonicalKVCacheTensor(packed_tensor, block_stride)],
|
||||
[
|
||||
[CanonicalKVCacheRef(0, block_stride)]
|
||||
for _ in kv_cache_config.kv_cache_groups
|
||||
],
|
||||
)
|
||||
)
|
||||
return
|
||||
|
||||
block_tensors: list[CanonicalKVCacheTensor] = []
|
||||
block_data_refs: dict[str, list[CanonicalKVCacheRef]] = defaultdict(list)
|
||||
for kv_cache_tensor in self.spec.kv_cache_config.kv_cache_tensors:
|
||||
for kv_cache_tensor in kv_cache_config.kv_cache_tensors:
|
||||
# Filter to layers that were actually processed above.
|
||||
# _get_kv_cache_config_deepseek_v4 emits KVCacheTensor entries for
|
||||
# every (tuple_idx, page_size) slot; slots where no group has a
|
||||
@@ -166,7 +193,7 @@ class OffloadingConnectorWorker:
|
||||
)
|
||||
|
||||
group_data_refs: list[list[CanonicalKVCacheRef]] = []
|
||||
for kv_cache_group in self.spec.kv_cache_config.kv_cache_groups:
|
||||
for kv_cache_group in kv_cache_config.kv_cache_groups:
|
||||
group_refs: list[CanonicalKVCacheRef] = []
|
||||
for layer_name in kv_cache_group.layer_names:
|
||||
group_refs += block_data_refs[layer_name]
|
||||
|
||||
@@ -43,6 +43,7 @@ from vllm.entrypoints.openai.engine.protocol import (
|
||||
JsonSchemaResponseFormat,
|
||||
ResponseFormat,
|
||||
StreamOptions,
|
||||
UsageInfo,
|
||||
)
|
||||
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
|
||||
from vllm.entrypoints.serve.utils.api_utils import sanitize_message
|
||||
@@ -54,6 +55,49 @@ if TYPE_CHECKING:
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _get_cached_tokens(usage: UsageInfo | None) -> int | None:
|
||||
"""Extract cached token count from OpenAI UsageInfo."""
|
||||
if usage is None or usage.prompt_tokens_details is None:
|
||||
return None
|
||||
return usage.prompt_tokens_details.cached_tokens
|
||||
|
||||
|
||||
def _build_anthropic_usage(
|
||||
prompt_tokens: int,
|
||||
completion_tokens: int | None,
|
||||
usage: UsageInfo | None,
|
||||
) -> AnthropicUsage:
|
||||
"""Build an AnthropicUsage from OpenAI-style token counts.
|
||||
|
||||
Anthropic defines ``total_input == input_tokens + cache_read +
|
||||
cache_creation``. vLLM's ``prompt_tokens`` is the total, so
|
||||
``input_tokens = prompt_tokens - cached_tokens``.
|
||||
|
||||
OpenAI usage only exposes ``cached_tokens`` (hits); there is no
|
||||
cache-creation analog, so ``cache_creation_input_tokens`` is ``0``
|
||||
when cache info is present. When cache info is absent (e.g.
|
||||
``--enable-prompt-tokens-details`` off, or a streaming chunk that
|
||||
hasn't carried it yet), cache fields are left **unset** so
|
||||
``exclude_unset=True`` serialization omits them entirely.
|
||||
|
||||
``completion_tokens`` follows ``UsageInfo`` and may be ``None`` on
|
||||
intermediate stream chunks; we coerce to ``0`` for the wire format.
|
||||
"""
|
||||
output_tokens = completion_tokens or 0
|
||||
cached = _get_cached_tokens(usage)
|
||||
if cached is not None:
|
||||
return AnthropicUsage(
|
||||
input_tokens=prompt_tokens - cached,
|
||||
output_tokens=output_tokens,
|
||||
cache_read_input_tokens=cached,
|
||||
cache_creation_input_tokens=0,
|
||||
)
|
||||
return AnthropicUsage(
|
||||
input_tokens=prompt_tokens,
|
||||
output_tokens=output_tokens,
|
||||
)
|
||||
|
||||
|
||||
def wrap_data_with_event(data: str, event: str):
|
||||
return f"event: {event}\ndata: {data}\n\n"
|
||||
|
||||
@@ -582,9 +626,10 @@ class AnthropicServingMessages(OpenAIServingChat):
|
||||
id=generator.id,
|
||||
content=[],
|
||||
model=generator.model,
|
||||
usage=AnthropicUsage(
|
||||
input_tokens=generator.usage.prompt_tokens,
|
||||
output_tokens=generator.usage.completion_tokens,
|
||||
usage=_build_anthropic_usage(
|
||||
generator.usage.prompt_tokens,
|
||||
generator.usage.completion_tokens,
|
||||
generator.usage,
|
||||
),
|
||||
kv_transfer_params=generator.kv_transfer_params,
|
||||
)
|
||||
@@ -765,11 +810,12 @@ class AnthropicServingMessages(OpenAIServingChat):
|
||||
model=origin_chunk.model,
|
||||
stop_reason=None,
|
||||
stop_sequence=None,
|
||||
usage=AnthropicUsage(
|
||||
input_tokens=origin_chunk.usage.prompt_tokens
|
||||
usage=_build_anthropic_usage(
|
||||
origin_chunk.usage.prompt_tokens
|
||||
if origin_chunk.usage
|
||||
else 0,
|
||||
output_tokens=0,
|
||||
0,
|
||||
origin_chunk.usage,
|
||||
),
|
||||
),
|
||||
)
|
||||
@@ -788,13 +834,14 @@ class AnthropicServingMessages(OpenAIServingChat):
|
||||
chunk = AnthropicStreamEvent(
|
||||
type="message_delta",
|
||||
delta=AnthropicDelta(stop_reason=stop_reason),
|
||||
usage=AnthropicUsage(
|
||||
input_tokens=origin_chunk.usage.prompt_tokens
|
||||
usage=_build_anthropic_usage(
|
||||
origin_chunk.usage.prompt_tokens
|
||||
if origin_chunk.usage
|
||||
else 0,
|
||||
output_tokens=origin_chunk.usage.completion_tokens
|
||||
origin_chunk.usage.completion_tokens
|
||||
if origin_chunk.usage
|
||||
else 0,
|
||||
origin_chunk.usage,
|
||||
),
|
||||
)
|
||||
data = chunk.model_dump_json(exclude_unset=True)
|
||||
|
||||
@@ -455,7 +455,7 @@ async def init_render_app_state(
|
||||
enable_auto_tools=args.enable_auto_tool_choice,
|
||||
exclude_tools_when_tool_choice_none=args.exclude_tools_when_tool_choice_none,
|
||||
tool_parser=args.tool_call_parser,
|
||||
reasoning_parser=args.structured_outputs_config.reasoning_parser,
|
||||
reasoning_parser=args.reasoning_parser,
|
||||
default_chat_template_kwargs=args.default_chat_template_kwargs,
|
||||
log_error_stack=args.log_error_stack,
|
||||
)
|
||||
|
||||
@@ -219,10 +219,14 @@ class GenerateResponse(BaseModel):
|
||||
|
||||
|
||||
class DerenderChatRequest(BaseModel):
|
||||
"""Request for the /v1/chat/completions/derender endpoint.
|
||||
"""Request for the /v1/chat/completions/derender endpoint (non-streaming).
|
||||
|
||||
Wraps a GenerateResponse and caller-supplied metadata needed to produce
|
||||
a fully-formed ChatCompletionResponse without a GPU.
|
||||
Wraps a complete GenerateResponse and caller-supplied metadata needed to
|
||||
produce a fully-formed ChatCompletionResponse without a GPU.
|
||||
|
||||
Streaming derender would require a separate endpoint design with
|
||||
incremental token delivery, ``OutputProcessor``-based detokenization,
|
||||
and ``parser.parse_delta()`` instead of ``parser.parse()``.
|
||||
"""
|
||||
|
||||
model: str
|
||||
@@ -244,7 +248,7 @@ class DerenderChatRequest(BaseModel):
|
||||
|
||||
|
||||
class DerenderCompletionRequest(BaseModel):
|
||||
"""Request for the /v1/completions/derender endpoint.
|
||||
"""Request for the /v1/completions/derender endpoint (non-streaming).
|
||||
|
||||
Parallel to DerenderChatRequest but handles the multi-prompt completions
|
||||
case: one GenerateResponse per prompt, mirroring the list[GenerateRequest]
|
||||
|
||||
@@ -27,6 +27,7 @@ from vllm.entrypoints.openai.completion.protocol import (
|
||||
)
|
||||
from vllm.entrypoints.openai.engine.protocol import (
|
||||
ErrorResponse,
|
||||
ToolCall,
|
||||
UsageInfo,
|
||||
)
|
||||
from vllm.entrypoints.openai.engine.serving import resolve_token_id_placeholder
|
||||
@@ -43,7 +44,6 @@ from vllm.entrypoints.serve.disagg.protocol import (
|
||||
DerenderChatRequest,
|
||||
DerenderCompletionRequest,
|
||||
GenerateRequest,
|
||||
GenerateResponseChoice,
|
||||
MultiModalFeatures,
|
||||
PlaceholderRangeInfo,
|
||||
)
|
||||
@@ -76,21 +76,83 @@ from vllm.utils.mistral import mt as _mt
|
||||
logger = init_logger(__name__)
|
||||
|
||||
|
||||
def _parse_token_id_placeholder(token: str) -> int | None:
|
||||
"""Extract token ID from a 'token_id:N' placeholder string."""
|
||||
if not token.startswith("token_id:"):
|
||||
return None
|
||||
try:
|
||||
return int(token[len("token_id:") :])
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def _correct_decoded_token(
|
||||
token_id: int, context_token_ids: list[int], tokenizer: TokenizerLike
|
||||
) -> str:
|
||||
"""Use preceding tokens as context to fix U+FFFD from byte-fallback.
|
||||
|
||||
Mirrors LogprobsProcessor._correct_decoded_token in v1/engine/logprobs.py.
|
||||
"""
|
||||
max_ctx = min(len(context_token_ids), 4)
|
||||
|
||||
for num_ctx in range(1, max_ctx + 1):
|
||||
context = context_token_ids[-num_ctx:]
|
||||
full_decoded = tokenizer.decode(context + [token_id])
|
||||
|
||||
if full_decoded.endswith("�"):
|
||||
continue
|
||||
|
||||
clean_end = len(context)
|
||||
for j in range(len(context) - 1, -1, -1):
|
||||
if tokenizer.decode([context[j]]).endswith("�"):
|
||||
clean_end = j
|
||||
else:
|
||||
break
|
||||
|
||||
clean_prefix = tokenizer.decode(context[:clean_end]) if clean_end > 0 else ""
|
||||
|
||||
if full_decoded.startswith(clean_prefix):
|
||||
return full_decoded[len(clean_prefix) :]
|
||||
|
||||
common_len = 0
|
||||
for a, b in zip(clean_prefix, full_decoded):
|
||||
if a != b:
|
||||
break
|
||||
common_len += 1
|
||||
return full_decoded[common_len:]
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def _resolve_logprobs(
|
||||
logprobs: ChatCompletionLogProbs, tokenizer: TokenizerLike
|
||||
) -> ChatCompletionLogProbs:
|
||||
"""Resolve all token_id:N placeholders in a ChatCompletionLogProbs object."""
|
||||
"""Resolve token_id:N placeholders in a ChatCompletionLogProbs object."""
|
||||
if logprobs.content is None:
|
||||
return logprobs
|
||||
|
||||
context_token_ids: list[int] = []
|
||||
resolved_content = []
|
||||
|
||||
for entry in logprobs.content:
|
||||
token_str, token_bytes = resolve_token_id_placeholder(entry.token, tokenizer)
|
||||
sampled_id = _parse_token_id_placeholder(entry.token)
|
||||
|
||||
if token_str.endswith("�") and sampled_id is not None:
|
||||
token_str = _correct_decoded_token(sampled_id, context_token_ids, tokenizer)
|
||||
token_bytes = list(token_str.encode("utf-8"))
|
||||
|
||||
resolved_top = []
|
||||
for top in entry.top_logprobs:
|
||||
top_str, top_bytes = resolve_token_id_placeholder(top.token, tokenizer)
|
||||
top_id = _parse_token_id_placeholder(top.token)
|
||||
if top_str.endswith("�") and top_id is not None:
|
||||
top_str = _correct_decoded_token(top_id, context_token_ids, tokenizer)
|
||||
top_bytes = list(top_str.encode("utf-8"))
|
||||
resolved_top.append(
|
||||
top.model_copy(update={"token": top_str, "bytes": top_bytes})
|
||||
)
|
||||
|
||||
resolved_content.append(
|
||||
entry.model_copy(
|
||||
update={
|
||||
@@ -100,6 +162,10 @@ def _resolve_logprobs(
|
||||
}
|
||||
)
|
||||
)
|
||||
|
||||
if sampled_id is not None:
|
||||
context_token_ids.append(sampled_id)
|
||||
|
||||
return ChatCompletionLogProbs(content=resolved_content)
|
||||
|
||||
|
||||
@@ -136,30 +202,6 @@ def _convert_chat_logprobs_to_completion_logprobs(
|
||||
)
|
||||
|
||||
|
||||
def _build_chat_choice(
|
||||
choice: GenerateResponseChoice, tokenizer: TokenizerLike
|
||||
) -> ChatCompletionResponseChoice:
|
||||
"""Detokenize and resolve logprobs for a single GenerateResponseChoice.
|
||||
|
||||
Raises:
|
||||
ValueError: if choice.token_ids is empty or None.
|
||||
"""
|
||||
if not choice.token_ids:
|
||||
raise ValueError(f"choice {choice.index} has empty or null token_ids")
|
||||
decoded_text = tokenizer.decode(choice.token_ids, skip_special_tokens=True)
|
||||
resolved_logprobs = (
|
||||
_resolve_logprobs(choice.logprobs, tokenizer)
|
||||
if choice.logprobs is not None
|
||||
else None
|
||||
)
|
||||
return ChatCompletionResponseChoice(
|
||||
index=choice.index,
|
||||
message=ChatMessage(role="assistant", content=decoded_text),
|
||||
logprobs=resolved_logprobs,
|
||||
finish_reason=choice.finish_reason,
|
||||
)
|
||||
|
||||
|
||||
class OpenAIServingRender:
|
||||
def __init__(
|
||||
self,
|
||||
@@ -536,9 +578,12 @@ class OpenAIServingRender:
|
||||
) -> ChatCompletionResponse | ErrorResponse:
|
||||
"""Postprocess a GenerateResponse into a ChatCompletionResponse.
|
||||
|
||||
This is the symmetric inverse of render_chat_request: it detokenizes
|
||||
output token IDs, resolves token_id:N logprob placeholders, and
|
||||
formats the result as an OpenAI-compatible chat completion response.
|
||||
Non-streaming only: expects the complete GenerateResponse with all
|
||||
token IDs present. Uses ``parser.parse()`` for one-shot extraction.
|
||||
|
||||
When ``request.chat_request`` is provided, the parser splits the
|
||||
output into (reasoning, content, tool_calls). Otherwise falls
|
||||
back to plain detokenization.
|
||||
"""
|
||||
error_check_ret = await self._check_model(request)
|
||||
if error_check_ret is not None:
|
||||
@@ -546,11 +591,89 @@ class OpenAIServingRender:
|
||||
|
||||
tokenizer = self.renderer.get_tokenizer()
|
||||
gen = request.generate_response
|
||||
chat_request = request.chat_request
|
||||
choices: list[ChatCompletionResponseChoice] = []
|
||||
|
||||
try:
|
||||
for choice in gen.choices:
|
||||
choices.append(_build_chat_choice(choice, tokenizer))
|
||||
if not choice.token_ids:
|
||||
raise ValueError(
|
||||
f"choice {choice.index} has empty or null token_ids"
|
||||
)
|
||||
|
||||
resolved_logprobs = (
|
||||
_resolve_logprobs(choice.logprobs, tokenizer)
|
||||
if choice.logprobs is not None
|
||||
else None
|
||||
)
|
||||
|
||||
if self.parser is not None and chat_request is not None:
|
||||
# Parser path: decode with special tokens preserved
|
||||
# so the parser can see markers like </think>,
|
||||
# <tool_call>, or Harmony channel tokens.
|
||||
decoded_text = tokenizer.decode(
|
||||
choice.token_ids, skip_special_tokens=False
|
||||
)
|
||||
|
||||
chat_template_kwargs: dict[str, Any] = {}
|
||||
if not self.use_harmony:
|
||||
chat_template_kwargs = (
|
||||
chat_request.build_chat_params(
|
||||
self.chat_template,
|
||||
self.chat_template_content_format,
|
||||
)
|
||||
.with_defaults(self.default_chat_template_kwargs)
|
||||
.chat_template_kwargs
|
||||
)
|
||||
|
||||
parser = self.parser(
|
||||
tokenizer,
|
||||
chat_request.tools,
|
||||
chat_template_kwargs=chat_template_kwargs,
|
||||
)
|
||||
reasoning, content, tool_calls = parser.parse(
|
||||
decoded_text,
|
||||
chat_request,
|
||||
enable_auto_tools=self.enable_auto_tools,
|
||||
model_output_token_ids=choice.token_ids,
|
||||
)
|
||||
|
||||
if not getattr(chat_request, "include_reasoning", True):
|
||||
reasoning = None
|
||||
|
||||
tc_items = (
|
||||
[
|
||||
ToolCall(
|
||||
id=random_uuid(),
|
||||
function=tc,
|
||||
)
|
||||
for tc in tool_calls
|
||||
]
|
||||
if tool_calls
|
||||
else []
|
||||
)
|
||||
|
||||
message = ChatMessage(
|
||||
role="assistant",
|
||||
reasoning=reasoning,
|
||||
content=content,
|
||||
tool_calls=tc_items,
|
||||
)
|
||||
else:
|
||||
# No parser: plain detokenization.
|
||||
decoded_text = tokenizer.decode(
|
||||
choice.token_ids, skip_special_tokens=True
|
||||
)
|
||||
message = ChatMessage(role="assistant", content=decoded_text)
|
||||
|
||||
choices.append(
|
||||
ChatCompletionResponseChoice(
|
||||
index=choice.index,
|
||||
message=message,
|
||||
logprobs=resolved_logprobs,
|
||||
finish_reason=choice.finish_reason,
|
||||
)
|
||||
)
|
||||
except ValueError as exc:
|
||||
return self.create_error_response(str(exc))
|
||||
|
||||
@@ -587,8 +710,9 @@ class OpenAIServingRender:
|
||||
) -> CompletionResponse | ErrorResponse:
|
||||
"""Postprocess a list of GenerateResponses into a CompletionResponse.
|
||||
|
||||
Mirrors the multi-prompt completions case: one GenerateResponse per
|
||||
prompt, parallel to the list[GenerateRequest] from /v1/completions/render.
|
||||
Non-streaming only. Mirrors the multi-prompt completions case: one
|
||||
GenerateResponse per prompt, parallel to the list[GenerateRequest]
|
||||
from /v1/completions/render.
|
||||
"""
|
||||
error_check_ret = await self._check_model(request)
|
||||
if error_check_ret is not None:
|
||||
|
||||
+7
-1
@@ -209,6 +209,7 @@ if TYPE_CHECKING:
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: int = 300
|
||||
VLLM_WORKER_SHUTDOWN_TIMEOUT_SECONDS: int = 5
|
||||
VLLM_KV_CACHE_LAYOUT: Literal["NHD", "HND"] | None = None
|
||||
VLLM_USE_PACKED_HMA_KV_CACHE: bool = False
|
||||
VLLM_SSM_CONV_STATE_LAYOUT: Literal["SD", "DS"] | None = None
|
||||
VLLM_COMPUTE_NANS_IN_LOGITS: bool = False
|
||||
VLLM_ROCM_QUICK_REDUCE_QUANTIZATION: Literal[
|
||||
@@ -485,7 +486,7 @@ def get_vllm_port() -> int | None:
|
||||
raise ValueError(
|
||||
f"VLLM_PORT '{port}' appears to be a URI. "
|
||||
"This may be caused by a Kubernetes service discovery issue,"
|
||||
"check the warning in: https://docs.vllm.ai/en/stable/serving/env_vars.html"
|
||||
"check the warning in: https://docs.vllm.ai/en/latest/configuration/env_vars.html"
|
||||
) from None
|
||||
raise ValueError(f"VLLM_PORT '{port}' must be a valid integer") from err
|
||||
|
||||
@@ -1608,6 +1609,11 @@ environment_variables: dict[str, Callable[[], Any]] = {
|
||||
"VLLM_KV_CACHE_LAYOUT": env_with_choices(
|
||||
"VLLM_KV_CACHE_LAYOUT", None, ["NHD", "HND"]
|
||||
),
|
||||
# Opt into packed per-block KV cache allocation for multi-group
|
||||
# attention-only HMA models (e.g. gpt-oss, Gemma 3/4).
|
||||
"VLLM_USE_PACKED_HMA_KV_CACHE": lambda: bool(
|
||||
int(os.getenv("VLLM_USE_PACKED_HMA_KV_CACHE", "0"))
|
||||
),
|
||||
# SSM conv state layout used for Mamba models.
|
||||
# - SD: (state_len, dim) — dim contiguous (default)
|
||||
# - DS: (dim, state_len) — TP-sharded dim on dim1,
|
||||
|
||||
@@ -17,7 +17,7 @@ from vllm.lora.utils import (
|
||||
)
|
||||
from vllm.model_executor.model_loader.tensorizer import TensorizerConfig
|
||||
from vllm.model_executor.models.utils import WeightsMapper
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
@@ -126,7 +126,7 @@ class LoRAModel:
|
||||
skip_prefixes: list[str] | None = None,
|
||||
) -> "LoRAModel":
|
||||
"""Create a LoRAModel from a dictionary of tensors."""
|
||||
pin_memory = str(device) == "cpu" and is_pin_memory_available()
|
||||
pin_memory = str(device) == "cpu" and PIN_MEMORY
|
||||
loras: dict[str, LoRALayerWeights] = {}
|
||||
for tensor_name, tensor in tensors.items():
|
||||
if is_base_embedding_weights(tensor_name):
|
||||
|
||||
@@ -7,7 +7,7 @@ import torch
|
||||
import torch.types
|
||||
|
||||
from vllm.lora.peft_helper import PEFTHelper
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
|
||||
class LoRALayerWeights:
|
||||
@@ -79,7 +79,7 @@ class LoRALayerWeights:
|
||||
dtype: torch.dtype,
|
||||
device: torch.types.Device,
|
||||
) -> "LoRALayerWeights":
|
||||
pin_memory = str(device) == "cpu" and is_pin_memory_available()
|
||||
pin_memory = str(device) == "cpu" and PIN_MEMORY
|
||||
lora_a = torch.zeros(
|
||||
[rank, input_dim], dtype=dtype, device=device, pin_memory=pin_memory
|
||||
)
|
||||
|
||||
@@ -42,7 +42,7 @@ from vllm.model_executor.models.utils import PPMissingLayer
|
||||
from vllm.multimodal import MULTIMODAL_REGISTRY
|
||||
from vllm.multimodal.encoder_budget import MultiModalBudget
|
||||
from vllm.utils.cache import LRUCache
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
@@ -801,7 +801,7 @@ class LoRAModelManager:
|
||||
# 2. The weight packing above (e.g., pack_moe) may invalidate the
|
||||
# pin_memory allocation, so we execute it after packing.
|
||||
|
||||
pin_memory = str(lora_device) == "cpu" and is_pin_memory_available()
|
||||
pin_memory = str(lora_device) == "cpu" and PIN_MEMORY
|
||||
if pin_memory:
|
||||
for lora in lora_model.loras.values():
|
||||
if isinstance(lora.lora_a, list):
|
||||
|
||||
@@ -1684,12 +1684,13 @@ class MLACommonMetadataBuilder(AttentionMetadataBuilder[M]):
|
||||
# [[0, 0, 0, 0], [256, 256, 256, 256], [512, 512, 512, 512]]
|
||||
# Note(simon): this is done in CPU because of downstream's
|
||||
# of `to_list`.
|
||||
chunk_starts = (
|
||||
chunk_starts = torch.empty(
|
||||
num_chunks, num_prefills, dtype=torch.int32, pin_memory=True
|
||||
).copy_(
|
||||
torch.arange(num_chunks, dtype=torch.int32)
|
||||
.multiply_(max_context_chunk)
|
||||
.unsqueeze(1)
|
||||
.expand(-1, num_prefills)
|
||||
* max_context_chunk
|
||||
).pin_memory()
|
||||
)
|
||||
chunk_ends = torch.min(
|
||||
context_lens_cpu.unsqueeze(0), chunk_starts + max_context_chunk
|
||||
)
|
||||
@@ -1746,12 +1747,13 @@ class MLACommonMetadataBuilder(AttentionMetadataBuilder[M]):
|
||||
)
|
||||
* self.dcp_local_block_size
|
||||
)
|
||||
local_chunk_starts = (
|
||||
local_chunk_starts = torch.empty(
|
||||
num_chunks, num_prefills, dtype=torch.int32, pin_memory=True
|
||||
).copy_(
|
||||
torch.arange(num_chunks, dtype=torch.int32)
|
||||
.multiply_(padded_local_max_context_chunk_across_ranks)
|
||||
.unsqueeze(1)
|
||||
.expand(-1, num_prefills)
|
||||
* padded_local_max_context_chunk_across_ranks
|
||||
).pin_memory()
|
||||
)
|
||||
local_chunk_ends = torch.min(
|
||||
padded_local_context_lens_cpu.unsqueeze(0),
|
||||
local_chunk_starts
|
||||
|
||||
@@ -28,6 +28,7 @@ from vllm.utils.flashinfer import (
|
||||
is_flashinfer_cudnn_fp8_prefill_attn_supported,
|
||||
)
|
||||
from vllm.utils.math_utils import round_up
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.v1.attention.backends.fa_utils import get_flash_attn_version
|
||||
from vllm.v1.attention.backends.registry import AttentionBackendEnum
|
||||
from vllm.v1.attention.ops.vit_attn_wrappers import (
|
||||
@@ -311,7 +312,7 @@ class MMEncoderAttention(CustomOp):
|
||||
)
|
||||
cu_seqlens = np.concatenate([cu_seqlens_qko, cu_seqlens_v])
|
||||
|
||||
cu_seqlens = torch.from_numpy(cu_seqlens).to(device, non_blocking=True)
|
||||
cu_seqlens = async_tensor_h2d(cu_seqlens, device=device)
|
||||
return cu_seqlens
|
||||
|
||||
def __init__(
|
||||
|
||||
@@ -21,7 +21,6 @@ from vllm.model_executor.layers.fused_moe.config import (
|
||||
FusedMoEQuantConfig,
|
||||
)
|
||||
from vllm.model_executor.layers.fused_moe.experts.triton_moe import TritonExperts
|
||||
from vllm.model_executor.layers.fused_moe.utils import moe_kernel_quantize_input
|
||||
from vllm.model_executor.layers.quantization.utils.nvfp4_emulation_utils import (
|
||||
dequantize_to_dtype,
|
||||
)
|
||||
@@ -135,14 +134,6 @@ class Nvfp4QuantizationEmulationTritonExperts(TritonExperts):
|
||||
swizzle=False,
|
||||
)
|
||||
|
||||
hidden_states, _ = moe_kernel_quantize_input(
|
||||
A=hidden_states,
|
||||
A_scale=self.quant_config.a1_gscale,
|
||||
quant_dtype="nvfp4",
|
||||
per_act_token_quant=False,
|
||||
quantization_emulation=True,
|
||||
)
|
||||
|
||||
# Activation quantization/dequantization is deferred to
|
||||
# `moe_kernel_quantize_input` in TritonExperts.apply.
|
||||
super().apply(
|
||||
|
||||
@@ -21,7 +21,6 @@ from vllm.model_executor.layers.fused_moe.config import (
|
||||
FusedMoEQuantConfig,
|
||||
)
|
||||
from vllm.model_executor.layers.fused_moe.experts.triton_moe import TritonExperts
|
||||
from vllm.model_executor.layers.fused_moe.utils import moe_kernel_quantize_input
|
||||
from vllm.model_executor.layers.quantization.utils.mxfp4_utils import dequant_mxfp4
|
||||
from vllm.model_executor.layers.quantization.utils.mxfp6_utils import dequant_mxfp6
|
||||
from vllm.model_executor.layers.quantization.utils.ocp_mx_utils import (
|
||||
@@ -155,16 +154,6 @@ class OCP_MXQuantizationEmulationTritonExperts(TritonExperts):
|
||||
w2, self.w2_scale_val, hidden_states.dtype
|
||||
)
|
||||
|
||||
# Apply activation QDQ if needed by the OCP MX scheme
|
||||
hidden_states, _ = moe_kernel_quantize_input(
|
||||
A=hidden_states,
|
||||
A_scale=None,
|
||||
quant_dtype=self.quant_config.quant_dtype,
|
||||
per_act_token_quant=False,
|
||||
ocp_mx_scheme=self.ocp_mx_scheme,
|
||||
quantization_emulation=True,
|
||||
)
|
||||
|
||||
# Activation quantization/dequantization is deferred to
|
||||
# `moe_kernel_quantize_input` in TritonExperts.apply.
|
||||
super().apply(
|
||||
|
||||
@@ -245,7 +245,7 @@ class TritonExperts(LoRAExpertsMixin, mk.FusedMoEExpertsModular):
|
||||
lora_unquantized_hidden_states = hidden_states
|
||||
hidden_states, a1q_scale = moe_kernel_quantize_input(
|
||||
hidden_states,
|
||||
self.a1_scale,
|
||||
self.a1_scale or self.a1_gscale,
|
||||
self.quant_dtype,
|
||||
self.per_act_token_quant,
|
||||
self.block_shape,
|
||||
|
||||
@@ -296,6 +296,7 @@ def moe_kernel_quantize_input(
|
||||
if not quantization_emulation:
|
||||
return _nvfp4_quantize(A, A_scale, is_sf_swizzled_layout=is_scale_swizzled)
|
||||
else:
|
||||
assert A_scale is not None
|
||||
A = ref_nvfp4_quant_dequant(A, A_scale, block_size=16)
|
||||
return A, None
|
||||
elif quant_dtype == "mxfp4":
|
||||
|
||||
@@ -10,6 +10,7 @@ import torch.nn as nn
|
||||
from vllm.config.pooler import SequencePoolingType
|
||||
from vllm.model_executor.layers.pooler import PoolingParamsUpdate
|
||||
from vllm.tasks import PoolingTask
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.v1.pool.metadata import PoolingMetadata
|
||||
|
||||
SequencePoolingMethodOutput: TypeAlias = torch.Tensor | list[torch.Tensor]
|
||||
@@ -74,15 +75,14 @@ class MeanPool(SequencePoolingMethod):
|
||||
# early return for empty batch
|
||||
return hidden_states.new_empty((0, hidden_size), dtype=torch.float32)
|
||||
|
||||
# Build segment_ids on CPU so repeat_interleave doesn't need to sync
|
||||
# GPU->CPU to learn its data-dependent output length, then upload
|
||||
# non-blocking. eg. [2, 1, 3] -> [0, 0, 1, 2, 2, 2]
|
||||
prompt_lens = async_tensor_h2d(
|
||||
prompt_lens_cpu, device=hidden_states.device, dtype=torch.int64
|
||||
)
|
||||
# eg. [2, 1, 3] -> [0, 0, 1, 2, 2, 2]
|
||||
segment_ids = torch.repeat_interleave(
|
||||
torch.arange(num_seqs, dtype=torch.long),
|
||||
prompt_lens_cpu,
|
||||
).to(hidden_states.device, non_blocking=True)
|
||||
prompt_lens = prompt_lens_cpu.to(
|
||||
hidden_states.device, dtype=torch.int64, non_blocking=True
|
||||
torch.arange(num_seqs, device=hidden_states.device, dtype=torch.long),
|
||||
prompt_lens,
|
||||
output_size=int(prompt_lens_cpu.sum()),
|
||||
)
|
||||
segment_sums = torch.zeros(
|
||||
(num_seqs, hidden_size),
|
||||
|
||||
@@ -1001,24 +1001,30 @@ class DeepseekV2MLAAttention(nn.Module):
|
||||
# IndexCache config
|
||||
# Refer: https://arxiv.org/abs/2603.12201 for more details.
|
||||
_skip_topk = False
|
||||
_index_topk_freq = getattr(config, "index_topk_freq", 1)
|
||||
_index_topk_pattern = getattr(config, "index_topk_pattern", None)
|
||||
_index_skip_topk_offset = getattr(config, "index_skip_topk_offset", 2)
|
||||
layer_id = extract_layer_index(prefix)
|
||||
is_mtp_layer = False
|
||||
if self.is_v32:
|
||||
_index_topk_freq = getattr(config, "index_topk_freq", 1)
|
||||
_index_topk_pattern = getattr(config, "index_topk_pattern", None)
|
||||
_index_skip_topk_offset = getattr(config, "index_skip_topk_offset", 2)
|
||||
layer_id = extract_layer_index(prefix)
|
||||
|
||||
if _index_topk_pattern is None:
|
||||
_skip_topk = (
|
||||
max(layer_id - _index_skip_topk_offset + 1, 0) % _index_topk_freq != 0
|
||||
if _index_topk_pattern is None:
|
||||
_skip_topk = (
|
||||
max(layer_id - _index_skip_topk_offset + 1, 0) % _index_topk_freq
|
||||
!= 0
|
||||
)
|
||||
elif 0 <= layer_id < len(_index_topk_pattern):
|
||||
_skip_topk = _index_topk_pattern[layer_id] == "S"
|
||||
|
||||
# The skip pattern only governs backbone layers. MTP/nextn
|
||||
# layers (layer_id >= num_hidden_layers) always build a full
|
||||
# indexer: they compute indices at draft step 0 and toggle
|
||||
# at runtime via set_skip_topk
|
||||
# (index_share_for_mtp_iteration).
|
||||
_num_hidden_layers = getattr(config, "num_hidden_layers", None)
|
||||
is_mtp_layer = (
|
||||
_num_hidden_layers is not None and layer_id >= _num_hidden_layers
|
||||
)
|
||||
elif 0 <= layer_id < len(_index_topk_pattern):
|
||||
_skip_topk = _index_topk_pattern[layer_id] == "S"
|
||||
|
||||
# The skip pattern only governs backbone layers. MTP/nextn layers
|
||||
# (layer_id >= num_hidden_layers) always build a full indexer: they
|
||||
# compute indices at draft step 0 and toggle at runtime via
|
||||
# set_skip_topk (index_share_for_mtp_iteration).
|
||||
_num_hidden_layers = getattr(config, "num_hidden_layers", None)
|
||||
is_mtp_layer = _num_hidden_layers is not None and layer_id >= _num_hidden_layers
|
||||
|
||||
if self.is_v32 and (not _skip_topk or is_mtp_layer):
|
||||
self.indexer_rope_emb = get_rope(
|
||||
|
||||
@@ -66,6 +66,7 @@ from vllm.model_executor.models.utils import maybe_prefix
|
||||
from vllm.model_executor.models.vision import is_vit_use_data_parallel
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.transformers_utils.configs.moonvit import MoonViTConfig
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
|
||||
|
||||
def _apply_rope_input_validation(x, freqs_cis):
|
||||
@@ -758,7 +759,7 @@ class MoonVitPretrainedModel(PreTrainedModel):
|
||||
),
|
||||
]
|
||||
)
|
||||
metadata["cu_seqlens"] = torch.from_numpy(cu_seqlens_np).to(device)
|
||||
metadata["cu_seqlens"] = async_tensor_h2d(cu_seqlens_np, device=device)
|
||||
|
||||
if max_seqlen_override is not None:
|
||||
max_seqlen_val = int(max_seqlen_override)
|
||||
@@ -770,7 +771,7 @@ class MoonVitPretrainedModel(PreTrainedModel):
|
||||
metadata["max_seqlen"] = torch.tensor(max_seqlen_val, dtype=torch.int32)
|
||||
|
||||
gather_idx_np = _build_merge_gather_idx(grid_pairs, self.merge_kernel_size)
|
||||
metadata["merge_gather_idx"] = torch.from_numpy(gather_idx_np).to(device)
|
||||
metadata["merge_gather_idx"] = async_tensor_h2d(gather_idx_np, device=device)
|
||||
|
||||
return metadata
|
||||
|
||||
|
||||
@@ -83,9 +83,8 @@ from vllm.multimodal.parse import MultiModalDataItems
|
||||
from vllm.multimodal.processing import PromptReplacement, PromptUpdate
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.sequence import IntermediateTensors
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.tensor_schema import TensorSchema, TensorShape
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.utils.torch_utils import PIN_MEMORY, async_tensor_h2d
|
||||
from vllm.v1.attention.backends.registry import AttentionBackendEnum
|
||||
from vllm.v1.worker.encoder_cudagraph_defs import EncoderCudaGraphReplayBuffers
|
||||
|
||||
@@ -825,7 +824,7 @@ class Qwen2_5_VisionTransformer(nn.Module):
|
||||
@staticmethod
|
||||
def invert_permutation(perm: torch.Tensor) -> torch.Tensor:
|
||||
# building the inverse permutation in O(n) time
|
||||
inv = torch.empty_like(perm, pin_memory=is_pin_memory_available())
|
||||
inv = torch.empty_like(perm, pin_memory=PIN_MEMORY)
|
||||
inv[perm] = torch.arange(perm.numel(), device=perm.device, dtype=perm.dtype)
|
||||
return inv
|
||||
|
||||
|
||||
@@ -13,6 +13,7 @@ from vllm.config.cache import CacheDType
|
||||
from vllm.platforms.interface import DeviceCapability
|
||||
from vllm.triton_utils import tl, triton
|
||||
from vllm.utils.math_utils import cdiv
|
||||
from vllm.utils.torch_utils import np_to_pinned_tensor
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
AttentionCGSupport,
|
||||
@@ -207,7 +208,7 @@ class DeepseekV4FlashMLAMetadataBuilder(
|
||||
# Zero-fill for cudagraphs
|
||||
self.req_id_per_token_buffer.fill_(0)
|
||||
self.req_id_per_token_buffer[: req_id_per_token.shape[0]].copy_(
|
||||
torch.from_numpy(req_id_per_token), non_blocking=True
|
||||
np_to_pinned_tensor(req_id_per_token), non_blocking=True
|
||||
)
|
||||
req_id_per_token = self.req_id_per_token_buffer[:num_tokens]
|
||||
|
||||
|
||||
@@ -488,7 +488,13 @@ class MultiModalBatchedField(BaseMultiModalField):
|
||||
# An optimization when `batch` contains only one tensor:
|
||||
# - produce exactly same result as `torch.stack(batch)`
|
||||
# - will achieve zero-copy if the tensor is contiguous
|
||||
return batch[0].unsqueeze(0).contiguous()
|
||||
out = batch[0].unsqueeze(0)
|
||||
if not pin_memory:
|
||||
return out.contiguous()
|
||||
# Avoid extra copy - pinning unpinned memory will make it contiguous
|
||||
if not out.is_contiguous() and out.is_pinned():
|
||||
out = out.contiguous()
|
||||
return out.pin_memory()
|
||||
first_shape = batch[0].shape
|
||||
if all(elem.shape == first_shape for elem in batch):
|
||||
out = torch.empty(
|
||||
@@ -538,7 +544,13 @@ class MultiModalFlatField(BaseMultiModalField):
|
||||
# An optimization when `batch` contains only one tensor:
|
||||
# - produce exactly same result as `torch.concat(batch)`
|
||||
# - will achieve zero-copy if the tensor is contiguous
|
||||
return batch[0].contiguous()
|
||||
out = batch[0]
|
||||
if not pin_memory:
|
||||
return out.contiguous()
|
||||
# Avoid extra copy - pinning unpinned memory will make it contiguous
|
||||
if not out.is_contiguous() and out.is_pinned():
|
||||
out = out.contiguous()
|
||||
return out.pin_memory()
|
||||
|
||||
dim = self.dim + (self.dim < 0) * len(batch[0].shape)
|
||||
|
||||
|
||||
@@ -0,0 +1,163 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
"""MiniMax M3 parser for reasoning markers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import functools
|
||||
from collections.abc import Iterable, Sequence
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from vllm.parser.engine.events import EventType
|
||||
from vllm.parser.engine.parser_engine import ParserEngine
|
||||
from vllm.parser.engine.parser_engine_config import (
|
||||
ParserEngineConfig,
|
||||
ParserState,
|
||||
Transition,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from vllm.tokenizers import TokenizerLike
|
||||
from vllm.tool_parsers.abstract_tool_parser import Tool
|
||||
|
||||
THINK_START = "<mm:think>"
|
||||
THINK_END = "</mm:think>"
|
||||
|
||||
|
||||
@functools.cache
|
||||
def minimax_m3_config(thinking: bool = False) -> ParserEngineConfig:
|
||||
return ParserEngineConfig(
|
||||
name="minimax_m3",
|
||||
initial_state=ParserState.REASONING if thinking else ParserState.CONTENT,
|
||||
terminals={
|
||||
"THINK_START": THINK_START,
|
||||
"THINK_END": THINK_END,
|
||||
},
|
||||
transitions={
|
||||
(ParserState.CONTENT, "THINK_START"): Transition(
|
||||
ParserState.REASONING,
|
||||
(EventType.REASONING_START,),
|
||||
),
|
||||
(ParserState.REASONING, "THINK_START"): Transition(
|
||||
ParserState.REASONING,
|
||||
(),
|
||||
),
|
||||
(ParserState.REASONING, "THINK_END"): Transition(
|
||||
ParserState.CONTENT,
|
||||
(EventType.REASONING_END,),
|
||||
),
|
||||
(ParserState.CONTENT, "THINK_END"): Transition(
|
||||
ParserState.CONTENT,
|
||||
(),
|
||||
),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
class MiniMaxM3Parser(ParserEngine):
|
||||
"""MiniMax M3 parser backed by the declarative parser engine."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
tokenizer: TokenizerLike,
|
||||
tools: list[Tool] | None = None,
|
||||
**kwargs,
|
||||
) -> None:
|
||||
chat_kwargs = kwargs.get("chat_template_kwargs", {}) or {}
|
||||
self._initial_in_reasoning = chat_kwargs.get("thinking_mode") == "enabled"
|
||||
kwargs.setdefault(
|
||||
"parser_engine_config",
|
||||
minimax_m3_config(thinking=self._initial_in_reasoning),
|
||||
)
|
||||
super().__init__(tokenizer, tools, **kwargs)
|
||||
self._start_token_ids = self._encode_marker(THINK_START)
|
||||
self._end_token_ids = self._encode_marker(THINK_END)
|
||||
|
||||
def _encode_marker(self, marker: str) -> tuple[int, ...]:
|
||||
try:
|
||||
token_ids = self.model_tokenizer.encode(marker, add_special_tokens=False)
|
||||
except TypeError:
|
||||
token_ids = self.model_tokenizer.encode(marker)
|
||||
return tuple(token_ids)
|
||||
|
||||
@staticmethod
|
||||
def _contains_token_sequence(
|
||||
token_ids: Sequence[int], marker_ids: Sequence[int]
|
||||
) -> bool:
|
||||
if not marker_ids or len(marker_ids) > len(token_ids):
|
||||
return False
|
||||
marker_len = len(marker_ids)
|
||||
return any(
|
||||
tuple(token_ids[i : i + marker_len]) == tuple(marker_ids)
|
||||
for i in range(len(token_ids) - marker_len + 1)
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _rfind_token_sequence(
|
||||
token_ids: Sequence[int], marker_ids: Sequence[int]
|
||||
) -> int:
|
||||
if not marker_ids or len(marker_ids) > len(token_ids):
|
||||
return -1
|
||||
marker_len = len(marker_ids)
|
||||
for i in range(len(token_ids) - marker_len, -1, -1):
|
||||
if tuple(token_ids[i : i + marker_len]) == tuple(marker_ids):
|
||||
return i
|
||||
return -1
|
||||
|
||||
def is_reasoning_end(self, input_ids: list[int]) -> bool:
|
||||
start_index = self._rfind_token_sequence(input_ids, self._start_token_ids)
|
||||
end_index = self._rfind_token_sequence(input_ids, self._end_token_ids)
|
||||
if end_index < 0:
|
||||
return False
|
||||
if start_index < 0:
|
||||
return True
|
||||
return end_index > start_index
|
||||
|
||||
def is_reasoning_end_streaming(
|
||||
self, input_ids: Sequence[int], delta_ids: Iterable[int]
|
||||
) -> bool:
|
||||
if self.reasoning_ended:
|
||||
return True
|
||||
if self._engine._lexer.buffer:
|
||||
return False
|
||||
if self._initial_in_reasoning:
|
||||
return False
|
||||
if self._engine.state == ParserState.CONTENT:
|
||||
return bool(input_ids)
|
||||
return False
|
||||
|
||||
def extract_content_ids(self, input_ids: list[int]) -> list[int]:
|
||||
end_index = self._rfind_token_sequence(input_ids, self._end_token_ids)
|
||||
if end_index >= 0:
|
||||
return input_ids[end_index + len(self._end_token_ids) :]
|
||||
|
||||
has_start = self._contains_token_sequence(input_ids, self._start_token_ids)
|
||||
if self._initial_in_reasoning and not has_start:
|
||||
return []
|
||||
|
||||
if not has_start:
|
||||
return input_ids
|
||||
return []
|
||||
|
||||
def count_reasoning_tokens(self, token_ids: Sequence[int]) -> int:
|
||||
count = 0
|
||||
depth = 1 if self._initial_in_reasoning else 0
|
||||
i = 0
|
||||
while i < len(token_ids):
|
||||
if tuple(token_ids[i : i + len(self._start_token_ids)]) == (
|
||||
self._start_token_ids
|
||||
):
|
||||
depth += 1
|
||||
i += len(self._start_token_ids)
|
||||
continue
|
||||
if tuple(token_ids[i : i + len(self._end_token_ids)]) == (
|
||||
self._end_token_ids
|
||||
):
|
||||
if depth > 0:
|
||||
depth -= 1
|
||||
i += len(self._end_token_ids)
|
||||
continue
|
||||
if depth > 0:
|
||||
count += 1
|
||||
i += 1
|
||||
return count
|
||||
@@ -9,7 +9,6 @@ from typing import TYPE_CHECKING
|
||||
from vllm import envs
|
||||
from vllm.plugins import PLATFORM_PLUGINS_GROUP, load_plugins_by_group
|
||||
from vllm.utils.import_utils import resolve_obj_by_qualname
|
||||
from vllm.utils.torch_utils import supports_xccl
|
||||
|
||||
from .interface import CpuArchEnum, Platform, PlatformEnum
|
||||
|
||||
@@ -135,7 +134,7 @@ def xpu_platform_plugin() -> str | None:
|
||||
try:
|
||||
import torch
|
||||
|
||||
if supports_xccl():
|
||||
if torch.distributed.is_xccl_available():
|
||||
dist_backend = "xccl"
|
||||
from vllm.platforms.xpu import XPUPlatform
|
||||
|
||||
|
||||
@@ -23,7 +23,6 @@ import vllm._C_stable_libtorch # noqa
|
||||
import vllm.envs as envs
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.import_utils import import_pynvml
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache
|
||||
from vllm.v1.attention.backends.registry import AttentionBackendEnum
|
||||
|
||||
from .interface import DeviceCapability, Platform, PlatformEnum, in_wsl
|
||||
@@ -88,6 +87,8 @@ def _get_backend_priorities(
|
||||
kv_cache_dtype: CacheDType | None = None,
|
||||
) -> list[AttentionBackendEnum]:
|
||||
"""Get backend priorities with lazy import to avoid circular dependency."""
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache
|
||||
|
||||
if use_mla:
|
||||
if device_capability.major == 10:
|
||||
# Sparse MLA backend priorities
|
||||
|
||||
@@ -14,7 +14,6 @@ import vllm_xpu_kernels._xpu_C # noqa
|
||||
|
||||
import vllm.envs as envs
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.torch_utils import supports_xpu_graph
|
||||
from vllm.v1.attention.backends.registry import AttentionBackendEnum
|
||||
|
||||
from .interface import DeviceCapability, Platform, PlatformEnum
|
||||
@@ -178,8 +177,6 @@ class XPUPlatform(Platform):
|
||||
|
||||
@classmethod
|
||||
def check_and_update_config(cls, vllm_config: VllmConfig) -> None:
|
||||
parallel_config = vllm_config.parallel_config
|
||||
|
||||
# lazy import to avoid circular import
|
||||
from vllm.config import CUDAGraphMode
|
||||
|
||||
@@ -190,6 +187,10 @@ class XPUPlatform(Platform):
|
||||
attention_config = vllm_config.attention_config
|
||||
if attention_config.backend is None:
|
||||
attention_config.backend = AttentionBackendEnum.FLASH_ATTN
|
||||
|
||||
# lazy import to avoid circular import
|
||||
from vllm.utils.torch_utils import supports_xpu_graph
|
||||
|
||||
if not supports_xpu_graph():
|
||||
compilation_config.cudagraph_mode = CUDAGraphMode.NONE
|
||||
logger.warning(
|
||||
@@ -324,9 +325,8 @@ class XPUPlatform(Platform):
|
||||
|
||||
@classmethod
|
||||
def get_device_communicator_cls(cls) -> str:
|
||||
from vllm.utils.torch_utils import supports_xccl
|
||||
|
||||
if not supports_xccl():
|
||||
if not torch.distributed.is_xccl_available():
|
||||
# Supports xccl with PyTorch versions >= 2.8.0.dev for XPU platform
|
||||
logger.warning(
|
||||
"xccl is not enabled in this torch build, communication"
|
||||
" is not available."
|
||||
|
||||
@@ -2,170 +2,19 @@
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
|
||||
from collections.abc import Iterable, Sequence
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from vllm.entrypoints.openai.engine.protocol import DeltaMessage
|
||||
from vllm.reasoning.basic_parsers import BaseThinkingReasoningParser
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest
|
||||
from vllm.entrypoints.openai.responses.protocol import ResponsesRequest
|
||||
from vllm.parser.engine.adapters import ParserEngineReasoningAdapter
|
||||
from vllm.parser.minimax_m3 import MiniMaxM3Parser
|
||||
|
||||
|
||||
class MiniMaxM3ReasoningParser(BaseThinkingReasoningParser):
|
||||
"""Reasoning parser for MiniMax M3 explicit thinking blocks.
|
||||
class MiniMaxM3ReasoningParser(ParserEngineReasoningAdapter):
|
||||
"""Reasoning parser adapter for MiniMax M3 explicit thinking blocks."""
|
||||
|
||||
MiniMax M3 emits reasoning as:
|
||||
|
||||
<mm:think>reasoning text</mm:think>assistant content
|
||||
|
||||
The M3 tokenizer exposes both markers as complete vocabulary tokens. The
|
||||
chat template may also prefill the start marker when
|
||||
``thinking_mode="enabled"``, so generated text can begin directly inside a
|
||||
reasoning block without emitting ``<mm:think>`` again.
|
||||
"""
|
||||
|
||||
@property
|
||||
def start_token(self) -> str:
|
||||
return "<mm:think>"
|
||||
|
||||
@property
|
||||
def end_token(self) -> str:
|
||||
return "</mm:think>"
|
||||
|
||||
def __init__(self, tokenizer, *args, **kwargs):
|
||||
super().__init__(tokenizer, *args, **kwargs)
|
||||
chat_kwargs = kwargs.get("chat_template_kwargs", {}) or {}
|
||||
self._initial_in_reasoning = chat_kwargs.get("thinking_mode") == "enabled"
|
||||
self._at_response_start = True
|
||||
|
||||
def extract_reasoning(
|
||||
self,
|
||||
model_output: str,
|
||||
request: "ChatCompletionRequest | ResponsesRequest",
|
||||
) -> tuple[str | None, str | None]:
|
||||
# MiniMax M3 can start a response with a stray closer. Drop that first
|
||||
# token only; later unmatched closers stay visible as content.
|
||||
if not self._initial_in_reasoning and model_output.startswith(self.end_token):
|
||||
content = model_output[len(self.end_token) :]
|
||||
return None, content or None
|
||||
|
||||
if self._initial_in_reasoning and self.start_token not in model_output:
|
||||
reasoning, end, content = model_output.partition(self.end_token)
|
||||
if not end:
|
||||
return model_output, None
|
||||
return reasoning, content or None
|
||||
|
||||
if self.start_token not in model_output:
|
||||
return None, model_output
|
||||
|
||||
content_before, _, after_start = model_output.partition(self.start_token)
|
||||
reasoning, end, content_after = after_start.partition(self.end_token)
|
||||
if not end:
|
||||
return reasoning, content_before or None
|
||||
|
||||
return reasoning, (content_before + content_after) or None
|
||||
_parser_engine_cls = MiniMaxM3Parser
|
||||
|
||||
def is_reasoning_end_streaming(
|
||||
self, input_ids: Sequence[int], delta_ids: Iterable[int]
|
||||
) -> bool:
|
||||
delta_ids = tuple(delta_ids)
|
||||
if self.end_token_id in delta_ids:
|
||||
return True
|
||||
if self.end_token_id in input_ids:
|
||||
return True
|
||||
if self._initial_in_reasoning:
|
||||
return False
|
||||
if self.start_token_id not in input_ids:
|
||||
return bool(input_ids)
|
||||
return False
|
||||
|
||||
def extract_content_ids(self, input_ids: list[int]) -> list[int]:
|
||||
if self.end_token_id in input_ids:
|
||||
end_index = len(input_ids) - 1 - input_ids[::-1].index(self.end_token_id)
|
||||
return input_ids[end_index + 1 :]
|
||||
|
||||
if self._initial_in_reasoning and self.start_token_id not in input_ids:
|
||||
return []
|
||||
|
||||
if self.start_token_id not in input_ids:
|
||||
return input_ids
|
||||
return []
|
||||
|
||||
def extract_reasoning_streaming(
|
||||
self,
|
||||
previous_text: str,
|
||||
current_text: str,
|
||||
delta_text: str,
|
||||
previous_token_ids: Sequence[int],
|
||||
current_token_ids: Sequence[int],
|
||||
delta_token_ids: Sequence[int],
|
||||
) -> DeltaMessage | None:
|
||||
if not delta_text:
|
||||
return None
|
||||
|
||||
if self._at_response_start and not self._initial_in_reasoning:
|
||||
# Apply the leading-closer tolerance once. Later unmatched closers
|
||||
# stay visible as content.
|
||||
self._at_response_start = False
|
||||
if delta_text.startswith(self.end_token):
|
||||
delta_text = delta_text[len(self.end_token) :]
|
||||
if not delta_text:
|
||||
return None
|
||||
if delta_token_ids and delta_token_ids[0] == self.end_token_id:
|
||||
delta_token_ids = delta_token_ids[1:]
|
||||
|
||||
if self.end_token_id in previous_token_ids:
|
||||
return DeltaMessage(content=delta_text)
|
||||
|
||||
if (
|
||||
self._initial_in_reasoning
|
||||
and self.start_token_id not in previous_token_ids
|
||||
and self.start_token_id not in delta_token_ids
|
||||
):
|
||||
if self.end_token_id in delta_token_ids:
|
||||
reasoning, _, content = delta_text.partition(self.end_token)
|
||||
return DeltaMessage(
|
||||
reasoning=reasoning or None,
|
||||
content=content or None,
|
||||
)
|
||||
return DeltaMessage(reasoning=delta_text)
|
||||
|
||||
if (
|
||||
self.start_token_id not in previous_token_ids
|
||||
and self.start_token_id not in delta_token_ids
|
||||
):
|
||||
return DeltaMessage(content=delta_text)
|
||||
|
||||
if self.end_token_id in delta_token_ids:
|
||||
reasoning_text, _, content = delta_text.partition(self.end_token)
|
||||
if self.start_token_id in delta_token_ids:
|
||||
_, _, reasoning_text = reasoning_text.partition(self.start_token)
|
||||
return DeltaMessage(
|
||||
reasoning=reasoning_text or None,
|
||||
content=content or None,
|
||||
)
|
||||
|
||||
if self.start_token_id in delta_token_ids:
|
||||
_, _, reasoning = delta_text.partition(self.start_token)
|
||||
return DeltaMessage(reasoning=reasoning) if reasoning else None
|
||||
|
||||
return DeltaMessage(reasoning=delta_text)
|
||||
|
||||
def count_reasoning_tokens(self, token_ids: Sequence[int]) -> int:
|
||||
if not self._initial_in_reasoning:
|
||||
return super().count_reasoning_tokens(token_ids)
|
||||
|
||||
count = 0
|
||||
depth = 1
|
||||
for token_id in token_ids:
|
||||
if token_id == self.start_token_id:
|
||||
depth += 1
|
||||
continue
|
||||
if token_id == self.end_token_id:
|
||||
if depth > 0:
|
||||
depth -= 1
|
||||
continue
|
||||
if depth > 0:
|
||||
count += 1
|
||||
return count
|
||||
return self._parser_engine.is_reasoning_end_streaming(
|
||||
list(input_ids), tuple(delta_ids)
|
||||
)
|
||||
|
||||
@@ -15,7 +15,7 @@ Register a lazy module mapping.
|
||||
Example:
|
||||
ToolParserManager.register_lazy_module(
|
||||
name="kimi_k2",
|
||||
module_path="vllm.tool_parsers.kimi_k2_parser",
|
||||
module_path="vllm.tool_parsers.kimi_k2_tool_parser",
|
||||
class_name="KimiK2ToolParser",
|
||||
)
|
||||
"""
|
||||
|
||||
+18
-15
@@ -3,7 +3,6 @@
|
||||
import contextlib
|
||||
import importlib.metadata
|
||||
import os
|
||||
import platform
|
||||
import random
|
||||
import threading
|
||||
from collections.abc import Callable, Collection
|
||||
@@ -18,6 +17,7 @@ from torch.library import Library, infer_schema
|
||||
|
||||
import vllm.envs as envs
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from vllm.config import ModelConfig
|
||||
@@ -68,9 +68,7 @@ MODELOPT_TO_VLLM_KV_CACHE_DTYPE_MAP = {
|
||||
T = TypeVar("T")
|
||||
|
||||
|
||||
# Pin memory in non-WSL case.
|
||||
# Logic duplicated here for now to avoid circular import.
|
||||
PIN_MEMORY = "microsoft" not in " ".join(platform.uname()).lower()
|
||||
PIN_MEMORY = is_pin_memory_available()
|
||||
|
||||
|
||||
def is_quantized_kv_cache(kv_cache_dtype: str) -> bool:
|
||||
@@ -606,14 +604,24 @@ def create_kv_caches_with_random(
|
||||
|
||||
|
||||
def async_tensor_h2d(
|
||||
data: list,
|
||||
dtype: torch.dtype,
|
||||
data: list | np.ndarray | torch.Tensor,
|
||||
device: str | torch.device,
|
||||
pin_memory: bool = PIN_MEMORY,
|
||||
dtype: torch.dtype | None = None,
|
||||
) -> torch.Tensor:
|
||||
"""Asynchronously create a tensor and copy it from host to device."""
|
||||
t = torch.tensor(data, dtype=dtype, pin_memory=pin_memory, device="cpu")
|
||||
return t.to(device=device, non_blocking=True)
|
||||
"""Copy list/numpy array/tensor async from host to device."""
|
||||
if isinstance(data, np.ndarray):
|
||||
data = torch.from_numpy(data)
|
||||
if isinstance(data, torch.Tensor):
|
||||
t = data.pin_memory() if PIN_MEMORY else data
|
||||
else:
|
||||
t = torch.tensor(data, dtype=dtype, pin_memory=PIN_MEMORY, device="cpu")
|
||||
assert t.is_cpu
|
||||
return t.to(device=device, dtype=dtype, non_blocking=True)
|
||||
|
||||
|
||||
def np_to_pinned_tensor(array: np.ndarray) -> torch.Tensor:
|
||||
t = torch.from_numpy(array)
|
||||
return t.pin_memory() if PIN_MEMORY else t
|
||||
|
||||
|
||||
def make_ndarray_with_pad(
|
||||
@@ -914,11 +922,6 @@ def _encode_layer_name(layer_name: str) -> str | LayerName:
|
||||
return LayerName(layer_name) if _USE_LAYERNAME else layer_name
|
||||
|
||||
|
||||
# Supports xccl with PyTorch versions >= 2.8.0.dev for XPU platform
|
||||
def supports_xccl() -> bool:
|
||||
return torch.distributed.is_xccl_available()
|
||||
|
||||
|
||||
# Supports XPU Graph with PyTorch versions >= 2.11.0.dev for XPU platform
|
||||
def supports_xpu_graph() -> bool:
|
||||
return is_torch_equal_or_newer("2.11.0.dev")
|
||||
|
||||
@@ -41,8 +41,8 @@ from vllm.utils.flashinfer import (
|
||||
use_trtllm_attention,
|
||||
)
|
||||
from vllm.utils.math_utils import cdiv
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import (
|
||||
PIN_MEMORY,
|
||||
canonicalize_singleton_dim_strides,
|
||||
is_quantized_kv_cache,
|
||||
is_strictly_contiguous,
|
||||
@@ -708,9 +708,7 @@ class FlashInferMetadataBuilder(AttentionMetadataBuilder[FlashInferMetadata]):
|
||||
# Since we do not have explicit synchronization in ModelRunnerV2, we do not pin
|
||||
# reused CPU buffers to avoid a race condition between step N async copies to
|
||||
# GPU and step N+1 buffer updates.
|
||||
self.pin_memory = (
|
||||
not vllm_config.use_v2_model_runner and is_pin_memory_available()
|
||||
)
|
||||
self.pin_memory = not vllm_config.use_v2_model_runner and PIN_MEMORY
|
||||
self.paged_kv_indptr = self._make_buffer(max_num_reqs + 1)
|
||||
self.paged_kv_indptr_cpu_buffer = torch.zeros_like(
|
||||
self.paged_kv_indptr.cpu, pin_memory=self.pin_memory
|
||||
|
||||
@@ -28,7 +28,11 @@ from vllm.logger import init_logger
|
||||
from vllm.model_executor.layers.attention import Attention
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.utils.math_utils import cdiv
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache, is_torch_equal_or_newer
|
||||
from vllm.utils.torch_utils import (
|
||||
async_tensor_h2d,
|
||||
is_quantized_kv_cache,
|
||||
is_torch_equal_or_newer,
|
||||
)
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
AttentionCGSupport,
|
||||
@@ -58,7 +62,7 @@ def _offsets_to_doc_ids_tensor(
|
||||
doc_ids = torch.repeat_interleave(
|
||||
torch.arange(len(counts), dtype=torch.int32), counts
|
||||
)
|
||||
return doc_ids.to(device, non_blocking=True)
|
||||
return async_tensor_h2d(doc_ids, device=device)
|
||||
|
||||
|
||||
def pad_to_multiple(x: torch.Tensor, multiple: int, dim: int):
|
||||
|
||||
@@ -8,6 +8,7 @@ from typing import Literal
|
||||
import torch
|
||||
|
||||
from vllm.config import VllmConfig
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
AttentionCGSupport,
|
||||
@@ -203,8 +204,8 @@ class GDNAttentionMetadataBuilder(AttentionMetadataBuilder[GDNAttentionMetadata]
|
||||
spec_sequence_masks = None
|
||||
spec_sequence_masks_cpu = None
|
||||
else:
|
||||
spec_sequence_masks = spec_sequence_masks_cpu.to(
|
||||
query_start_loc.device, non_blocking=True
|
||||
spec_sequence_masks = async_tensor_h2d(
|
||||
spec_sequence_masks_cpu, device=query_start_loc.device
|
||||
)
|
||||
|
||||
if spec_sequence_masks is None:
|
||||
@@ -376,12 +377,14 @@ class GDNAttentionMetadataBuilder(AttentionMetadataBuilder[GDNAttentionMetadata]
|
||||
)
|
||||
|
||||
assert prefill_query_start_loc_cpu is not None
|
||||
chunk_indices = prepare_chunk_indices(
|
||||
prefill_query_start_loc_cpu, FLA_CHUNK_SIZE
|
||||
).to(device=gpu_device, non_blocking=True)
|
||||
chunk_offsets = prepare_chunk_offsets(
|
||||
prefill_query_start_loc_cpu, FLA_CHUNK_SIZE
|
||||
).to(device=gpu_device, non_blocking=True)
|
||||
chunk_indices = async_tensor_h2d(
|
||||
prepare_chunk_indices(prefill_query_start_loc_cpu, FLA_CHUNK_SIZE),
|
||||
device=gpu_device,
|
||||
)
|
||||
chunk_offsets = async_tensor_h2d(
|
||||
prepare_chunk_offsets(prefill_query_start_loc_cpu, FLA_CHUNK_SIZE),
|
||||
device=gpu_device,
|
||||
)
|
||||
|
||||
if num_prefills > 0:
|
||||
has_initial_state = context_lens_tensor > 0
|
||||
|
||||
@@ -7,6 +7,7 @@ from typing import Any
|
||||
import torch
|
||||
|
||||
from vllm.config import VllmConfig
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
CommonAttentionMetadata,
|
||||
@@ -68,22 +69,22 @@ def compute_varlen_chunk_metadata(
|
||||
|
||||
# Exclusive prefix sum over logical-chunk lengths
|
||||
if chunk_lens:
|
||||
cu_chunk_seqlens = torch.tensor(
|
||||
[0] + list(itertools.accumulate(chunk_lens)),
|
||||
device=device,
|
||||
dtype=torch.int32,
|
||||
)
|
||||
# Final boundary must equal total tokens
|
||||
assert int(cu_chunk_seqlens[-1].item()) == total
|
||||
cu_chunk_seqlens_list = [0] + list(itertools.accumulate(chunk_lens))
|
||||
# Final boundary must equal total tokens (check on host to avoid a sync)
|
||||
assert cu_chunk_seqlens_list[-1] == total
|
||||
else:
|
||||
cu_chunk_seqlens = torch.tensor([0], device=device, dtype=torch.int32)
|
||||
cu_chunk_seqlens_list = [0]
|
||||
cu_chunk_seqlens = async_tensor_h2d(
|
||||
cu_chunk_seqlens_list, dtype=torch.int32, device=device
|
||||
)
|
||||
|
||||
last_chunk_indices_t = (
|
||||
torch.tensor(last_chunk_indices, device=device, dtype=torch.int32)
|
||||
if len(starts) > 0
|
||||
else torch.empty((0,), device=device, dtype=torch.int32)
|
||||
# last_chunk_indices is empty when there are no sequences (len(starts) == 0).
|
||||
last_chunk_indices_t = async_tensor_h2d(
|
||||
last_chunk_indices, dtype=torch.int32, device=device
|
||||
)
|
||||
seq_idx_chunks_t = async_tensor_h2d(
|
||||
seq_idx_chunks, dtype=torch.int32, device=device
|
||||
)
|
||||
seq_idx_chunks_t = torch.tensor(seq_idx_chunks, device=device, dtype=torch.int32)
|
||||
return cu_chunk_seqlens, last_chunk_indices_t, seq_idx_chunks_t
|
||||
|
||||
|
||||
|
||||
@@ -26,7 +26,7 @@ from vllm.model_executor.layers.attention.mla_attention import (
|
||||
get_mla_dims,
|
||||
)
|
||||
from vllm.platforms.interface import DeviceCapability
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache, np_to_pinned_tensor
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
AttentionCGSupport,
|
||||
@@ -217,7 +217,7 @@ class FlashInferMLASparseMetadataBuilder(
|
||||
# Zero-fill for cudagraphs
|
||||
self.req_id_per_token_buffer.fill_(0)
|
||||
self.req_id_per_token_buffer[: req_id_per_token.shape[0]].copy_(
|
||||
torch.from_numpy(req_id_per_token), non_blocking=True
|
||||
np_to_pinned_tensor(req_id_per_token), non_blocking=True
|
||||
)
|
||||
req_id_per_token_tensor = self.req_id_per_token_buffer[:num_tokens]
|
||||
|
||||
|
||||
@@ -16,7 +16,7 @@ from vllm.model_executor.layers.attention.mla_attention import (
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.platforms.interface import DeviceCapability
|
||||
from vllm.utils.platform_utils import num_compute_units
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache
|
||||
from vllm.utils.torch_utils import is_quantized_kv_cache, np_to_pinned_tensor
|
||||
from vllm.v1.attention.backend import (
|
||||
AttentionBackend,
|
||||
AttentionCGSupport,
|
||||
@@ -503,7 +503,7 @@ class FlashMLASparseMetadataBuilder(AttentionMetadataBuilder[FlashMLASparseMetad
|
||||
# Zero-fill for cudagraphs
|
||||
self.req_id_per_token_buffer.fill_(0)
|
||||
self.req_id_per_token_buffer[: req_id_per_token.shape[0]].copy_(
|
||||
torch.from_numpy(req_id_per_token), non_blocking=True
|
||||
np_to_pinned_tensor(req_id_per_token), non_blocking=True
|
||||
)
|
||||
req_id_per_token = self.req_id_per_token_buffer[:num_tokens]
|
||||
|
||||
|
||||
@@ -17,7 +17,7 @@ from typing_extensions import runtime_checkable
|
||||
|
||||
from vllm.config import VllmConfig, get_layers_from_vllm_config
|
||||
from vllm.utils.math_utils import cdiv
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.utils.torch_utils import PIN_MEMORY, async_tensor_h2d, np_to_pinned_tensor
|
||||
from vllm.v1.kv_cache_interface import KVCacheSpec, MambaSpec
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -364,8 +364,8 @@ def make_local_attention_virtual_batches(
|
||||
# tensor first, which recovers perf.
|
||||
# Upload the index tensors to the block_table's device up-front so that the
|
||||
# fancy indexing below doesn't implicitly force a synchronous H2D copy.
|
||||
batch_indices_torch = torch.from_numpy(batch_indices).to(device, non_blocking=True)
|
||||
block_indices_torch = torch.from_numpy(block_indices).to(device, non_blocking=True)
|
||||
batch_indices_torch = async_tensor_h2d(batch_indices, device=device)
|
||||
block_indices_torch = async_tensor_h2d(block_indices, device=device)
|
||||
|
||||
# Save as a lambda so we can return this for update_block_table
|
||||
make_block_table = lambda block_table: block_table[
|
||||
@@ -379,8 +379,8 @@ def make_local_attention_virtual_batches(
|
||||
|
||||
return CommonAttentionMetadata(
|
||||
query_start_loc_cpu=query_start_loc_cpu,
|
||||
query_start_loc=query_start_loc_cpu.to(device=device, non_blocking=True),
|
||||
seq_lens=seq_lens_cpu.to(device=device, non_blocking=True),
|
||||
query_start_loc=async_tensor_h2d(query_start_loc_cpu, device=device),
|
||||
seq_lens=async_tensor_h2d(seq_lens_cpu, device=device),
|
||||
num_reqs=len(seq_lens_cpu),
|
||||
num_actual_tokens=common_attn_metadata.num_actual_tokens,
|
||||
max_query_len=seqlens_q_local.max(),
|
||||
@@ -808,14 +808,12 @@ def create_fast_prefill_custom_backend(
|
||||
|
||||
|
||||
def compute_causal_conv1d_metadata(
|
||||
query_start_loc_p_cpu: torch.Tensor,
|
||||
*,
|
||||
device: torch.device,
|
||||
):
|
||||
query_start_loc_p_cpu: torch.Tensor, *, device: torch.device
|
||||
) -> tuple[dict[int, dict[str, Any]], torch.Tensor, torch.Tensor]:
|
||||
# Needed for causal_conv1d. Use the CPU query_start_loc to avoid DtoH sync.
|
||||
assert query_start_loc_p_cpu.device.type == "cpu"
|
||||
seqlens = query_start_loc_p_cpu.diff()
|
||||
nums_dict = {} # type: ignore
|
||||
nums_dict: dict[int, dict[str, Any]] = {}
|
||||
batch_ptr = None
|
||||
token_chunk_offset_ptr = None
|
||||
for BLOCK_M in [8]: # cover all BLOCK_M values
|
||||
@@ -823,7 +821,7 @@ def compute_causal_conv1d_metadata(
|
||||
nums_dict[BLOCK_M] = {}
|
||||
nums_dict[BLOCK_M]["nums"] = nums
|
||||
nums_dict[BLOCK_M]["tot"] = nums.sum().item()
|
||||
mlist = torch.from_numpy(np.repeat(np.arange(len(nums)), nums))
|
||||
mlist = np_to_pinned_tensor(np.repeat(np.arange(len(nums)), nums))
|
||||
nums_dict[BLOCK_M]["mlist"] = mlist
|
||||
mlist_len = len(nums_dict[BLOCK_M]["mlist"])
|
||||
nums_dict[BLOCK_M]["mlist_len"] = mlist_len
|
||||
@@ -831,7 +829,7 @@ def compute_causal_conv1d_metadata(
|
||||
offsetlist = [] # type: ignore
|
||||
for idx, num in enumerate(nums):
|
||||
offsetlist.extend(range(num))
|
||||
offsetlist = torch.tensor(offsetlist, dtype=torch.int32)
|
||||
offsetlist = torch.tensor(offsetlist, dtype=torch.int32, pin_memory=PIN_MEMORY)
|
||||
nums_dict[BLOCK_M]["offsetlist"] = offsetlist
|
||||
|
||||
if batch_ptr is None:
|
||||
@@ -845,16 +843,15 @@ def compute_causal_conv1d_metadata(
|
||||
else:
|
||||
if batch_ptr.nelement() < MAX_NUM_PROGRAMS:
|
||||
batch_ptr.resize_(MAX_NUM_PROGRAMS).fill_(PAD_SLOT_ID)
|
||||
token_chunk_offset_ptr.resize_( # type: ignore
|
||||
MAX_NUM_PROGRAMS
|
||||
).fill_(PAD_SLOT_ID)
|
||||
assert token_chunk_offset_ptr is not None
|
||||
token_chunk_offset_ptr.resize_(MAX_NUM_PROGRAMS).fill_(PAD_SLOT_ID)
|
||||
|
||||
assert batch_ptr is not None
|
||||
batch_ptr[0:mlist_len].copy_(mlist, non_blocking=True)
|
||||
token_chunk_offset_ptr[ # type: ignore
|
||||
0:mlist_len
|
||||
].copy_(offsetlist, non_blocking=True)
|
||||
assert token_chunk_offset_ptr is not None
|
||||
token_chunk_offset_ptr[0:mlist_len].copy_(offsetlist, non_blocking=True)
|
||||
nums_dict[BLOCK_M]["batch_ptr"] = batch_ptr
|
||||
nums_dict[BLOCK_M]["token_chunk_offset_ptr"] = token_chunk_offset_ptr # type: ignore
|
||||
nums_dict[BLOCK_M]["token_chunk_offset_ptr"] = token_chunk_offset_ptr
|
||||
|
||||
return nums_dict, batch_ptr, token_chunk_offset_ptr
|
||||
|
||||
|
||||
@@ -58,7 +58,9 @@ def _indexer_k_quant_and_cache_kernel(
|
||||
slot_id = tl.load(slot_mapping_ptr + tid)
|
||||
if slot_id < 0:
|
||||
return
|
||||
block_id = slot_id // block_size
|
||||
# The packed KV layout makes per-block strides large
|
||||
# enough that block_id * stride can exceed 32-bit range.
|
||||
block_id = (slot_id // block_size).to(tl.int64)
|
||||
block_offset = slot_id % block_size
|
||||
tile_block_id = block_offset // BLOCK_TILE_SIZE
|
||||
tile_block_offset = block_offset % BLOCK_TILE_SIZE
|
||||
@@ -179,7 +181,9 @@ def _cp_gather_indexer_quant_cache_kernel(
|
||||
block_table_ptr + block_table_offset, mask=valid_block_table, other=-1
|
||||
)
|
||||
valid_block = valid_block_table & (block_id >= 0) & (block_id < NUM_BLOCKS)
|
||||
safe_block_id = tl.where(valid_block, block_id, 0)
|
||||
# The packed KV layout makes per-block strides large
|
||||
# enough that block_id * stride can exceed 32-bit range.
|
||||
safe_block_id = tl.where(valid_block, block_id, 0).to(tl.int64)
|
||||
safe_block_offset = tl.where(valid_block, block_offset, 0)
|
||||
tiled_block_offset = safe_block_offset % BLOCK_TILE_SIZE
|
||||
if LAYOUT == "SHUFFLE":
|
||||
|
||||
@@ -938,9 +938,7 @@ def _pool_bytes_per_block(kv_cache_groups: list[KVCacheGroupSpec]) -> int:
|
||||
kv_cache_groups[0].kv_cache_spec, UniformTypeKVCacheSpecs
|
||||
):
|
||||
return kv_cache_groups[0].kv_cache_spec.page_size_bytes
|
||||
if all(
|
||||
isinstance(g.kv_cache_spec, UniformTypeKVCacheSpecs) for g in kv_cache_groups
|
||||
):
|
||||
if _use_packed_kv_cache_groups(kv_cache_groups):
|
||||
# buckets = {page_size: [[layer_names], [layer_names], ...]}
|
||||
buckets = _bucket_layers_by_page_size(kv_cache_groups)
|
||||
return sum(ps * len(slots) for ps, slots in buckets.items())
|
||||
@@ -1218,16 +1216,29 @@ def _bucket_layers_by_page_size(
|
||||
return buckets
|
||||
|
||||
|
||||
def _get_kv_cache_config_deepseek_v4(
|
||||
def _use_packed_kv_cache_groups(
|
||||
kv_cache_groups: list[KVCacheGroupSpec],
|
||||
) -> bool:
|
||||
is_dsv4 = all(
|
||||
isinstance(group.kv_cache_spec, UniformTypeKVCacheSpecs)
|
||||
for group in kv_cache_groups
|
||||
)
|
||||
return is_dsv4 or (
|
||||
bool(envs.VLLM_USE_PACKED_HMA_KV_CACHE) and len(kv_cache_groups) > 1
|
||||
)
|
||||
|
||||
|
||||
def _get_kv_cache_config_packed(
|
||||
vllm_config: VllmConfig,
|
||||
kv_cache_groups: list[KVCacheGroupSpec],
|
||||
available_memory: int,
|
||||
) -> tuple[int, list[KVCacheTensor]]:
|
||||
"""DeepseekV4 KV cache tensor layout planning.
|
||||
"""Plan a packed per-block KV cache tensor layout.
|
||||
|
||||
Emit one KVCacheTensor per (slot_idx, page_size). Layers from different
|
||||
groups at the same slot share a tensor (they have independent block
|
||||
tables so block-id namespaces never collide).
|
||||
tables so block-id namespaces never collide). Each emitted tensor aliases
|
||||
one physical backing allocation, with per-block data laid out contiguously.
|
||||
"""
|
||||
# buckets = {page_size: [[layer_names], [layer_names], ...]}
|
||||
buckets = _bucket_layers_by_page_size(kv_cache_groups)
|
||||
@@ -1255,6 +1266,9 @@ def _get_kv_cache_config_deepseek_v4(
|
||||
return num_blocks, kv_cache_tensors
|
||||
|
||||
|
||||
_get_kv_cache_config_deepseek_v4 = _get_kv_cache_config_packed
|
||||
|
||||
|
||||
def get_kv_cache_config_from_groups(
|
||||
vllm_config: VllmConfig,
|
||||
kv_cache_groups: list[KVCacheGroupSpec],
|
||||
@@ -1299,13 +1313,11 @@ def get_kv_cache_config_from_groups(
|
||||
)
|
||||
for layer_name in kv_cache_groups[0].layer_names
|
||||
]
|
||||
elif all(
|
||||
isinstance(group.kv_cache_spec, UniformTypeKVCacheSpecs)
|
||||
for group in kv_cache_groups
|
||||
):
|
||||
# DeepseekV4: UniformTypeKVCacheSpecs but multiple groups.
|
||||
# Delegate to the DeepseekV4-specific allocator.
|
||||
num_blocks, kv_cache_tensors = _get_kv_cache_config_deepseek_v4(
|
||||
elif _use_packed_kv_cache_groups(kv_cache_groups):
|
||||
# DeepSeek V4 keeps the existing packed layout. Other multi-group
|
||||
# attention-only HMA layouts can opt in with
|
||||
# VLLM_USE_PACKED_HMA_KV_CACHE=1.
|
||||
num_blocks, kv_cache_tensors = _get_kv_cache_config_packed(
|
||||
vllm_config, kv_cache_groups, available_memory
|
||||
)
|
||||
else:
|
||||
|
||||
@@ -118,8 +118,8 @@ class CachedRequestData:
|
||||
# NOTE(woosuk): new_token_ids is only used for pipeline parallelism.
|
||||
# When PP is not used, new_token_ids will be empty.
|
||||
new_token_ids: list[list[int]]
|
||||
# For requests not scheduled in the last step, propagate the token ids to the
|
||||
# connector. Won't contain requests that were scheduled in the prior step.
|
||||
# MRV1-only: For requests not scheduled in the last step, propagate the token ids
|
||||
# to the connector. Won't contain requests scheduled in the prior step.
|
||||
all_token_ids: dict[str, list[int]]
|
||||
new_block_ids: list[tuple[list[int], ...] | None]
|
||||
num_computed_tokens: list[int]
|
||||
|
||||
@@ -101,6 +101,7 @@ class Scheduler(SchedulerInterface):
|
||||
self.finished_req_ids_dict: dict[int, set[str]] | None = (
|
||||
defaultdict(set) if include_finished_set else None
|
||||
)
|
||||
# Track requests scheduled in prior step (MRV1-only).
|
||||
self.prev_step_scheduled_req_ids: set[str] = set()
|
||||
|
||||
# Scheduling constraints.
|
||||
@@ -1010,8 +1011,8 @@ class Scheduler(SchedulerInterface):
|
||||
|
||||
# Construct the scheduler output.
|
||||
if self.use_v2_model_runner:
|
||||
scheduled_new_reqs = scheduled_new_reqs + scheduled_resumed_reqs
|
||||
scheduled_resumed_reqs = []
|
||||
scheduled_new_reqs.extend(scheduled_resumed_reqs)
|
||||
scheduled_resumed_reqs.clear()
|
||||
new_reqs_data = [
|
||||
NewRequestData.from_request(
|
||||
req,
|
||||
@@ -1037,9 +1038,10 @@ class Scheduler(SchedulerInterface):
|
||||
req_to_new_blocks,
|
||||
)
|
||||
|
||||
# Record the request ids that were scheduled in this step.
|
||||
self.prev_step_scheduled_req_ids.clear()
|
||||
self.prev_step_scheduled_req_ids.update(num_scheduled_tokens.keys())
|
||||
# Record the request ids that were scheduled in this step (MRV1-only).
|
||||
if not self.use_v2_model_runner:
|
||||
self.prev_step_scheduled_req_ids.clear()
|
||||
self.prev_step_scheduled_req_ids.update(num_scheduled_tokens.keys())
|
||||
|
||||
new_block_ids_to_zero = (
|
||||
(self.kv_cache_manager.take_new_block_ids() or None)
|
||||
@@ -1252,12 +1254,11 @@ class Scheduler(SchedulerInterface):
|
||||
req.num_computed_tokens : req.num_computed_tokens + num_tokens
|
||||
]
|
||||
new_token_ids.append(token_ids)
|
||||
scheduled_in_prev_step = req_id in self.prev_step_scheduled_req_ids
|
||||
if idx >= num_running_reqs:
|
||||
assert not scheduled_in_prev_step
|
||||
resumed_req_ids.add(req_id)
|
||||
if not scheduled_in_prev_step:
|
||||
all_token_ids[req_id] = req.all_token_ids.copy()
|
||||
if not self.use_v2_model_runner: # noqa: SIM102
|
||||
if req_id not in self.prev_step_scheduled_req_ids:
|
||||
all_token_ids[req_id] = req.all_token_ids.copy()
|
||||
new_block_ids.append(
|
||||
req_to_new_blocks[req_id].get_block_ids(allow_none=True)
|
||||
)
|
||||
|
||||
@@ -4,7 +4,10 @@ from typing_extensions import override
|
||||
|
||||
from vllm.v1.kv_offload.base import BlockIDsLoadStoreSpec
|
||||
|
||||
METRIC_STORES_SKIPPED = "vllm:kv_offload_stores_skipped"
|
||||
|
||||
class CPUOffloadingMetrics:
|
||||
STORES_SKIPPED = "vllm:kv_offload_stores_skipped"
|
||||
CPU_CACHE_USAGE_PERC = "vllm:kv_offload_cpu_cache_usage_perc"
|
||||
|
||||
|
||||
class CPULoadStoreSpec(BlockIDsLoadStoreSpec):
|
||||
|
||||
@@ -14,7 +14,7 @@ from vllm.logger import init_logger
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.triton_utils import HAS_TRITON, triton
|
||||
from vllm.utils.math_utils import cdiv
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
from vllm.v1.kv_offload.base import (
|
||||
BlockIDsLoadStoreSpec,
|
||||
CanonicalKVCacheRef,
|
||||
@@ -156,7 +156,7 @@ def pin_mmap_region(region: SharedOffloadRegion) -> None:
|
||||
def _new_descriptor_buffers(
|
||||
num_copy_ops: int,
|
||||
) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
pin = is_pin_memory_available()
|
||||
pin = PIN_MEMORY
|
||||
# CUDA cache_kernels.cu requires int64; XPU DMA engine requires uint64.
|
||||
ptr_dtype = torch.uint64 if current_platform.is_xpu() else torch.int64
|
||||
return (
|
||||
@@ -482,7 +482,7 @@ class CpuGpuOffloadingHandlers:
|
||||
num_cpu_blocks: int,
|
||||
mmap_region: SharedOffloadRegion | None = None,
|
||||
):
|
||||
pin_memory = is_pin_memory_available()
|
||||
pin_memory = PIN_MEMORY
|
||||
logger.info("Allocating %d CPU tensors...", len(kv_caches.tensors))
|
||||
self._mmap_region = mmap_region
|
||||
if mmap_region is not None and pin_memory:
|
||||
|
||||
@@ -18,7 +18,10 @@ from vllm.v1.kv_offload.base import (
|
||||
ReqContext,
|
||||
RequestOffloadingContext,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.common import METRIC_STORES_SKIPPED, CPULoadStoreSpec
|
||||
from vllm.v1.kv_offload.cpu.common import (
|
||||
CPULoadStoreSpec,
|
||||
CPUOffloadingMetrics,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.policies.arc import ARCCachePolicy
|
||||
from vllm.v1.kv_offload.cpu.policies.base import BlockStatus, CachePolicy
|
||||
from vllm.v1.kv_offload.cpu.policies.lru import LRUCachePolicy
|
||||
@@ -282,13 +285,21 @@ class CPUOffloadingManager(OffloadingManager):
|
||||
self.events.clear()
|
||||
|
||||
def get_stats(self) -> OffloadingConnectorStats | None:
|
||||
if self.store_threshold < 2:
|
||||
return None
|
||||
|
||||
stats = OffloadingConnectorStats()
|
||||
stats.increase_counter(
|
||||
METRIC_STORES_SKIPPED,
|
||||
self.stores_skipped_in_current_batch,
|
||||
|
||||
# Compute cache usage.
|
||||
num_used = (
|
||||
self._num_allocated_blocks
|
||||
- len(self._free_list)
|
||||
- self._num_evictable_cache_blocks
|
||||
)
|
||||
self.stores_skipped_in_current_batch = 0
|
||||
usage = num_used / self._num_blocks if self._num_blocks > 0 else 0.0
|
||||
stats.set_gauge(CPUOffloadingMetrics.CPU_CACHE_USAGE_PERC, usage)
|
||||
|
||||
if self.store_threshold >= 2:
|
||||
stats.increase_counter(
|
||||
CPUOffloadingMetrics.STORES_SKIPPED,
|
||||
self.stores_skipped_in_current_batch,
|
||||
)
|
||||
self.stores_skipped_in_current_batch = 0
|
||||
return stats
|
||||
|
||||
@@ -14,11 +14,15 @@ from vllm.v1.kv_offload.base import (
|
||||
GPULoadStoreSpec,
|
||||
LoadStoreSpec,
|
||||
OffloadingCounterMetadata,
|
||||
OffloadingGaugeMetadata,
|
||||
OffloadingManager,
|
||||
OffloadingMetricMetadata,
|
||||
OffloadingSpec,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.common import METRIC_STORES_SKIPPED, CPULoadStoreSpec
|
||||
from vllm.v1.kv_offload.cpu.common import (
|
||||
CPULoadStoreSpec,
|
||||
CPUOffloadingMetrics,
|
||||
)
|
||||
from vllm.v1.kv_offload.cpu.gpu_worker import CpuGpuOffloadingHandlers
|
||||
from vllm.v1.kv_offload.cpu.manager import CPUOffloadingManager
|
||||
from vllm.v1.kv_offload.worker.worker import OffloadingHandler
|
||||
@@ -31,17 +35,27 @@ class CPUOffloadingSpec(OffloadingSpec):
|
||||
def build_metric_definitions(
|
||||
cls, extra_config: dict[str, Any]
|
||||
) -> dict[str, OffloadingMetricMetadata]:
|
||||
store_threshold = int(extra_config.get("store_threshold", 0))
|
||||
if store_threshold < 2:
|
||||
return {}
|
||||
return {
|
||||
METRIC_STORES_SKIPPED: OffloadingCounterMetadata(
|
||||
definitions: dict[str, OffloadingMetricMetadata] = {
|
||||
CPUOffloadingMetrics.CPU_CACHE_USAGE_PERC: OffloadingGaugeMetadata(
|
||||
documentation=(
|
||||
"Number of KV offload stores skipped because the reuse "
|
||||
"threshold was not reached."
|
||||
"Fraction of CPU KV-cache space currently pinned by active "
|
||||
"transfers (0.0 = idle, 1.0 = saturated). Sustained high "
|
||||
"values indicate transfers (stores or promotions) may be "
|
||||
"dropped due to insufficient capacity."
|
||||
),
|
||||
)
|
||||
}
|
||||
store_threshold = int(extra_config.get("store_threshold", 0))
|
||||
if store_threshold >= 2:
|
||||
definitions[CPUOffloadingMetrics.STORES_SKIPPED] = (
|
||||
OffloadingCounterMetadata(
|
||||
documentation=(
|
||||
"Number of KV offload stores skipped because the reuse "
|
||||
"threshold was not reached."
|
||||
),
|
||||
)
|
||||
)
|
||||
return definitions
|
||||
|
||||
def __init__(self, vllm_config: VllmConfig, kv_cache_config: KVCacheConfig):
|
||||
super().__init__(vllm_config, kv_cache_config)
|
||||
@@ -58,7 +72,15 @@ class CPUOffloadingSpec(OffloadingSpec):
|
||||
self.cpu_page_size_per_worker = 0
|
||||
assert kv_cache_config is not None
|
||||
if kv_cache_config.num_blocks > 0 and world_size > 0:
|
||||
total_gpu_kv_bytes = sum(t.size for t in kv_cache_config.kv_cache_tensors)
|
||||
is_packed = any(t.block_stride for t in kv_cache_config.kv_cache_tensors)
|
||||
assert not is_packed or all(
|
||||
t.block_stride for t in kv_cache_config.kv_cache_tensors
|
||||
)
|
||||
total_gpu_kv_bytes = (
|
||||
kv_cache_config.kv_cache_tensors[0].size
|
||||
if is_packed
|
||||
else sum(t.size for t in kv_cache_config.kv_cache_tensors)
|
||||
)
|
||||
kv_bytes_per_block = (
|
||||
total_gpu_kv_bytes // kv_cache_config.num_blocks
|
||||
) * world_size
|
||||
|
||||
@@ -7,9 +7,7 @@ import torch
|
||||
|
||||
from vllm.pooling_params import PoolingParams
|
||||
from vllm.tasks import PoolingTask
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
|
||||
pin_memory = is_pin_memory_available()
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -134,7 +132,7 @@ class PoolingMetadata:
|
||||
num_scheduled_tokens_cpu = torch.from_numpy(num_scheduled_tokens_np)
|
||||
if query_start_loc_gpu is None:
|
||||
cumsum = torch.zeros(
|
||||
n_seq + 1, dtype=torch.int64, pin_memory=pin_memory, device="cpu"
|
||||
n_seq + 1, dtype=torch.int64, pin_memory=PIN_MEMORY, device="cpu"
|
||||
)
|
||||
torch.cumsum(num_scheduled_tokens_cpu, dim=0, out=cumsum[1:])
|
||||
cumsum = cumsum.to(device, non_blocking=True)
|
||||
|
||||
@@ -7,6 +7,7 @@ import numpy as np
|
||||
import torch
|
||||
|
||||
from vllm import SamplingParams
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.v1.sample.logits_processor.interface import (
|
||||
BatchUpdate,
|
||||
LogitsProcessor,
|
||||
@@ -118,7 +119,6 @@ class MinPLogitsProcessor(LogitsProcessor):
|
||||
class LogitBiasLogitsProcessor(LogitsProcessor):
|
||||
def __init__(self, _, device: torch.device, is_pin_memory: bool):
|
||||
self.device = device
|
||||
self.pin_memory = is_pin_memory
|
||||
self.biases: dict[int, dict[int, float]] = {}
|
||||
|
||||
self.bias_tensor: torch.Tensor = torch.tensor(())
|
||||
@@ -154,9 +154,7 @@ class LogitBiasLogitsProcessor(LogitsProcessor):
|
||||
)
|
||||
|
||||
def _device_tensor(self, data: list, dtype: torch.dtype) -> torch.Tensor:
|
||||
return torch.tensor(
|
||||
data, device="cpu", dtype=dtype, pin_memory=self.pin_memory
|
||||
).to(device=self.device, non_blocking=True)
|
||||
return async_tensor_h2d(data, device=self.device, dtype=dtype)
|
||||
|
||||
def apply(self, logits: torch.Tensor) -> torch.Tensor:
|
||||
if self.biases:
|
||||
@@ -170,7 +168,6 @@ class MinTokensLogitsProcessor(LogitsProcessor):
|
||||
):
|
||||
# index -> (min_toks, output_token_ids, stop_token_ids)
|
||||
self.device = device
|
||||
self.pin_memory = is_pin_memory
|
||||
self.min_toks: dict[int, tuple[int, Sequence[int], set[int]]] = {}
|
||||
|
||||
# (req_idx_tensor,eos_tok_id_tensor)
|
||||
@@ -227,9 +224,7 @@ class MinTokensLogitsProcessor(LogitsProcessor):
|
||||
)
|
||||
|
||||
def _device_tensor(self, data: list, dtype: torch.dtype) -> torch.Tensor:
|
||||
return torch.tensor(
|
||||
data, device="cpu", dtype=dtype, pin_memory=self.pin_memory
|
||||
).to(device=self.device, non_blocking=True)
|
||||
return async_tensor_h2d(data, device=self.device, dtype=dtype)
|
||||
|
||||
def apply(self, logits: torch.Tensor) -> torch.Tensor:
|
||||
if self.min_toks:
|
||||
@@ -283,8 +278,8 @@ class MinTokensLogitsProcessor(LogitsProcessor):
|
||||
toks_arr = np.concatenate(all_toks)
|
||||
# (row_indices, token_indices) for index_put_ to set -inf.
|
||||
logits_slice = (
|
||||
torch.from_numpy(rows_arr).to(self.device, non_blocking=True),
|
||||
torch.from_numpy(toks_arr).to(self.device, non_blocking=True),
|
||||
async_tensor_h2d(rows_arr, device=self.device),
|
||||
async_tensor_h2d(toks_arr, device=self.device),
|
||||
)
|
||||
logits.index_put_(logits_slice, self.neg_inf_tensor)
|
||||
|
||||
|
||||
@@ -4,8 +4,7 @@
|
||||
import torch
|
||||
|
||||
from vllm.model_executor.layers.utils import apply_penalties
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import make_tensor_with_pad
|
||||
from vllm.utils.torch_utils import PIN_MEMORY, make_tensor_with_pad
|
||||
|
||||
|
||||
def apply_all_penalties(
|
||||
@@ -52,6 +51,6 @@ def _convert_to_tensors(
|
||||
pad=vocab_size,
|
||||
device="cpu",
|
||||
dtype=torch.int64,
|
||||
pin_memory=is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
return output_tokens_tensor.to(device, non_blocking=True)
|
||||
|
||||
@@ -6,7 +6,7 @@ import torch
|
||||
import torch.nn as nn
|
||||
|
||||
from vllm.config.model import LogprobsMode
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
from vllm.v1.outputs import LogprobsTensors, SamplerOutput
|
||||
from vllm.v1.sample.metadata import SamplingMetadata
|
||||
from vllm.v1.sample.ops.bad_words import apply_bad_words
|
||||
@@ -65,7 +65,7 @@ class Sampler(nn.Module):
|
||||
):
|
||||
super().__init__()
|
||||
self.topk_topp_sampler = TopKTopPSampler(logprobs_mode, use_fp64_gumbel)
|
||||
self.pin_memory = is_pin_memory_available()
|
||||
self.pin_memory = PIN_MEMORY
|
||||
self.logprobs_mode = logprobs_mode
|
||||
self.use_fp64_gumbel = use_fp64_gumbel
|
||||
|
||||
|
||||
@@ -7,7 +7,7 @@ from typing import TYPE_CHECKING, Any
|
||||
import torch
|
||||
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.utils.torch_utils import async_tensor_h2d
|
||||
from vllm.utils.torch_utils import PIN_MEMORY, async_tensor_h2d
|
||||
from vllm.v1.sample.logits_processor.interface import (
|
||||
BatchUpdate,
|
||||
MoveDirectionality,
|
||||
@@ -22,12 +22,11 @@ def maybe_create_thinking_budget_state_holder(
|
||||
max_num_seqs: int,
|
||||
num_spec_tokens: int,
|
||||
device: torch.device,
|
||||
is_pin_memory: bool,
|
||||
) -> "ThinkingBudgetStateHolder | None":
|
||||
if reasoning_config is None:
|
||||
return None
|
||||
return ThinkingBudgetStateHolder(
|
||||
reasoning_config, max_num_seqs, num_spec_tokens, device, is_pin_memory
|
||||
reasoning_config, max_num_seqs, num_spec_tokens, device, PIN_MEMORY
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -33,7 +33,7 @@ from vllm.multimodal.inputs import (
|
||||
MultiModalSharedField,
|
||||
NestedTensors,
|
||||
)
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
from vllm.v1.utils import tensor_data
|
||||
|
||||
logger = init_logger(__name__)
|
||||
@@ -327,7 +327,7 @@ class MsgpackDecoder:
|
||||
oob_tensor_provider: OOBTensorProvider | None = None,
|
||||
):
|
||||
self.share_mem = share_mem
|
||||
self.pin_tensors = is_pin_memory_available()
|
||||
self.pin_tensors = PIN_MEMORY
|
||||
args = () if t is None else (t,)
|
||||
self.decoder = msgpack.Decoder(
|
||||
*args, ext_hook=self.ext_hook, dec_hook=self.dec_hook
|
||||
|
||||
@@ -82,6 +82,9 @@ class SimpleCPUOffloadScheduler:
|
||||
vllm_config.kv_events_config is not None
|
||||
and vllm_config.kv_events_config.enable_kv_cache_events
|
||||
)
|
||||
dcp_world_size = vllm_config.parallel_config.decode_context_parallel_size
|
||||
pcp_world_size = vllm_config.parallel_config.prefill_context_parallel_size
|
||||
self.cp_world_size = dcp_world_size * pcp_world_size
|
||||
self.block_size = scheduler_block_size
|
||||
self.hash_block_size = hash_block_size
|
||||
assert self.block_size % self.hash_block_size == 0
|
||||
@@ -113,9 +116,6 @@ class SimpleCPUOffloadScheduler:
|
||||
)
|
||||
|
||||
# TODO (yifan): maybe need to enable kv_cache_events and metrics_collector here.
|
||||
dcp_world_size = vllm_config.parallel_config.decode_context_parallel_size
|
||||
pcp_world_size = vllm_config.parallel_config.prefill_context_parallel_size
|
||||
assert dcp_world_size == 1 and pcp_world_size == 1
|
||||
self.cpu_coordinator: KVCacheCoordinator = get_kv_cache_coordinator(
|
||||
kv_cache_config=self.cpu_kv_cache_config,
|
||||
max_model_len=vllm_config.model_config.max_model_len,
|
||||
@@ -155,6 +155,7 @@ class SimpleCPUOffloadScheduler:
|
||||
self._target_free = self._estimate_lazy_target_blocks(
|
||||
kv_cache_config,
|
||||
vllm_config.scheduler_config.max_num_batched_tokens,
|
||||
self.cp_world_size,
|
||||
)
|
||||
else:
|
||||
self._target_free = 0
|
||||
@@ -187,7 +188,13 @@ class SimpleCPUOffloadScheduler:
|
||||
|
||||
assert len(gpu_config.kv_cache_tensors) > 0
|
||||
|
||||
gpu_total_bytes = sum(t.size for t in gpu_config.kv_cache_tensors)
|
||||
is_packed = any(t.block_stride for t in gpu_config.kv_cache_tensors)
|
||||
assert not is_packed or all(t.block_stride for t in gpu_config.kv_cache_tensors)
|
||||
gpu_total_bytes = (
|
||||
gpu_config.kv_cache_tensors[0].size
|
||||
if is_packed
|
||||
else sum(t.size for t in gpu_config.kv_cache_tensors)
|
||||
)
|
||||
num_gpu_blocks = gpu_config.num_blocks
|
||||
num_cpu_blocks = max(1, num_gpu_blocks * cpu_capacity_bytes // gpu_total_bytes)
|
||||
# Create CPU kv_cache_tensors mirroring GPU by scaling size proportionally.
|
||||
@@ -195,6 +202,8 @@ class SimpleCPUOffloadScheduler:
|
||||
KVCacheTensor(
|
||||
size=t.size // num_gpu_blocks * num_cpu_blocks,
|
||||
shared_by=list(t.shared_by),
|
||||
offset=t.offset,
|
||||
block_stride=t.block_stride,
|
||||
)
|
||||
for t in gpu_config.kv_cache_tensors
|
||||
]
|
||||
@@ -207,19 +216,22 @@ class SimpleCPUOffloadScheduler:
|
||||
|
||||
@staticmethod
|
||||
def _estimate_lazy_target_blocks(
|
||||
kv_cache_config: "KVCacheConfig", max_num_batched_tokens: int
|
||||
kv_cache_config: "KVCacheConfig",
|
||||
max_num_batched_tokens: int,
|
||||
cp_world_size: int = 1,
|
||||
) -> int:
|
||||
"""GPU blocks to keep available (free/offloaded) per step in lazy mode."""
|
||||
WATERMARK_RATIO = 1.0 # Reserve larger space to avoid running out of GPU blocks
|
||||
target = 0
|
||||
for g in kv_cache_config.kv_cache_groups:
|
||||
spec = g.kv_cache_spec
|
||||
block_size = spec.block_size * cp_world_size
|
||||
if isinstance(spec, MambaSpec):
|
||||
target += 2
|
||||
elif isinstance(spec, SlidingWindowSpec):
|
||||
target += cdiv(spec.sliding_window, spec.block_size) + 1
|
||||
target += cdiv(spec.sliding_window, block_size) + 1
|
||||
else:
|
||||
target += cdiv(max_num_batched_tokens, spec.block_size)
|
||||
target += cdiv(max_num_batched_tokens, block_size)
|
||||
return int(target * (1 + WATERMARK_RATIO))
|
||||
|
||||
def bind_gpu_block_pool(self, gpu_block_pool: BlockPool) -> None:
|
||||
@@ -355,7 +367,9 @@ class SimpleCPUOffloadScheduler:
|
||||
continue
|
||||
|
||||
# Number of blocks in the computed range for this group.
|
||||
g_block_size = kv_cache_groups[g].kv_cache_spec.block_size
|
||||
g_block_size = (
|
||||
kv_cache_groups[g].kv_cache_spec.block_size * self.cp_world_size
|
||||
)
|
||||
n_computed_g = cdiv(total_computed_tokens, g_block_size)
|
||||
|
||||
# Back-trace: ext blocks sit at the tail of the computed range.
|
||||
|
||||
@@ -8,7 +8,7 @@ import torch
|
||||
|
||||
from vllm.config import VllmConfig
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
from vllm.v1.simple_kv_offload.copy_backend import DmaCopyBackend
|
||||
from vllm.v1.simple_kv_offload.cuda_mem_ops import pin_tensor
|
||||
from vllm.v1.simple_kv_offload.metadata import (
|
||||
@@ -149,7 +149,7 @@ class SimpleCPUOffloadWorker:
|
||||
(self.num_cpu_blocks * total_bytes_per_block) / (1024**3),
|
||||
)
|
||||
|
||||
pin_memory = is_pin_memory_available()
|
||||
pin_memory = PIN_MEMORY
|
||||
if not pin_memory:
|
||||
logger.warning(
|
||||
"Pinned memory not available. CPU offload performance may be degraded."
|
||||
|
||||
@@ -12,7 +12,7 @@ from vllm.config import CUDAGraphMode, VllmConfig, get_layers_from_vllm_config
|
||||
from vllm.forward_context import set_forward_context
|
||||
from vllm.model_executor.layers.attention_layer_base import AttentionLayerBase
|
||||
from vllm.model_executor.model_loader import get_model
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
from vllm.v1.attention.backend import AttentionMetadataBuilder, CommonAttentionMetadata
|
||||
from vllm.v1.cudagraph_dispatcher import CudagraphDispatcher
|
||||
from vllm.v1.utils import CpuGpuBuffer
|
||||
@@ -58,7 +58,7 @@ class ExtractHiddenStatesProposer:
|
||||
self.backup_next_token_ids = CpuGpuBuffer(
|
||||
max_batch_size,
|
||||
dtype=torch.int32,
|
||||
pin_memory=is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
device=device,
|
||||
with_numpy=True,
|
||||
)
|
||||
@@ -317,7 +317,6 @@ class ExtractHiddenStatesProposer:
|
||||
(batch_size, 1). For each request we either use the sampled token
|
||||
(if valid and not discarded) or a backup token from the request state.
|
||||
"""
|
||||
num_reqs = gpu_input_batch.num_reqs
|
||||
|
||||
# Precompute backup token IDs for discarded requests.
|
||||
num_reqs = gpu_input_batch.num_reqs
|
||||
|
||||
@@ -26,7 +26,7 @@ from vllm.model_executor.models.llama_eagle3 import Eagle3LlamaForCausalLM
|
||||
from vllm.model_executor.models.qwen3_dflash import DFlashQwen3ForCausalLM
|
||||
from vllm.multimodal import MULTIMODAL_REGISTRY
|
||||
from vllm.platforms import current_platform
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY, async_tensor_h2d
|
||||
from vllm.v1.attention.backend import CommonAttentionMetadata
|
||||
from vllm.v1.attention.backends.registry import AttentionBackendEnum
|
||||
from vllm.v1.attention.backends.triton_attn import TritonAttentionMetadata
|
||||
@@ -228,7 +228,7 @@ class SpecDecodeBaseProposer:
|
||||
self.backup_next_token_ids = CpuGpuBuffer(
|
||||
self.max_batch_size,
|
||||
dtype=torch.int32,
|
||||
pin_memory=is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
device=device,
|
||||
with_numpy=True,
|
||||
)
|
||||
@@ -239,9 +239,7 @@ class SpecDecodeBaseProposer:
|
||||
self._last_draft_probs: torch.Tensor | None = None
|
||||
|
||||
self._slot_mapping_buffer = torch.zeros(
|
||||
self.max_positions,
|
||||
dtype=torch.int64,
|
||||
device=device,
|
||||
self.max_positions, dtype=torch.int64, device=device
|
||||
)
|
||||
|
||||
# Determine allowed attention backends once during initialization.
|
||||
@@ -1127,7 +1125,7 @@ class SpecDecodeBaseProposer:
|
||||
new_query_start_loc_cpu = torch.zeros(
|
||||
query_start_loc_cpu.shape,
|
||||
dtype=torch.int32,
|
||||
pin_memory=is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
new_query_start_loc_np = new_query_start_loc_cpu.numpy()
|
||||
np.cumsum(new_num_tokens_per_req_np, out=new_query_start_loc_np[1:])
|
||||
@@ -1160,11 +1158,11 @@ class SpecDecodeBaseProposer:
|
||||
# q1 + 0, q1 + 1, q1 + 2, q1 + 3, // req 2
|
||||
# q1 + q2 + 0, q1 + q2 + 1, q1 + q2 + 2] // req 3
|
||||
token_indices_np = token_offsets + old_query_start_locs_expanded
|
||||
token_indices = torch.from_numpy(token_indices_np).to(device, non_blocking=True)
|
||||
token_indices = async_tensor_h2d(token_indices_np, device=device)
|
||||
|
||||
spec_common_attn_metadata = CommonAttentionMetadata(
|
||||
query_start_loc=new_query_start_loc_cpu.to(device, non_blocking=True),
|
||||
seq_lens=new_seq_lens_cpu.to(device, non_blocking=True),
|
||||
query_start_loc=async_tensor_h2d(new_query_start_loc_cpu, device=device),
|
||||
seq_lens=async_tensor_h2d(new_seq_lens_cpu, device=device),
|
||||
query_start_loc_cpu=new_query_start_loc_cpu,
|
||||
_seq_lens_cpu=new_seq_lens_cpu,
|
||||
_num_computed_tokens_cpu=common_attn_metadata._num_computed_tokens_cpu,
|
||||
|
||||
@@ -545,7 +545,7 @@ def update_ngram_gpu_tensors_incremental(
|
||||
num_tokens = input_batch.num_tokens_no_spec[idx]
|
||||
if num_tokens > 0:
|
||||
token_ids_gpu_tensor[idx, :num_tokens].copy_(
|
||||
input_batch.token_ids_cpu_tensor[idx, :num_tokens],
|
||||
input_batch.token_ids_cpu_tensor[idx, :num_tokens].pin_memory(),
|
||||
non_blocking=True,
|
||||
)
|
||||
|
||||
@@ -591,7 +591,7 @@ def update_ngram_gpu_tensors_incremental(
|
||||
num_tokens = input_batch.num_tokens_no_spec[new_req_idx]
|
||||
if num_tokens > 0:
|
||||
token_ids_gpu_tensor[new_req_idx, :num_tokens].copy_(
|
||||
input_batch.token_ids_cpu_tensor[new_req_idx, :num_tokens],
|
||||
input_batch.token_ids_cpu_tensor[new_req_idx, :num_tokens].pin_memory(),
|
||||
non_blocking=True,
|
||||
)
|
||||
|
||||
|
||||
@@ -11,7 +11,7 @@ from transformers import PreTrainedTokenizerBase
|
||||
|
||||
from vllm.sampling_params import SamplingParams
|
||||
from vllm.utils.import_utils import LazyLoader
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
from vllm.v1.structured_output.backend_types import (
|
||||
StructuredOutputBackend,
|
||||
StructuredOutputGrammar,
|
||||
@@ -139,7 +139,7 @@ class LMFormatEnforcerBackend(StructuredOutputBackend):
|
||||
(max_num_seqs, (self.vocab_size + 31) // 32),
|
||||
-1,
|
||||
dtype=torch.int32,
|
||||
pin_memory=is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
|
||||
def destroy(self):
|
||||
|
||||
@@ -15,7 +15,7 @@ from regex import escape as regex_escape
|
||||
|
||||
from vllm.sampling_params import SamplingParams
|
||||
from vllm.utils.import_utils import LazyLoader
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
from vllm.v1.structured_output.backend_types import (
|
||||
StructuredOutputBackend,
|
||||
StructuredOutputGrammar,
|
||||
@@ -101,7 +101,7 @@ class OutlinesBackend(StructuredOutputBackend):
|
||||
(max_num_seqs, (self.vocab_size + 31) // 32),
|
||||
-1,
|
||||
dtype=torch.int32,
|
||||
pin_memory=is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
|
||||
def destroy(self):
|
||||
|
||||
@@ -10,7 +10,6 @@ from collections.abc import Callable
|
||||
from concurrent.futures import ThreadPoolExecutor, TimeoutError
|
||||
from typing import TYPE_CHECKING, TypeVar
|
||||
|
||||
import numpy as np
|
||||
import regex as re
|
||||
import torch
|
||||
from cachetools import LRUCache
|
||||
@@ -18,7 +17,7 @@ from cachetools import LRUCache
|
||||
import vllm.envs as envs
|
||||
from vllm.logger import init_logger
|
||||
from vllm.utils.import_utils import LazyLoader
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import PIN_MEMORY, async_tensor_h2d
|
||||
from vllm.v1.core.sched.output import GrammarOutput, SchedulerOutput
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -123,11 +122,13 @@ def apply_grammar_bitmask(
|
||||
out_indices = []
|
||||
|
||||
# Reorder the bitmask to match the order of the requests in the batch.
|
||||
sorted_bitmask = np.full(
|
||||
shape=(logits.shape[0], grammar_bitmask.shape[1]),
|
||||
fill_value=-1,
|
||||
dtype=grammar_bitmask.dtype,
|
||||
sorted_bitmask_tensor = torch.full(
|
||||
(logits.shape[0], grammar_bitmask.shape[1]),
|
||||
-1,
|
||||
dtype=torch.from_numpy(grammar_bitmask[:0]).dtype,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
sorted_bitmask = sorted_bitmask_tensor.numpy()
|
||||
cumulative_index = 0
|
||||
for req_id in grammar_output.structured_output_request_ids:
|
||||
num_spec_tokens = len(spec_tokens.get(req_id, ()))
|
||||
@@ -138,10 +139,8 @@ def apply_grammar_bitmask(
|
||||
out_indices.append(bitmask_index)
|
||||
cumulative_index += 1 + num_spec_tokens
|
||||
|
||||
# Copy async to device as tensor.
|
||||
grammar_bitmask = torch.from_numpy(sorted_bitmask).to(
|
||||
logits.device, non_blocking=True
|
||||
)
|
||||
# Copy async to device.
|
||||
grammar_bitmask = sorted_bitmask_tensor.to(logits.device, non_blocking=True)
|
||||
|
||||
# If the length of out indices and the logits have the same shape
|
||||
# we don't need to pass indices to the kernel,
|
||||
@@ -154,11 +153,9 @@ def apply_grammar_bitmask(
|
||||
# xgrammar expects a python list of indices but it will actually work with
|
||||
# a tensor. If we copy the tensor ourselves here we can do it in a
|
||||
# non_blocking manner and there should be no cpu sync within xgrammar.
|
||||
pin_memory = is_pin_memory_available()
|
||||
index_tensor = torch.tensor(
|
||||
out_indices, dtype=torch.int32, device="cpu", pin_memory=pin_memory
|
||||
index_tensor = async_tensor_h2d(
|
||||
out_indices, dtype=torch.int32, device=logits.device
|
||||
)
|
||||
index_tensor = index_tensor.to(logits.device, non_blocking=True)
|
||||
|
||||
xgr.apply_token_bitmask_inplace(logits, grammar_bitmask, indices=index_tensor)
|
||||
return
|
||||
|
||||
+2
-1
@@ -31,6 +31,7 @@ from vllm.logger import init_logger
|
||||
from vllm.usage.usage_lib import UsageContext, is_usage_stats_enabled, usage_message
|
||||
from vllm.utils.network_utils import get_open_zmq_ipc_path, get_tcp_uri
|
||||
from vllm.utils.system_utils import decorate_logs, kill_process_tree, set_process_title
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
from vllm.v1.core.sched.output import SchedulerOutput
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -114,7 +115,7 @@ class CpuGpuBuffer:
|
||||
*size: int | torch.SymInt,
|
||||
dtype: torch.dtype,
|
||||
device: torch.device,
|
||||
pin_memory: bool,
|
||||
pin_memory: bool = PIN_MEMORY,
|
||||
with_numpy: bool = True,
|
||||
) -> None:
|
||||
# these buffers are mutable runtime state, so allocate them as normal
|
||||
|
||||
@@ -7,6 +7,8 @@
|
||||
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
|
||||
# Patch torch APIs
|
||||
import torch
|
||||
|
||||
@@ -45,11 +47,14 @@ import vllm.utils.torch_utils as torch_utils
|
||||
|
||||
|
||||
def async_tensor_h2d(
|
||||
data: list,
|
||||
dtype: torch.dtype,
|
||||
data: list | np.ndarray | torch.Tensor,
|
||||
device: str | torch.device,
|
||||
pin_memory: bool = False,
|
||||
dtype: torch.dtype | None = None,
|
||||
) -> torch.Tensor:
|
||||
if isinstance(data, np.ndarray):
|
||||
data = torch.from_numpy(data)
|
||||
if isinstance(data, torch.Tensor):
|
||||
return data.to(dtype=dtype)
|
||||
return torch.tensor(data, dtype=dtype, device="cpu")
|
||||
|
||||
|
||||
|
||||
@@ -36,10 +36,9 @@ def async_copy_to_gpu(
|
||||
assert device is not None
|
||||
out = torch.empty_like(x, device=device)
|
||||
|
||||
# Copy directly to GPU — explicit pin_memory() causes sporadic stalls
|
||||
# under high concurrency due to CUDA driver contention. The driver
|
||||
# handles the transfer efficiently without manual pinning.
|
||||
return out.copy_(x, non_blocking=True)
|
||||
# pin_memory() is no-op if the memory is already pinned.
|
||||
pinned = x.pin_memory()
|
||||
return out.copy_(pinned, non_blocking=True)
|
||||
|
||||
|
||||
class UvaBuffer:
|
||||
@@ -183,7 +182,7 @@ class StagedWriteTensor:
|
||||
|
||||
# Special handling for write_contents
|
||||
write_contents = async_tensor_h2d(
|
||||
self._staged_write_contents, self.dtype, self.device
|
||||
self._staged_write_contents, device=self.device, dtype=self.dtype
|
||||
)
|
||||
|
||||
# Write diffs to the GPU buffer
|
||||
@@ -255,7 +254,7 @@ class FusedStagedWriter:
|
||||
indices_uva = self.indices.copy_to_uva(indices)
|
||||
starts_uva = self.starts.copy_to_uva(starts)
|
||||
cu_lens_uva = self.cu_lens.copy_to_uva(cu_lens)
|
||||
contents_gpu = async_tensor_h2d(contents, torch.int32, self.device)
|
||||
contents_gpu = async_tensor_h2d(contents, device=self.device, dtype=torch.int32)
|
||||
|
||||
_apply_write_kernel[(len(group_ids),)](
|
||||
output_ptrs,
|
||||
|
||||
@@ -49,12 +49,11 @@ class EncoderRunner:
|
||||
|
||||
@torch.inference_mode()
|
||||
def execute_mm_encoder(
|
||||
self,
|
||||
mm_kwargs: list[tuple[str, MultiModalKwargsItem]],
|
||||
self, mm_kwargs: list[tuple[str, MultiModalKwargsItem]]
|
||||
) -> list[torch.Tensor]:
|
||||
encoder_outputs: list[torch.Tensor] = []
|
||||
for modality, num_items, mm_kwargs_batch in group_and_batch_mm_kwargs(
|
||||
mm_kwargs, device=self.device, pin_memory=False
|
||||
mm_kwargs, device=self.device, pin_memory=True
|
||||
):
|
||||
batch_outputs = self.model.embed_multimodal(**mm_kwargs_batch)
|
||||
sanity_check_mm_encoder_outputs(batch_outputs, expected_num_items=num_items)
|
||||
|
||||
@@ -46,8 +46,7 @@ from vllm.sequence import IntermediateTensors
|
||||
from vllm.tasks import SupportedTask
|
||||
from vllm.utils.math_utils import cdiv
|
||||
from vllm.utils.mem_utils import DeviceMemoryProfiler, format_gib
|
||||
from vllm.utils.platform_utils import is_pin_memory_available
|
||||
from vllm.utils.torch_utils import STR_DTYPE_TO_TORCH_DTYPE
|
||||
from vllm.utils.torch_utils import PIN_MEMORY, STR_DTYPE_TO_TORCH_DTYPE
|
||||
from vllm.v1.core.sched.output import GrammarOutput, SchedulerOutput
|
||||
from vllm.v1.kv_cache_interface import KVCacheConfig, MambaSpec
|
||||
from vllm.v1.outputs import DraftTokenIds, ModelRunnerOutput
|
||||
@@ -498,7 +497,7 @@ class GPUModelRunner(LoRAModelRunnerMixin):
|
||||
"""Build KV-block zeroing metadata; invoked from gpu_worker."""
|
||||
self.kv_block_zeroer = KVBlockZeroer(
|
||||
self.device,
|
||||
is_pin_memory_available(),
|
||||
pin_memory=PIN_MEMORY,
|
||||
attn_groups_iter=(g for groups in self.attn_groups for g in groups),
|
||||
kernel_block_sizes=self.kernel_block_sizes,
|
||||
cache_dtype=self.cache_config.cache_dtype,
|
||||
|
||||
@@ -170,7 +170,8 @@ class WhisperModelState(ModelState):
|
||||
for_capture: bool,
|
||||
num_reqs: int,
|
||||
) -> dict[int, tuple[torch.Tensor, np.ndarray]]:
|
||||
encoder_seq_lens_np = np.zeros(num_reqs, dtype=np.int32)
|
||||
encoder_seq_lens = torch.zeros(num_reqs, dtype=torch.int32, pin_memory=True)
|
||||
encoder_seq_lens_np = encoder_seq_lens.numpy()
|
||||
if not for_capture:
|
||||
# During normal execution, use actual encoder lengths.
|
||||
for i, req_id in enumerate(req_ids):
|
||||
@@ -183,9 +184,7 @@ class WhisperModelState(ModelState):
|
||||
# is captured with the correct value for cross-attention.
|
||||
encoder_seq_lens_np[:] = self.max_encoder_len
|
||||
|
||||
self.encoder_seq_lens_gpu[:num_reqs].copy_(
|
||||
torch.from_numpy(encoder_seq_lens_np), non_blocking=True
|
||||
)
|
||||
self.encoder_seq_lens_gpu[:num_reqs].copy_(encoder_seq_lens, non_blocking=True)
|
||||
self.encoder_seq_lens_gpu[num_reqs:].fill_(0)
|
||||
encoder_seq_lens_gpu = self.encoder_seq_lens_gpu[:num_reqs]
|
||||
|
||||
|
||||
@@ -222,7 +222,7 @@ def _bias_kernel(
|
||||
num_stop_token_ids = tl.load(num_stop_token_ids_ptr + req_state_idx)
|
||||
pos = tl.load(pos_ptr + token_idx)
|
||||
min_len = tl.load(min_lens_ptr + req_state_idx)
|
||||
if num_stop_token_ids > 0 and pos < min_len:
|
||||
if num_stop_token_ids > 0 and pos + 1 < min_len:
|
||||
mask = block < num_stop_token_ids
|
||||
stop_token_ids = tl.load(
|
||||
stop_token_ids_ptr + req_state_idx * stop_token_ids_stride + block,
|
||||
|
||||
@@ -15,6 +15,7 @@ from vllm.pooling_params import PoolingParams
|
||||
from vllm.sampling_params import SamplingParams, SamplingType
|
||||
from vllm.utils import length_from_prompt_token_ids_or_embeds
|
||||
from vllm.utils.collection_utils import swap_dict_values
|
||||
from vllm.utils.torch_utils import PIN_MEMORY
|
||||
from vllm.v1.outputs import LogprobsTensors
|
||||
from vllm.v1.pool.metadata import PoolingMetadata, PoolingStates
|
||||
from vllm.v1.sample.logits_processor import (
|
||||
@@ -95,7 +96,6 @@ class InputBatch:
|
||||
max_model_len: int,
|
||||
max_num_batched_tokens: int,
|
||||
device: torch.device,
|
||||
pin_memory: bool,
|
||||
vocab_size: int,
|
||||
block_sizes: list[int], # The block_size of each kv cache group
|
||||
kernel_block_sizes: list[int],
|
||||
@@ -112,7 +112,6 @@ class InputBatch:
|
||||
max_num_reqs,
|
||||
num_spec_tokens,
|
||||
device,
|
||||
pin_memory,
|
||||
)
|
||||
self.thinking_token_budget_reqs: set[str] = set()
|
||||
self.is_pooling_model = is_pooling_model
|
||||
@@ -120,7 +119,6 @@ class InputBatch:
|
||||
self.max_model_len = max_model_len
|
||||
self.max_num_batched_tokens = max_num_batched_tokens
|
||||
self.device = device
|
||||
self.pin_memory = pin_memory
|
||||
self.vocab_size = vocab_size
|
||||
|
||||
self._req_ids: list[str | None] = []
|
||||
@@ -138,7 +136,10 @@ class InputBatch:
|
||||
)
|
||||
self.token_ids_cpu = self.token_ids_cpu_tensor.numpy()
|
||||
self.is_token_ids_tensor = torch.zeros(
|
||||
(max_num_reqs, max_model_len), device="cpu", dtype=bool, pin_memory=False
|
||||
(max_num_reqs, max_model_len),
|
||||
device="cpu",
|
||||
dtype=bool,
|
||||
pin_memory=False,
|
||||
)
|
||||
self.is_token_ids = self.is_token_ids_tensor.numpy()
|
||||
# Store prompt embeddings per request to avoid OOM from large upfront
|
||||
@@ -149,21 +150,21 @@ class InputBatch:
|
||||
(max_num_reqs,),
|
||||
device="cpu",
|
||||
dtype=torch.int32,
|
||||
pin_memory=pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
self.num_tokens_no_spec = self.num_tokens_no_spec_cpu_tensor.numpy()
|
||||
self.num_prompt_tokens_cpu_tensor = torch.zeros(
|
||||
(max_num_reqs,),
|
||||
device="cpu",
|
||||
dtype=torch.int32,
|
||||
pin_memory=pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
self.num_prompt_tokens = self.num_prompt_tokens_cpu_tensor.numpy()
|
||||
self.num_computed_tokens_cpu_tensor = torch.zeros(
|
||||
(max_num_reqs,),
|
||||
device="cpu",
|
||||
dtype=torch.int32,
|
||||
pin_memory=pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
self.num_computed_tokens_cpu = self.num_computed_tokens_cpu_tensor.numpy()
|
||||
|
||||
@@ -172,7 +173,7 @@ class InputBatch:
|
||||
max_num_reqs=max_num_reqs,
|
||||
max_model_len=max_model_len,
|
||||
max_num_batched_tokens=max_num_batched_tokens,
|
||||
pin_memory=pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
device=device,
|
||||
block_sizes=block_sizes,
|
||||
kernel_block_sizes=kernel_block_sizes,
|
||||
@@ -185,7 +186,7 @@ class InputBatch:
|
||||
(max_num_reqs,), dtype=torch.float32, device=device
|
||||
)
|
||||
self.temperature_cpu_tensor = torch.empty(
|
||||
(max_num_reqs,), dtype=torch.float32, device="cpu", pin_memory=pin_memory
|
||||
(max_num_reqs,), dtype=torch.float32, device="cpu", pin_memory=PIN_MEMORY
|
||||
)
|
||||
self.temperature_cpu = self.temperature_cpu_tensor.numpy()
|
||||
self.greedy_reqs: set[str] = set()
|
||||
@@ -193,14 +194,14 @@ class InputBatch:
|
||||
|
||||
self.top_p = torch.empty((max_num_reqs,), dtype=torch.float32, device=device)
|
||||
self.top_p_cpu_tensor = torch.empty(
|
||||
(max_num_reqs,), dtype=torch.float32, device="cpu", pin_memory=pin_memory
|
||||
(max_num_reqs,), dtype=torch.float32, device="cpu", pin_memory=PIN_MEMORY
|
||||
)
|
||||
self.top_p_cpu = self.top_p_cpu_tensor.numpy()
|
||||
self.top_p_reqs: set[str] = set()
|
||||
|
||||
self.top_k = torch.empty((max_num_reqs,), dtype=torch.int32, device=device)
|
||||
self.top_k_cpu_tensor = torch.empty(
|
||||
(max_num_reqs,), dtype=torch.int32, device="cpu", pin_memory=pin_memory
|
||||
(max_num_reqs,), dtype=torch.int32, device="cpu", pin_memory=PIN_MEMORY
|
||||
)
|
||||
self.top_k_cpu = self.top_k_cpu_tensor.numpy()
|
||||
self.top_k_reqs: set[str] = set()
|
||||
@@ -210,7 +211,7 @@ class InputBatch:
|
||||
(max_num_reqs,), dtype=torch.float, device=device
|
||||
)
|
||||
self.frequency_penalties_cpu_tensor = torch.empty(
|
||||
(max_num_reqs,), dtype=torch.float, device="cpu", pin_memory=pin_memory
|
||||
(max_num_reqs,), dtype=torch.float, device="cpu", pin_memory=PIN_MEMORY
|
||||
)
|
||||
self.frequency_penalties_cpu = self.frequency_penalties_cpu_tensor.numpy()
|
||||
self.frequency_penalties_reqs: set[str] = set()
|
||||
@@ -220,7 +221,7 @@ class InputBatch:
|
||||
(max_num_reqs,), dtype=torch.float, device=device
|
||||
)
|
||||
self.presence_penalties_cpu_tensor = torch.empty(
|
||||
(max_num_reqs,), dtype=torch.float, device="cpu", pin_memory=pin_memory
|
||||
(max_num_reqs,), dtype=torch.float, device="cpu", pin_memory=PIN_MEMORY
|
||||
)
|
||||
self.presence_penalties_cpu = self.presence_penalties_cpu_tensor.numpy()
|
||||
self.presence_penalties_reqs: set[str] = set()
|
||||
@@ -230,14 +231,14 @@ class InputBatch:
|
||||
(max_num_reqs,), dtype=torch.float, device=device
|
||||
)
|
||||
self.repetition_penalties_cpu_tensor = torch.empty(
|
||||
(max_num_reqs,), dtype=torch.float, device="cpu", pin_memory=pin_memory
|
||||
(max_num_reqs,), dtype=torch.float, device="cpu", pin_memory=PIN_MEMORY
|
||||
)
|
||||
self.repetition_penalties_cpu = self.repetition_penalties_cpu_tensor.numpy()
|
||||
self.repetition_penalties_reqs: set[str] = set()
|
||||
|
||||
# Speculative decoding
|
||||
self.num_accepted_tokens_cpu_tensor = torch.ones(
|
||||
(max_num_reqs,), dtype=torch.int32, device="cpu", pin_memory=pin_memory
|
||||
(max_num_reqs,), dtype=torch.int32, device="cpu", pin_memory=PIN_MEMORY
|
||||
)
|
||||
self.num_accepted_tokens_cpu = self.num_accepted_tokens_cpu_tensor.numpy()
|
||||
|
||||
@@ -963,7 +964,7 @@ class InputBatch:
|
||||
(self.num_reqs, max_prompt_len),
|
||||
device="cpu",
|
||||
dtype=torch.int64,
|
||||
pin_memory=self.pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
prompt_token_ids = prompt_token_ids_cpu_tensor.numpy()
|
||||
prompt_token_ids[:] = self.token_ids_cpu[:num_reqs, :max_prompt_len]
|
||||
|
||||
@@ -116,8 +116,10 @@ from vllm.utils import length_from_prompt_token_ids_or_embeds
|
||||
from vllm.utils.math_utils import cdiv, round_up
|
||||
from vllm.utils.mem_utils import DeviceMemoryProfiler, format_gib
|
||||
from vllm.utils.nvtx_pytorch_hooks import PytHooks
|
||||
from vllm.utils.platform_utils import is_pin_memory_available, num_compute_units
|
||||
from vllm.utils.platform_utils import num_compute_units
|
||||
from vllm.utils.torch_utils import (
|
||||
PIN_MEMORY,
|
||||
async_tensor_h2d,
|
||||
get_dtype_size,
|
||||
is_quantized_kv_cache,
|
||||
kv_cache_dtype_str_to_dtype,
|
||||
@@ -441,7 +443,6 @@ class GPUModelRunner(
|
||||
scheduler_config = self.scheduler_config
|
||||
parallel_config = self.parallel_config
|
||||
self.device = device
|
||||
self.pin_memory = is_pin_memory_available()
|
||||
self.dtype = self.model_config.dtype
|
||||
|
||||
self.kv_cache_dtype = kv_cache_dtype_str_to_dtype(
|
||||
@@ -666,7 +667,6 @@ class GPUModelRunner(
|
||||
max_model_len=max(self.max_model_len, self.max_encoder_len),
|
||||
max_num_batched_tokens=self.max_num_tokens,
|
||||
device=self.device,
|
||||
pin_memory=self.pin_memory,
|
||||
vocab_size=self.model_config.get_vocab_size(),
|
||||
block_sizes=[placeholder_block_size],
|
||||
kernel_block_sizes=[placeholder_block_size],
|
||||
@@ -674,7 +674,7 @@ class GPUModelRunner(
|
||||
logitsprocs=build_logitsprocs(
|
||||
self.vllm_config,
|
||||
self.device,
|
||||
self.pin_memory,
|
||||
PIN_MEMORY,
|
||||
self.is_pooling_model,
|
||||
custom_logitsprocs,
|
||||
),
|
||||
@@ -729,7 +729,7 @@ class GPUModelRunner(
|
||||
self.max_num_reqs, dtype=torch.int32, device=self.device
|
||||
)
|
||||
self.optimistic_seq_lens_cpu = torch.zeros(
|
||||
self.max_num_reqs, dtype=torch.int32, pin_memory=self.pin_memory
|
||||
self.max_num_reqs, dtype=torch.int32, pin_memory=PIN_MEMORY
|
||||
)
|
||||
self.num_computed_tokens = torch.zeros(
|
||||
self.max_num_reqs, dtype=torch.int32, device=self.device
|
||||
@@ -846,7 +846,7 @@ class GPUModelRunner(
|
||||
and self.speculative_config.use_ngram_gpu()
|
||||
):
|
||||
self._num_valid_draft_tokens_cpu = torch.empty(
|
||||
self.max_num_reqs, dtype=torch.int32, pin_memory=self.pin_memory
|
||||
self.max_num_reqs, dtype=torch.int32, pin_memory=PIN_MEMORY
|
||||
)
|
||||
self._num_valid_draft_tokens_event = torch.cuda.Event()
|
||||
self._num_valid_draft_tokens_copy_stream = torch.cuda.Stream()
|
||||
@@ -857,7 +857,7 @@ class GPUModelRunner(
|
||||
(self.max_num_reqs, 1),
|
||||
dtype=torch.int64,
|
||||
device="cpu",
|
||||
pin_memory=self.pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
|
||||
# Pre-allocated tensor for copying valid sampled token counts to CPU,
|
||||
@@ -879,7 +879,7 @@ class GPUModelRunner(
|
||||
(self.max_num_reqs, self.num_spec_tokens),
|
||||
dtype=torch.int64,
|
||||
device="cpu",
|
||||
pin_memory=self.pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
if self.use_async_scheduling:
|
||||
self.valid_sampled_token_count_event = torch.Event()
|
||||
@@ -888,7 +888,7 @@ class GPUModelRunner(
|
||||
self.max_num_reqs,
|
||||
dtype=torch.int32,
|
||||
device="cpu",
|
||||
pin_memory=self.pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
|
||||
# Model weight offloader
|
||||
@@ -1000,7 +1000,6 @@ class GPUModelRunner(
|
||||
*size,
|
||||
dtype=dtype,
|
||||
device=self.device,
|
||||
pin_memory=self.pin_memory,
|
||||
with_numpy=numpy,
|
||||
)
|
||||
|
||||
@@ -1055,7 +1054,7 @@ class GPUModelRunner(
|
||||
token_type_ids.append(ids)
|
||||
|
||||
token_type_ids_cpu = torch.empty(
|
||||
sum(seq_lens_cpu), dtype=torch.int32, pin_memory=self.pin_memory
|
||||
sum(seq_lens_cpu), dtype=torch.int32, pin_memory=PIN_MEMORY
|
||||
)
|
||||
torch.cat(token_type_ids, out=token_type_ids_cpu)
|
||||
model_kwargs["token_type_ids"] = token_type_ids_cpu.to(
|
||||
@@ -1095,12 +1094,12 @@ class GPUModelRunner(
|
||||
"""
|
||||
self._kv_block_zeroer = KVBlockZeroer(
|
||||
self.device,
|
||||
self.pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
attn_groups_iter=self._kv_cache_spec_attn_group_iterator(),
|
||||
kernel_block_sizes=self._kernel_block_sizes,
|
||||
cache_dtype=self.cache_config.cache_dtype,
|
||||
runner_only_attn_layers=self.runner_only_attn_layers,
|
||||
static_forward_context=(self.compilation_config.static_forward_context),
|
||||
static_forward_context=self.compilation_config.static_forward_context,
|
||||
)
|
||||
|
||||
def _zero_block_ids(self, block_ids: list[int]) -> None:
|
||||
@@ -1651,7 +1650,7 @@ class GPUModelRunner(
|
||||
for _, _, mm_kwargs_batch in group_and_batch_mm_kwargs(
|
||||
mm_kwargs,
|
||||
device=self.device,
|
||||
pin_memory=self.pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
):
|
||||
mm_kwargs_combined.update(mm_kwargs_batch)
|
||||
|
||||
@@ -1807,10 +1806,10 @@ class GPUModelRunner(
|
||||
return
|
||||
# Upload the index tensors asynchronously so the scatter can be non-blocking.
|
||||
sampled_tokens_index_tensor = torch.tensor(
|
||||
sample_flattened_indices, dtype=torch.int64, pin_memory=self.pin_memory
|
||||
sample_flattened_indices, dtype=torch.int64, pin_memory=PIN_MEMORY
|
||||
).to(self.device, non_blocking=True)
|
||||
prev_common_req_indices_tensor = torch.tensor(
|
||||
prev_indices, dtype=torch.int64, pin_memory=self.pin_memory
|
||||
prev_indices, dtype=torch.int64, pin_memory=PIN_MEMORY
|
||||
).to(self.device, non_blocking=True)
|
||||
self.input_ids.gpu.scatter_(
|
||||
dim=0,
|
||||
@@ -1826,10 +1825,10 @@ class GPUModelRunner(
|
||||
|
||||
assert isinstance(self._draft_token_ids, torch.Tensor)
|
||||
draft_tokens_index_tensor = torch.tensor(
|
||||
spec_flattened_indices, dtype=torch.int64, pin_memory=self.pin_memory
|
||||
spec_flattened_indices, dtype=torch.int64, pin_memory=PIN_MEMORY
|
||||
).to(self.device, non_blocking=True)
|
||||
prev_draft_token_indices_tensor = torch.tensor(
|
||||
prev_draft_token_indices, dtype=torch.int64, pin_memory=self.pin_memory
|
||||
prev_draft_token_indices, dtype=torch.int64, pin_memory=PIN_MEMORY
|
||||
).to(self.device, non_blocking=True)
|
||||
|
||||
# because input_ids dtype is torch.int32,
|
||||
@@ -2788,21 +2787,16 @@ class GPUModelRunner(
|
||||
# [0, 1, 2, 5, 6, 9]
|
||||
target_logits_indices += self._arange_scratch[: cu_num_draft_tokens[-1]]
|
||||
|
||||
# TODO: Optimize the CPU -> GPU copy.
|
||||
cu_num_draft_tokens = torch.from_numpy(cu_num_draft_tokens).to(
|
||||
self.device, non_blocking=True
|
||||
cu_num_draft_tokens = async_tensor_h2d(cu_num_draft_tokens, device=self.device)
|
||||
cu_num_sampled_tokens = async_tensor_h2d(
|
||||
cu_num_sampled_tokens, device=self.device
|
||||
)
|
||||
cu_num_sampled_tokens = torch.from_numpy(cu_num_sampled_tokens).to(
|
||||
self.device, non_blocking=True
|
||||
logits_indices = async_tensor_h2d(logits_indices, device=self.device)
|
||||
target_logits_indices = async_tensor_h2d(
|
||||
target_logits_indices, device=self.device
|
||||
)
|
||||
logits_indices = torch.from_numpy(logits_indices).to(
|
||||
self.device, non_blocking=True
|
||||
)
|
||||
target_logits_indices = torch.from_numpy(target_logits_indices).to(
|
||||
self.device, non_blocking=True
|
||||
)
|
||||
bonus_logits_indices = torch.from_numpy(bonus_logits_indices).to(
|
||||
self.device, non_blocking=True
|
||||
bonus_logits_indices = async_tensor_h2d(
|
||||
bonus_logits_indices, device=self.device
|
||||
)
|
||||
|
||||
# Compute the draft token ids.
|
||||
@@ -3012,9 +3006,7 @@ class GPUModelRunner(
|
||||
# Track the current index in mm_kwargs/mm_lora_refs to map groups to request IDs
|
||||
current_item_idx = 0
|
||||
for modality, num_items, mm_kwargs_batch in group_and_batch_mm_kwargs(
|
||||
mm_kwargs,
|
||||
device=self.device,
|
||||
pin_memory=self.pin_memory,
|
||||
mm_kwargs, device=self.device, pin_memory=PIN_MEMORY
|
||||
):
|
||||
batch_outputs: MultiModalEmbeddings
|
||||
|
||||
@@ -3048,7 +3040,7 @@ class GPUModelRunner(
|
||||
group_and_batch_mm_kwargs(
|
||||
[video_mm_kwargs_item],
|
||||
device=self.device,
|
||||
pin_memory=self.pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -3107,7 +3099,10 @@ class GPUModelRunner(
|
||||
|
||||
mm_embeds = list[torch.Tensor]()
|
||||
is_mm_embed = torch.zeros(
|
||||
total_num_scheduled_tokens, dtype=torch.bool, device="cpu"
|
||||
total_num_scheduled_tokens,
|
||||
dtype=torch.bool,
|
||||
device="cpu",
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
|
||||
req_start_idx = 0
|
||||
@@ -3515,8 +3510,7 @@ class GPUModelRunner(
|
||||
token_ids_idx_np = np.nonzero(is_token_ids)[0]
|
||||
# Some tokens ids may need to become embeds
|
||||
if token_ids_idx_np.size > 0:
|
||||
token_ids_idx = torch.from_numpy(token_ids_idx_np)
|
||||
token_ids_idx = token_ids_idx.to(self.device, non_blocking=True)
|
||||
token_ids_idx = async_tensor_h2d(token_ids_idx_np, device=self.device)
|
||||
token_ids = self.input_ids.gpu[token_ids_idx]
|
||||
tokens_to_embeds = self.model.embed_input_ids(input_ids=token_ids)
|
||||
self.inputs_embeds.gpu[token_ids_idx] = tokens_to_embeds
|
||||
@@ -4953,7 +4947,7 @@ class GPUModelRunner(
|
||||
):
|
||||
indices.append(offset + len(tokens) - 1)
|
||||
offset += num_draft + 1
|
||||
indices = torch.tensor(indices, device=self.device)
|
||||
indices = async_tensor_h2d(indices, device=self.device)
|
||||
hidden_states = sample_hidden_states[indices]
|
||||
|
||||
draft_token_ids = self.drafter.propose(
|
||||
@@ -5483,8 +5477,8 @@ class GPUModelRunner(
|
||||
continue
|
||||
|
||||
num_prompt_tokens = len(request.prompt_token_ids)
|
||||
prompt_token_ids = torch.tensor(request.prompt_token_ids).to(
|
||||
self.device, non_blocking=True
|
||||
prompt_token_ids = async_tensor_h2d(
|
||||
request.prompt_token_ids, device=self.device
|
||||
)
|
||||
|
||||
# Set up target LogprobsTensors object.
|
||||
@@ -5651,7 +5645,7 @@ class GPUModelRunner(
|
||||
for _, _, mm_kwargs_batch in group_and_batch_mm_kwargs(
|
||||
[(modality, dummy_mm_item)] * max_items_per_batch,
|
||||
device=self.device,
|
||||
pin_memory=self.pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -6995,7 +6989,6 @@ class GPUModelRunner(
|
||||
max_model_len=max_model_len,
|
||||
max_num_batched_tokens=self.max_num_tokens,
|
||||
device=self.device,
|
||||
pin_memory=self.pin_memory,
|
||||
vocab_size=self.model_config.get_vocab_size(),
|
||||
block_sizes=block_sizes,
|
||||
kernel_block_sizes=kernel_block_sizes,
|
||||
@@ -7430,7 +7423,7 @@ class GPUModelRunner(
|
||||
self.routed_experts_capturer.device_buffer.shape,
|
||||
dtype=self.routed_experts_capturer.device_buffer.dtype,
|
||||
device="cpu",
|
||||
pin_memory=self.pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
# ``slot_mapping`` dtype is fixed to int64 by
|
||||
# ``block_table.slot_mapping``; we mirror that here.
|
||||
@@ -7439,7 +7432,7 @@ class GPUModelRunner(
|
||||
(max_tokens,),
|
||||
dtype=torch.int64,
|
||||
device="cpu",
|
||||
pin_memory=self.pin_memory,
|
||||
pin_memory=PIN_MEMORY,
|
||||
)
|
||||
# Private device buffer so the shared ``block_table.slot_mapping``
|
||||
# can be overwritten by the next ``_prepare_inputs`` while the
|
||||
|
||||
Reference in New Issue
Block a user