forked from Karylab-cklius/vllm
Compare commits
4
Commits
v0.18.0rc1
...
v0.18.0
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bcf2be9612 | ||
|
|
89138b21cc | ||
|
|
6edd43de3c | ||
|
|
16c971dbc7 |
@@ -45,6 +45,22 @@ steps:
|
||||
commands:
|
||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-blackwell.txt
|
||||
|
||||
- label: LM Eval Qwen3.5 Models (B200)
|
||||
timeout_in_minutes: 120
|
||||
device: b200
|
||||
optional: true
|
||||
num_devices: 2
|
||||
source_file_dependencies:
|
||||
- vllm/model_executor/models/qwen3_5.py
|
||||
- vllm/model_executor/models/qwen3_5_mtp.py
|
||||
- vllm/transformers_utils/configs/qwen3_5.py
|
||||
- vllm/transformers_utils/configs/qwen3_5_moe.py
|
||||
- vllm/model_executor/models/qwen3_next.py
|
||||
- vllm/model_executor/models/qwen3_next_mtp.py
|
||||
- vllm/model_executor/layers/fla/ops/
|
||||
commands:
|
||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-qwen35-blackwell.txt
|
||||
|
||||
- label: LM Eval Large Models (H200)
|
||||
timeout_in_minutes: 60
|
||||
device: h200
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
model_name: "Qwen/Qwen3.5-35B-A3B"
|
||||
accuracy_threshold: 0.86
|
||||
num_questions: 1319
|
||||
num_fewshot: 5
|
||||
server_args: >-
|
||||
--max-model-len 4096
|
||||
--data-parallel-size 2
|
||||
--enable-expert-parallel
|
||||
@@ -0,0 +1,9 @@
|
||||
model_name: "Qwen/Qwen3.5-35B-A3B-FP8"
|
||||
accuracy_threshold: 0.86
|
||||
num_questions: 1319
|
||||
num_fewshot: 5
|
||||
server_args: >-
|
||||
--max-model-len 4096
|
||||
--data-parallel-size 2
|
||||
--enable-expert-parallel
|
||||
--kv-cache-dtype fp8
|
||||
@@ -0,0 +1,2 @@
|
||||
Qwen3.5-35B-A3B-DEP2.yaml
|
||||
Qwen3.5-35B-A3B-FP8-DEP2.yaml
|
||||
@@ -777,6 +777,7 @@ VLM_TEST_SETTINGS = {
|
||||
max_model_len=8192,
|
||||
max_num_seqs=2,
|
||||
auto_cls=AutoModelForCausalLM,
|
||||
patch_hf_runner=model_utils.paddleocr_vl_patch_hf_runner,
|
||||
image_size_factors=[(0.25,)],
|
||||
marks=[
|
||||
pytest.mark.skipif(
|
||||
|
||||
@@ -1149,6 +1149,31 @@ def ovis2_5_patch_hf_runner(hf_model: HfRunner) -> HfRunner:
|
||||
return hf_model
|
||||
|
||||
|
||||
def paddleocr_vl_patch_hf_runner(hf_model: HfRunner) -> HfRunner:
|
||||
"""Patches the HfRunner to fix create_causal_mask API mismatch.
|
||||
|
||||
The PaddleOCR-VL HF model passes `inputs_embeds` to create_causal_mask,
|
||||
but transformers renamed this parameter to `input_embeds`.
|
||||
"""
|
||||
import sys
|
||||
|
||||
model_module = sys.modules.get(type(hf_model.model.model).__module__)
|
||||
if model_module is None:
|
||||
return hf_model
|
||||
|
||||
original_create_causal_mask = getattr(model_module, "create_causal_mask", None)
|
||||
if original_create_causal_mask is None:
|
||||
return hf_model
|
||||
|
||||
def patched_create_causal_mask(*args, **kwargs):
|
||||
if "inputs_embeds" in kwargs:
|
||||
kwargs["input_embeds"] = kwargs.pop("inputs_embeds")
|
||||
return original_create_causal_mask(*args, **kwargs)
|
||||
|
||||
model_module.create_causal_mask = patched_create_causal_mask # type: ignore[attr-defined]
|
||||
return hf_model
|
||||
|
||||
|
||||
def qwen2_5_omni_patch_hf_runner(hf_model: HfRunner) -> HfRunner:
|
||||
"""Patches and returns an instance of the HfRunner for Qwen2.5-Omni."""
|
||||
thinker = hf_model.model.thinker
|
||||
|
||||
@@ -253,23 +253,25 @@ class TrtLlmFp8ExpertsMonolithic(TrtLlmFp8ExpertsBase, mk.FusedMoEExpertsMonolit
|
||||
weight_key: QuantKey | None,
|
||||
activation_key: QuantKey | None,
|
||||
) -> bool:
|
||||
"""Monolithic kernels need to express router support."""
|
||||
"""Monolithic kernels need to express router support.
|
||||
Renormalize/RenormalizeNaive are excluded: the monolithic kernel's
|
||||
internal routing for these methods produces output uncorrelated
|
||||
with the modular kernel's output and with Triton kernel's output
|
||||
for Qwen3.5-35B-A3B-FP8.
|
||||
See: https://github.com/vllm-project/vllm/issues/37591
|
||||
"""
|
||||
# NOTE(dbari): TopK routing could also be enabled, but need to validate models
|
||||
# NOTE(dbari): Default is not implemented and should not be enabled until it is
|
||||
if (weight_key, activation_key) == (kFp8Static128BlockSym, kFp8Dynamic128Sym):
|
||||
# NOTE(rob): potentially allow others here. This is a conservative list.
|
||||
return routing_method in [
|
||||
RoutingMethodType.DeepSeekV3,
|
||||
RoutingMethodType.Renormalize,
|
||||
RoutingMethodType.RenormalizeNaive,
|
||||
]
|
||||
elif (weight_key, activation_key) == (kFp8StaticTensorSym, kFp8StaticTensorSym):
|
||||
# NOTE(dbari): as above, potentially allow others here.
|
||||
return routing_method in [
|
||||
RoutingMethodType.DeepSeekV3,
|
||||
RoutingMethodType.Llama4,
|
||||
RoutingMethodType.Renormalize,
|
||||
RoutingMethodType.RenormalizeNaive,
|
||||
]
|
||||
else:
|
||||
raise ValueError("Unsupported quantization scheme.")
|
||||
|
||||
@@ -162,6 +162,11 @@ class CutlassMLAImpl(MLACommonImpl[MLACommonMetadata]):
|
||||
# Share workspace buffer across all executions
|
||||
self._workspace = g_sm100_workspace
|
||||
|
||||
# Pre-allocated output buffer, lazily sized on first call.
|
||||
# Zero-init once to prevent NaN in padding slots (seq_lens=0)
|
||||
# from contaminating downstream per-tensor reductions.
|
||||
self._decode_out: torch.Tensor | None = None
|
||||
|
||||
def _sm100_cutlass_mla_decode(
|
||||
self,
|
||||
q_nope: torch.Tensor,
|
||||
@@ -218,7 +223,15 @@ class CutlassMLAImpl(MLACommonImpl[MLACommonMetadata]):
|
||||
if is_quantized_kv_cache(self.kv_cache_dtype)
|
||||
else q_nope.dtype
|
||||
)
|
||||
out = q_nope.new_empty((B_q, MAX_HEADS, D_latent), dtype=dtype)
|
||||
# Reuse pre-allocated zero-init output buffer to avoid a memset
|
||||
# kernel on every CUDA graph replay.
|
||||
if (
|
||||
self._decode_out is None
|
||||
or self._decode_out.shape[0] < B_q
|
||||
or self._decode_out.dtype != dtype
|
||||
):
|
||||
self._decode_out = q_nope.new_zeros((B_q, MAX_HEADS, D_latent), dtype=dtype)
|
||||
out = self._decode_out[:B_q]
|
||||
lse = (
|
||||
torch.empty((B_q, MAX_HEADS), dtype=torch.float32, device=q_nope.device)
|
||||
if self.need_to_return_lse_for_decode
|
||||
|
||||
@@ -21,6 +21,7 @@ from vllm.v1.attention.backend import (
|
||||
AttentionLayer,
|
||||
AttentionType,
|
||||
MultipleOf,
|
||||
is_quantized_kv_cache,
|
||||
)
|
||||
from vllm.v1.attention.backends.utils import KVCacheLayoutType
|
||||
|
||||
@@ -151,6 +152,11 @@ class FlashInferMLAImpl(MLACommonImpl[MLACommonMetadata]):
|
||||
self.bmm1_scale: float | None = None
|
||||
self.bmm2_scale: float | None = None
|
||||
|
||||
# Pre-allocated output buffer, lazily sized on first call.
|
||||
# Zero-init once to prevent NaN in padding slots (seq_lens=0)
|
||||
# from contaminating downstream per-tensor reductions.
|
||||
self._decode_out: torch.Tensor | None = None
|
||||
|
||||
def forward_mqa(
|
||||
self,
|
||||
q: torch.Tensor | tuple[torch.Tensor, torch.Tensor],
|
||||
@@ -181,6 +187,37 @@ class FlashInferMLAImpl(MLACommonImpl[MLACommonMetadata]):
|
||||
if self.bmm2_scale is None:
|
||||
self.bmm2_scale = layer._v_scale_float
|
||||
|
||||
# Reuse pre-allocated zero-init output buffer to avoid a memset
|
||||
# kernel on every CUDA graph replay.
|
||||
# q is 4D: (batch, q_len_per_req, num_heads, head_dim)
|
||||
# FlashInfer has a bug where out= validation hardcodes 3D shape
|
||||
# (batch, num_heads, kv_lora_rank), but the kernel writes 4D
|
||||
# (batch, q_len, num_heads, kv_lora_rank) when q_len > 1.
|
||||
# So we can only pass out= for single-token decode (q_len == 1).
|
||||
# For q_len > 1, we zero padding slots after the kernel returns.
|
||||
# TODO: upstream fix to FlashInfer
|
||||
B, q_len_per_req = q.shape[0], q.shape[1]
|
||||
out_kwargs: dict[str, torch.Tensor] = {}
|
||||
if q_len_per_req == 1:
|
||||
dtype = (
|
||||
torch.bfloat16
|
||||
if is_quantized_kv_cache(self.kv_cache_dtype)
|
||||
else q.dtype
|
||||
)
|
||||
if (
|
||||
self._decode_out is None
|
||||
or self._decode_out.shape[0] < B
|
||||
or self._decode_out.dtype != dtype
|
||||
):
|
||||
self._decode_out = torch.zeros(
|
||||
B,
|
||||
q.shape[2],
|
||||
self.kv_lora_rank,
|
||||
dtype=dtype,
|
||||
device=q.device,
|
||||
)
|
||||
out_kwargs["out"] = self._decode_out[:B]
|
||||
|
||||
o = trtllm_batch_decode_with_kv_cache_mla(
|
||||
query=q,
|
||||
kv_cache=kv_c_and_k_pe_cache.unsqueeze(1),
|
||||
@@ -193,8 +230,15 @@ class FlashInferMLAImpl(MLACommonImpl[MLACommonMetadata]):
|
||||
max_seq_len=attn_metadata.max_seq_len,
|
||||
bmm1_scale=self.bmm1_scale,
|
||||
bmm2_scale=self.bmm2_scale,
|
||||
**out_kwargs,
|
||||
)
|
||||
|
||||
# For q_len > 1, we can't pass out= so we work around by zeroing padding slots
|
||||
if not out_kwargs:
|
||||
num_real = attn_metadata.num_decodes
|
||||
if num_real < o.shape[0]:
|
||||
o[num_real:] = 0
|
||||
|
||||
# Flatten the output for consistent shape
|
||||
o = o.view(-1, o.shape[-2], o.shape[-1])
|
||||
|
||||
|
||||
@@ -392,8 +392,10 @@ class Worker(WorkerBase):
|
||||
)
|
||||
|
||||
# Profile CUDA graph memory if graphs will be captured.
|
||||
# Skip on ROCm/HIP as graph pool handles and mem_get_info behave
|
||||
# differently and can produce incorrect/negative estimates.
|
||||
cudagraph_memory_estimate = 0
|
||||
if not self.model_config.enforce_eager:
|
||||
if not self.model_config.enforce_eager and not current_platform.is_rocm():
|
||||
cudagraph_memory_estimate = self.model_runner.profile_cudagraph_memory()
|
||||
|
||||
# Use the pre-cudagraph torch peak to avoid double-counting.
|
||||
@@ -406,6 +408,8 @@ class Worker(WorkerBase):
|
||||
+ profile_result.weights_memory
|
||||
)
|
||||
|
||||
# On ROCm, cudagraph_memory_estimate is always 0 so this is a no-op.
|
||||
# On CUDA, respect the opt-in flag as originally designed.
|
||||
cudagraph_memory_estimate_applied = (
|
||||
cudagraph_memory_estimate
|
||||
if envs.VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS
|
||||
@@ -517,7 +521,6 @@ class Worker(WorkerBase):
|
||||
|
||||
def update_max_model_len(self, max_model_len: int) -> None:
|
||||
"""Update max_model_len after auto-fit to GPU memory.
|
||||
|
||||
This is called when max_model_len=-1 is used and the engine
|
||||
automatically determines the maximum context length that fits
|
||||
in GPU memory. Workers need to update their cached max_model_len
|
||||
|
||||
Reference in New Issue
Block a user