diff --git a/vllm/envs.py b/vllm/envs.py index 1f4c1b29006..806aed2a041 100755 --- a/vllm/envs.py +++ b/vllm/envs.py @@ -245,7 +245,6 @@ if TYPE_CHECKING: VLLM_DEBUG_WORKSPACE: bool = False VLLM_DISABLE_SHARED_EXPERTS_STREAM: bool = False VLLM_SHARED_EXPERTS_STREAM_TOKEN_THRESHOLD: int = 256 - VLLM_ENABLE_MULTI_STREAM_GEMM: bool = False VLLM_COMPILE_CACHE_SAVE_FORMAT: Literal["binary", "unpacked"] = "binary" VLLM_USE_V2_MODEL_RUNNER: bool = False VLLM_LOG_MODEL_INSPECTION: bool = False @@ -1663,13 +1662,6 @@ environment_variables: dict[str, Callable[[], Any]] = { "VLLM_SHARED_EXPERTS_STREAM_TOKEN_THRESHOLD": lambda: int( int(os.getenv("VLLM_SHARED_EXPERTS_STREAM_TOKEN_THRESHOLD", 256)) ), - # Enables multi-stream overlap of the attention input GEMM with auxiliary - # GEMMs (e.g. fused_wqa_wkv overlapped with indexer weights / kv-score - # projections in DeepSeek-V4). When unset, those callables run - # sequentially on the current stream even if aux streams are provided. - "VLLM_ENABLE_MULTI_STREAM_GEMM": lambda: bool( - int(os.getenv("VLLM_ENABLE_MULTI_STREAM_GEMM", "0")) - ), # Format for saving torch.compile cache artifacts # - "binary": saves as binary file # Safe for multiple vllm serve processes accessing the same torch compile cache. diff --git a/vllm/model_executor/layers/deepseek_v4_attention.py b/vllm/model_executor/layers/deepseek_v4_attention.py index 21f5e0e2fb5..a968a06bb65 100644 --- a/vllm/model_executor/layers/deepseek_v4_attention.py +++ b/vllm/model_executor/layers/deepseek_v4_attention.py @@ -13,7 +13,6 @@ import torch.nn as nn import torch.nn.functional as F from transformers import DeepseekV2Config, DeepseekV3Config -import vllm.envs as envs from vllm.model_executor.layers.linear import ( ReplicatedLinear, ) @@ -383,7 +382,6 @@ class DeepseekV4MultiHeadLatentAttentionWrapper(PluggableLayer): self.ln_events[0], self.ln_events[1:4], self.aux_stream_list[:3], - enable=envs.VLLM_ENABLE_MULTI_STREAM_GEMM, ) return qr_kv, kv_score, indexer_kv_score, indexer_weights diff --git a/vllm/utils/multi_stream_utils.py b/vllm/utils/multi_stream_utils.py index 2203221c5a1..c00f08f9332 100644 --- a/vllm/utils/multi_stream_utils.py +++ b/vllm/utils/multi_stream_utils.py @@ -64,7 +64,6 @@ def execute_in_parallel( start_event: torch.cuda.Event, done_events: list[torch.cuda.Event], aux_streams: list[torch.cuda.Stream] | None = None, - enable: bool = False, ) -> tuple[Any, list[Any]]: """Run default_fn on the current stream and aux_fns concurrently on aux_streams. @@ -75,9 +74,8 @@ def execute_in_parallel( start_event fans out from the current stream to every launched aux stream; done_events[i] is recorded after aux_fns[i] so the current stream joins - before returning. Falls back to sequential execution on the current stream - when aux_streams is None or enable is False; in that case default_fn runs - first, then aux_fns in order. + before returning. When aux_streams is None, all aux_fns run sequentially + on the current stream. Args: default_fn: Callable for the default (current) stream. @@ -88,17 +86,13 @@ def execute_in_parallel( corresponding aux_fn. Length must match aux_fns. aux_streams: Per-aux CUDA streams. Length must match aux_fns. Multi-stream is disabled when None. - enable: Opt-in switch for the multi-stream path. Defaults to False, - so callers that pass aux_streams must also pass enable=True - (typically gated by an env var) to actually overlap. When False, - execution falls back to sequential on the current stream. Returns: Tuple of (default_result, aux_results) where aux_results[i] is the result of aux_fns[i] (or None when skipped). """ aux_results: list[Any] - if aux_streams is None or not enable: + if aux_streams is None: default_result = default_fn() aux_results = [fn() if fn is not None else None for fn in aux_fns] return default_result, aux_results