forked from Karylab-cklius/vllm
Remove grok model arch from vllm (#46706)
Signed-off-by: Xianbao QIAN <xianbao.qian@gmail.com>
This commit is contained in:
@@ -297,7 +297,7 @@ The `fastokens` Python package (>= 0.2.0) must be installed; if it isn't,
|
||||
vLLM raises a clear `ImportError` at tokenizer load. The override applies to
|
||||
any `--tokenizer-mode` that ends up loading an HF fast tokenizer (`hf`,
|
||||
`deepseek_v32`, `deepseek_v4`, …). Models that don't use the HF
|
||||
fast tokenizer (`mistral`, `grok2`, `kimi_audio`) ignore the flag.
|
||||
fast tokenizer (`mistral`, `kimi_audio`) ignore the flag.
|
||||
|
||||
Tokenizer-bound workloads — long shared prefixes, bursty short prompts,
|
||||
batch detokenization — see the largest wins. If your bottleneck is GPU
|
||||
|
||||
@@ -414,8 +414,6 @@ th {
|
||||
| `GraniteMoeHybridForCausalLM` | Granite 4.0 MoE Hybrid | `ibm-granite/granite-4.0-tiny-preview`, etc. | ✅︎ | ✅︎ |
|
||||
| `GraniteMoeSharedForCausalLM` | Granite MoE Shared | `ibm-research/moe-7b-1b-active-shared-experts` (test model) | ✅︎ | ✅︎ |
|
||||
| `GritLM` | GritLM | `parasail-ai/GritLM-7B-vllm`. | ✅︎ | ✅︎ |
|
||||
| `Grok1ModelForCausalLM` | Grok1 | `hpcai-tech/grok-1`. | ✅︎ | ✅︎ |
|
||||
| `Grok1ForCausalLM` | Grok2 | `xai-org/grok-2` | ✅︎ | ✅︎ |
|
||||
| `HrmTextForCausalLM` | HRM-Text | `sapientinc/HRM-Text-1B`, etc. | | |
|
||||
| `HunYuanDenseV1ForCausalLM` | Hunyuan Dense | `tencent/Hunyuan-7B-Instruct` | ✅︎ | ✅︎ |
|
||||
| `HunYuanMoEV1ForCausalLM` | Hunyuan-A13B | `tencent/Hunyuan-A13B-Instruct`, `tencent/Hunyuan-A13B-Pretrain`, `tencent/Hunyuan-A13B-Instruct-FP8`, etc. | ✅︎ | ✅︎ |
|
||||
@@ -488,9 +486,6 @@ th {
|
||||
| `TeleFLMForCausalLM` | TeleFLM | `CofeAI/FLM-2-52B-Instruct-2407`, `CofeAI/Tele-FLM`, etc. | ✅︎ | ✅︎ |
|
||||
| `Zamba2ForCausalLM` | Zamba2 | `Zyphra/Zamba2-7B-instruct`, `Zyphra/Zamba2-2.7B-instruct`, `Zyphra/Zamba2-1.2B-instruct`, etc. | | |
|
||||
|
||||
!!! note
|
||||
Grok2 requires `tokenizer.tok.json` with `tiktoken` installed. You can optionally override MoE router renormalization with `moe_router_renormalize`.
|
||||
|
||||
Some models are supported only via the [Transformers modeling backend](#transformers). The purpose of the table below is to acknowledge models which we officially support in this way. The logs will say that the Transformers modeling backend is being used, and you will see no warning that this is fallback behaviour. This means that, if you have issues with any of the models listed below, please [make an issue](https://github.com/vllm-project/vllm/issues/new/choose) and we'll do our best to fix it!
|
||||
|
||||
| Architecture | Models | Example HF Models | [LoRA](../features/lora.md) | [PP](../serving/parallelism_scaling.md) |
|
||||
|
||||
@@ -1,43 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
import pytest
|
||||
|
||||
from ...utils import dummy_hf_overrides
|
||||
|
||||
MODELS = ["xai-org/grok-2"]
|
||||
|
||||
|
||||
def _grok2_dummy_overrides(hf_config):
|
||||
hf_config = dummy_hf_overrides(hf_config, model_arch="Grok1ForCausalLM")
|
||||
text_config = hf_config.get_text_config()
|
||||
text_config.update(
|
||||
{
|
||||
"hidden_size": 256,
|
||||
"intermediate_size": 512,
|
||||
"moe_intermediate_size": 256,
|
||||
"num_attention_heads": 4,
|
||||
"num_key_value_heads": 2,
|
||||
"head_dim": 64,
|
||||
}
|
||||
)
|
||||
return hf_config
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", MODELS)
|
||||
def test_dummy_generate(vllm_runner, monkeypatch, model: str) -> None:
|
||||
with monkeypatch.context() as m:
|
||||
m.setenv("VLLM_ALLOW_INSECURE_SERIALIZATION", "1")
|
||||
with vllm_runner(
|
||||
model,
|
||||
load_format="dummy",
|
||||
max_model_len=128,
|
||||
hf_overrides=_grok2_dummy_overrides,
|
||||
enforce_eager=True,
|
||||
) as llm:
|
||||
prompt = "Hello from Grok-2"
|
||||
tokenizer = llm.get_llm().get_tokenizer()
|
||||
prompt_len = len(tokenizer.encode(prompt))
|
||||
outputs = llm.generate_greedy([prompt], max_tokens=1)
|
||||
output_ids, output_str = outputs[0]
|
||||
assert len(output_ids) > prompt_len
|
||||
assert output_str is not None
|
||||
@@ -319,10 +319,6 @@ _TEXT_GENERATION_EXAMPLE_MODELS = {
|
||||
"GraniteMoeSharedForCausalLM": _HfExamplesInfo(
|
||||
"ibm-research/moe-7b-1b-active-shared-experts"
|
||||
),
|
||||
"Grok1ModelForCausalLM": _HfExamplesInfo(
|
||||
"hpcai-tech/grok-1", trust_remote_code=True
|
||||
),
|
||||
"Grok1ForCausalLM": _HfExamplesInfo("xai-org/grok-2", trust_remote_code=True),
|
||||
"HrmTextForCausalLM": _HfExamplesInfo(
|
||||
"sapientinc/HRM-Text-1B",
|
||||
min_transformers_version="5.9.0",
|
||||
|
||||
@@ -9,7 +9,6 @@ from transformers import (
|
||||
)
|
||||
|
||||
from vllm.tokenizers import TokenizerLike, get_tokenizer
|
||||
from vllm.tokenizers.grok2 import Grok2Tokenizer
|
||||
from vllm.tokenizers.hf import HfTokenizer
|
||||
from vllm.tokenizers.mistral import MistralTokenizer
|
||||
|
||||
@@ -35,10 +34,6 @@ def test_tokenizer_like_protocol():
|
||||
assert isinstance(tokenizer, MistralTokenizer)
|
||||
_assert_tokenizer_like(tokenizer)
|
||||
|
||||
tokenizer = get_tokenizer("xai-org/grok-2", tokenizer_mode="grok2")
|
||||
assert isinstance(tokenizer, Grok2Tokenizer)
|
||||
_assert_tokenizer_like(tokenizer)
|
||||
|
||||
tokenizer = get_tokenizer("deepseek-ai/DeepSeek-V3", tokenizer_mode="deepseek_v32")
|
||||
assert isinstance(tokenizer, HfTokenizer)
|
||||
|
||||
|
||||
@@ -601,8 +601,6 @@ class ModelConfig:
|
||||
if self.tokenizer_mode == "auto":
|
||||
if self.model_impl == "terratorch":
|
||||
self.tokenizer_mode = "terratorch"
|
||||
elif arch == "Grok1ForCausalLM":
|
||||
self.tokenizer_mode = "grok2"
|
||||
elif arch == "MoonshotKimiaForCausalLM":
|
||||
self.tokenizer_mode = "kimi_audio"
|
||||
elif arch == "DeepseekV32ForCausalLM":
|
||||
|
||||
@@ -1,792 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
# Adapted from
|
||||
# https://github.com/ROCm/vllm/blob/cea7419f151cc50293a05b7fac8547f8f887c9f6/vllm/model_executor/models/grok1.py
|
||||
# Copyright 2023 The vLLM team.
|
||||
# Copyright 2022 EleutherAI and the HuggingFace Inc. team. All rights reserved.
|
||||
#
|
||||
# This code is based on EleutherAI's GPT-NeoX library and the GPT-NeoX
|
||||
# and OPT implementations in this library. It has been modified from its
|
||||
# original forms to accommodate minor architectural differences compared
|
||||
# to GPT-NeoX and OPT used by the Meta AI team that trained the model.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""Inference-only Grok (Grok1/Grok2) model."""
|
||||
|
||||
import math
|
||||
from collections.abc import Iterable
|
||||
from itertools import islice
|
||||
from typing import Any
|
||||
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
from torch import nn
|
||||
|
||||
from vllm.compilation.decorators import support_torch_compile
|
||||
from vllm.config import CacheConfig, VllmConfig
|
||||
from vllm.distributed import get_pp_group, get_tensor_model_parallel_world_size
|
||||
from vllm.logger import init_logger
|
||||
from vllm.model_executor.layers.activation import GeluAndMul
|
||||
from vllm.model_executor.layers.attention import Attention
|
||||
from vllm.model_executor.layers.fused_moe import (
|
||||
FusedMoE,
|
||||
fused_moe_make_expert_params_mapping,
|
||||
)
|
||||
from vllm.model_executor.layers.layernorm import RMSNorm
|
||||
from vllm.model_executor.layers.linear import (
|
||||
MergedColumnParallelLinear,
|
||||
QKVParallelLinear,
|
||||
ReplicatedLinear,
|
||||
RowParallelLinear,
|
||||
)
|
||||
from vllm.model_executor.layers.logits_processor import LogitsProcessor
|
||||
from vllm.model_executor.layers.quantization import QuantizationConfig
|
||||
from vllm.model_executor.layers.rotary_embedding import get_rope
|
||||
from vllm.model_executor.layers.vocab_parallel_embedding import (
|
||||
ParallelLMHead,
|
||||
VocabParallelEmbedding,
|
||||
)
|
||||
from vllm.model_executor.model_loader.weight_utils import (
|
||||
default_weight_loader,
|
||||
maybe_remap_kv_scale_name,
|
||||
)
|
||||
from vllm.sequence import IntermediateTensors
|
||||
|
||||
from .interfaces import SupportsLoRA, SupportsPP
|
||||
from .utils import (
|
||||
AutoWeightsLoader,
|
||||
is_pp_missing_parameter,
|
||||
make_empty_intermediate_tensors_factory,
|
||||
make_layers,
|
||||
maybe_prefix,
|
||||
)
|
||||
|
||||
# Default Grok1-specific constants, overridden by config values if present
|
||||
DEFAULT_ATTN_OUTPUT_MULTIPLIER = 0.08838834764831845
|
||||
DEFAULT_OUTPUT_MULTIPLIER_SCALE = 0.5773502691896257
|
||||
DEFAULT_EMBEDDING_MULTIPLIER_SCALE = 78.38367176906169
|
||||
DEFAULT_ROUTER_LOGIT_SOFTCAP = 30.0
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
|
||||
def _get_num_experts(config) -> int:
|
||||
return getattr(config, "num_experts", getattr(config, "num_local_experts", 8))
|
||||
|
||||
|
||||
def _get_moe_intermediate_size(config) -> int:
|
||||
return getattr(config, "moe_intermediate_size", config.intermediate_size)
|
||||
|
||||
|
||||
def _get_grok_version(config) -> str:
|
||||
"""Detect Grok version from HF config using multiple heuristics."""
|
||||
# Check for Grok2-specific attributes (both for robust detection)
|
||||
has_residual_moe = getattr(config, "residual_moe", False)
|
||||
has_moe_intermediate_size = hasattr(config, "moe_intermediate_size")
|
||||
|
||||
if has_residual_moe or has_moe_intermediate_size:
|
||||
return "grok2"
|
||||
|
||||
return "grok1" # Default to Grok1
|
||||
|
||||
|
||||
def _get_rope_parameters(config) -> dict[str, Any] | None:
|
||||
rope_parameters = getattr(config, "rope_parameters", None)
|
||||
if rope_parameters is None:
|
||||
rope_type = getattr(config, "rope_type", None)
|
||||
if rope_type is None:
|
||||
return None
|
||||
rope_parameters = {"rope_type": rope_type}
|
||||
rope_theta = getattr(config, "rope_theta", None)
|
||||
if rope_theta is not None:
|
||||
rope_parameters["rope_theta"] = rope_theta
|
||||
scaling_factor = getattr(config, "scaling_factor", None)
|
||||
if scaling_factor is not None:
|
||||
rope_parameters["factor"] = scaling_factor
|
||||
for name in (
|
||||
"original_max_position_embeddings",
|
||||
"extrapolation_factor",
|
||||
"attn_factor",
|
||||
"beta_fast",
|
||||
"beta_slow",
|
||||
):
|
||||
value = getattr(config, name, None)
|
||||
if value is not None:
|
||||
rope_parameters[name] = value
|
||||
|
||||
if rope_parameters.get("rope_type") == "original":
|
||||
rope_parameters = dict(rope_parameters)
|
||||
rope_parameters["rope_type"] = "default"
|
||||
return rope_parameters
|
||||
|
||||
|
||||
def _get_moe_renormalize(config) -> bool:
|
||||
explicit_value = getattr(
|
||||
config, "moe_router_renormalize", getattr(config, "moe_renormalize", None)
|
||||
)
|
||||
if explicit_value is not None:
|
||||
return bool(explicit_value)
|
||||
return not getattr(config, "residual_moe", False)
|
||||
|
||||
|
||||
class Grok1MLP(nn.Module):
|
||||
def __init__(
|
||||
self,
|
||||
hidden_size: int,
|
||||
intermediate_size: int,
|
||||
quant_config: QuantizationConfig | None = None,
|
||||
prefix: str = "",
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.gate_up_proj = MergedColumnParallelLinear(
|
||||
input_size=hidden_size,
|
||||
output_sizes=[intermediate_size] * 2,
|
||||
bias=False,
|
||||
quant_config=quant_config,
|
||||
prefix=f"{prefix}.gate_up_proj",
|
||||
)
|
||||
self.down_proj = RowParallelLinear(
|
||||
input_size=intermediate_size,
|
||||
output_size=hidden_size,
|
||||
bias=False,
|
||||
quant_config=quant_config,
|
||||
prefix=f"{prefix}.down_proj",
|
||||
)
|
||||
self.act_fn = GeluAndMul()
|
||||
|
||||
def forward(self, x: torch.Tensor) -> torch.Tensor:
|
||||
x, _ = self.gate_up_proj(x)
|
||||
x = self.act_fn(x)
|
||||
x, _ = self.down_proj(x)
|
||||
return x
|
||||
|
||||
|
||||
class Grok1MoE(nn.Module):
|
||||
"""A tensor-parallel MoE implementation for Grok1 that shards each expert
|
||||
across all ranks.
|
||||
|
||||
Each expert's weights are sharded across all ranks and a fused MoE
|
||||
kernel is used for the forward pass, and finally we reduce the outputs
|
||||
across ranks.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
num_experts: int,
|
||||
top_k: int,
|
||||
hidden_size: int,
|
||||
intermediate_size: int,
|
||||
router_logit_soft_cap: float = 0.0,
|
||||
params_dtype: torch.dtype | None = None,
|
||||
quant_config: QuantizationConfig | None = None,
|
||||
tp_size: int | None = None,
|
||||
renormalize: bool = False,
|
||||
prefix: str = "",
|
||||
):
|
||||
super().__init__()
|
||||
self.hidden_size = hidden_size
|
||||
|
||||
# Gate always runs at half / full precision for now.
|
||||
self.gate = ReplicatedLinear(
|
||||
hidden_size,
|
||||
num_experts,
|
||||
bias=False,
|
||||
params_dtype=params_dtype,
|
||||
quant_config=None,
|
||||
prefix=f"{prefix}.gate",
|
||||
)
|
||||
|
||||
self.experts = FusedMoE(
|
||||
num_experts=num_experts,
|
||||
top_k=top_k,
|
||||
hidden_size=hidden_size,
|
||||
intermediate_size=intermediate_size,
|
||||
params_dtype=params_dtype,
|
||||
renormalize=renormalize,
|
||||
quant_config=quant_config,
|
||||
tp_size=tp_size,
|
||||
activation="gelu",
|
||||
prefix=f"{prefix}.experts",
|
||||
)
|
||||
self.router_logit_soft_cap = router_logit_soft_cap
|
||||
|
||||
def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
|
||||
# NOTE: hidden_states can have either 1D or 2D shape.
|
||||
orig_shape = hidden_states.shape
|
||||
hidden_states = hidden_states.view(-1, self.hidden_size)
|
||||
# router_logits: (num_tokens, n_experts)
|
||||
router_logits, _ = self.gate(hidden_states)
|
||||
if self.router_logit_soft_cap > 0:
|
||||
router_logits = self.router_logit_soft_cap * F.tanh(
|
||||
router_logits / self.router_logit_soft_cap
|
||||
)
|
||||
final_hidden_states = self.experts(hidden_states, router_logits)
|
||||
return final_hidden_states.view(orig_shape)
|
||||
|
||||
|
||||
class Grok1Attention(nn.Module):
|
||||
def __init__(
|
||||
self,
|
||||
hidden_size: int,
|
||||
num_heads: int,
|
||||
num_kv_heads: int,
|
||||
max_position: int = 4096 * 32,
|
||||
rope_parameters: dict[str, Any] | None = None,
|
||||
cache_config: CacheConfig | None = None,
|
||||
quant_config: QuantizationConfig | None = None,
|
||||
prefix: str = "",
|
||||
config=None, # Added config parameter
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.hidden_size = hidden_size
|
||||
self.config = config # Store config reference
|
||||
tp_size = get_tensor_model_parallel_world_size()
|
||||
self.total_num_heads = num_heads
|
||||
assert self.total_num_heads % tp_size == 0
|
||||
self.num_heads = self.total_num_heads // tp_size
|
||||
self.total_num_kv_heads = num_kv_heads
|
||||
if self.total_num_kv_heads >= tp_size:
|
||||
# Number of KV heads is greater than TP size, so we partition
|
||||
# the KV heads across multiple tensor parallel GPUs.
|
||||
assert self.total_num_kv_heads % tp_size == 0
|
||||
else:
|
||||
# Number of KV heads is less than TP size, so we replicate
|
||||
# the KV heads across multiple tensor parallel GPUs.
|
||||
assert tp_size % self.total_num_kv_heads == 0
|
||||
self.num_kv_heads = max(1, self.total_num_kv_heads // tp_size)
|
||||
self.head_dim = hidden_size // self.total_num_heads
|
||||
self.q_size = self.num_heads * self.head_dim
|
||||
self.kv_size = self.num_kv_heads * self.head_dim
|
||||
self.scaling = self.head_dim**-0.5
|
||||
|
||||
self.qkv_proj = QKVParallelLinear(
|
||||
hidden_size,
|
||||
self.head_dim,
|
||||
self.total_num_heads,
|
||||
self.total_num_kv_heads,
|
||||
bias=False,
|
||||
quant_config=quant_config,
|
||||
prefix=f"{prefix}.qkv_proj",
|
||||
)
|
||||
self.o_proj = RowParallelLinear(
|
||||
self.total_num_heads * self.head_dim,
|
||||
hidden_size,
|
||||
bias=False,
|
||||
quant_config=quant_config,
|
||||
prefix=f"{prefix}.o_proj",
|
||||
)
|
||||
self.rotary_emb = get_rope(
|
||||
self.head_dim,
|
||||
max_position=max_position,
|
||||
rope_parameters=rope_parameters,
|
||||
is_neox_style=True,
|
||||
)
|
||||
|
||||
attn_logits_soft_cap = max(getattr(config, "attn_logit_softcapping", 30.0), 0.0)
|
||||
attn_logit_softcapping_method = getattr(
|
||||
config, "attn_logit_softcapping_method", None
|
||||
)
|
||||
if attn_logit_softcapping_method not in (None, "tanh"):
|
||||
logger.warning_once(
|
||||
"Grok attention logit softcapping method '%s' is not "
|
||||
"supported; falling back to default behavior.",
|
||||
attn_logit_softcapping_method,
|
||||
)
|
||||
|
||||
self.attn = Attention(
|
||||
self.num_heads,
|
||||
self.head_dim,
|
||||
self.scaling,
|
||||
num_kv_heads=self.num_kv_heads,
|
||||
cache_config=cache_config,
|
||||
quant_config=quant_config,
|
||||
logits_soft_cap=attn_logits_soft_cap,
|
||||
prefix=f"{prefix}.attn",
|
||||
)
|
||||
self.attn_multiplier = (
|
||||
getattr(self.config, "attn_output_multiplier", 1.0) if self.config else 1.0
|
||||
)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
positions: torch.Tensor,
|
||||
hidden_states: torch.Tensor,
|
||||
) -> torch.Tensor:
|
||||
qkv, _ = self.qkv_proj(hidden_states)
|
||||
q, k, v = qkv.split([self.q_size, self.kv_size, self.kv_size], dim=-1)
|
||||
q, k = self.rotary_emb(positions, q, k)
|
||||
attn_output = self.attn(q, k, v)
|
||||
output, _ = self.o_proj(attn_output)
|
||||
output *= self.attn_multiplier
|
||||
return output
|
||||
|
||||
|
||||
class Grok1DecoderLayer(nn.Module):
|
||||
def __init__(
|
||||
self,
|
||||
config,
|
||||
cache_config: CacheConfig | None = None,
|
||||
quant_config: QuantizationConfig | None = None,
|
||||
prefix: str = "",
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.hidden_size = config.hidden_size
|
||||
# Check for fp8 quantization
|
||||
self.use_fp8 = False
|
||||
if quant_config is not None:
|
||||
self.use_fp8 = getattr(quant_config, "is_fp8_w8a8", lambda: False)()
|
||||
if not self.use_fp8 and hasattr(quant_config, "is_fp8"):
|
||||
self.use_fp8 = quant_config.is_fp8
|
||||
|
||||
self.attn = Grok1Attention(
|
||||
hidden_size=self.hidden_size,
|
||||
num_heads=config.num_attention_heads,
|
||||
max_position=config.max_position_embeddings,
|
||||
num_kv_heads=config.num_key_value_heads,
|
||||
rope_parameters=_get_rope_parameters(config),
|
||||
cache_config=cache_config,
|
||||
quant_config=quant_config,
|
||||
prefix=f"{prefix}.attn",
|
||||
config=config,
|
||||
) # Pass config to Grok1Attention
|
||||
|
||||
num_experts = _get_num_experts(config)
|
||||
num_experts_per_tok = getattr(config, "num_experts_per_tok", 2)
|
||||
moe_intermediate_size = _get_moe_intermediate_size(config)
|
||||
moe_renormalize = _get_moe_renormalize(config)
|
||||
|
||||
self.moe_block = Grok1MoE(
|
||||
num_experts=num_experts,
|
||||
top_k=num_experts_per_tok,
|
||||
hidden_size=config.hidden_size,
|
||||
intermediate_size=moe_intermediate_size,
|
||||
router_logit_soft_cap=max(
|
||||
getattr(
|
||||
config,
|
||||
"router_logit_softcapping",
|
||||
DEFAULT_ROUTER_LOGIT_SOFTCAP,
|
||||
),
|
||||
0.0,
|
||||
),
|
||||
quant_config=quant_config,
|
||||
renormalize=moe_renormalize,
|
||||
prefix=f"{prefix}.moe_block",
|
||||
)
|
||||
self.residual_moe = getattr(config, "residual_moe", False)
|
||||
self.residual_moe_scale = 1.0 / math.sqrt(2.0)
|
||||
|
||||
self.pre_attn_norm = RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
||||
self.post_attn_norm = RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
||||
self.pre_moe_norm = RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
||||
self.post_moe_norm = RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
||||
self.mlp = None
|
||||
if self.residual_moe:
|
||||
self.mlp = Grok1MLP(
|
||||
hidden_size=config.hidden_size,
|
||||
intermediate_size=config.intermediate_size,
|
||||
quant_config=quant_config,
|
||||
prefix=f"{prefix}.mlp",
|
||||
)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
positions: torch.Tensor,
|
||||
hidden_states: torch.Tensor,
|
||||
residual: torch.Tensor | None,
|
||||
) -> tuple[torch.Tensor, torch.Tensor]:
|
||||
# Self Attention
|
||||
if residual is None:
|
||||
residual = hidden_states
|
||||
hidden_states = self.pre_attn_norm(hidden_states)
|
||||
else:
|
||||
hidden_states, residual = self.pre_attn_norm(hidden_states, residual)
|
||||
|
||||
hidden_states = self.attn(
|
||||
positions=positions,
|
||||
hidden_states=hidden_states,
|
||||
)
|
||||
|
||||
# Post attention normalization
|
||||
hidden_states = self.post_attn_norm(hidden_states)
|
||||
|
||||
# MoE block with normalization
|
||||
hidden_states, residual = self.pre_moe_norm(hidden_states, residual)
|
||||
if self.residual_moe:
|
||||
assert self.mlp is not None
|
||||
hidden_states = (
|
||||
self.moe_block(hidden_states) + self.mlp(hidden_states)
|
||||
) * self.residual_moe_scale
|
||||
else:
|
||||
hidden_states = self.moe_block(hidden_states)
|
||||
hidden_states = self.post_moe_norm(hidden_states)
|
||||
|
||||
return hidden_states, residual
|
||||
|
||||
|
||||
@support_torch_compile
|
||||
class Grok1Model(nn.Module):
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
vllm_config: VllmConfig,
|
||||
prefix: str = "",
|
||||
ckpt_gate_proj_name: str = "linear",
|
||||
ckpt_down_proj_name: str = "linear_1",
|
||||
ckpt_up_proj_name: str = "linear_v",
|
||||
weight_name_remapping: dict[str, str] | None = None,
|
||||
):
|
||||
super().__init__()
|
||||
|
||||
config = vllm_config.model_config.hf_config
|
||||
cache_config = vllm_config.cache_config
|
||||
quant_config = vllm_config.quant_config
|
||||
|
||||
self.config = config
|
||||
self.quant_config = quant_config
|
||||
|
||||
# Store expert naming for weight loading
|
||||
self.ckpt_gate_proj_name = ckpt_gate_proj_name
|
||||
self.ckpt_down_proj_name = ckpt_down_proj_name
|
||||
self.ckpt_up_proj_name = ckpt_up_proj_name
|
||||
self.weight_name_remapping = weight_name_remapping or {}
|
||||
|
||||
self.vocab_size = config.vocab_size
|
||||
|
||||
self.embedding_multiplier_scale = getattr(
|
||||
config, "embedding_multiplier_scale", DEFAULT_EMBEDDING_MULTIPLIER_SCALE
|
||||
)
|
||||
|
||||
self.embed_tokens = VocabParallelEmbedding(
|
||||
self.vocab_size,
|
||||
config.hidden_size,
|
||||
quant_config=quant_config,
|
||||
)
|
||||
|
||||
self.start_layer, self.end_layer, self.layers = make_layers(
|
||||
config.num_hidden_layers,
|
||||
lambda prefix: Grok1DecoderLayer(
|
||||
config, cache_config, quant_config=quant_config, prefix=prefix
|
||||
),
|
||||
prefix=f"{prefix}.layers",
|
||||
)
|
||||
|
||||
self.norm = RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
||||
self.make_empty_intermediate_tensors = make_empty_intermediate_tensors_factory(
|
||||
["hidden_states", "residual"], config.hidden_size
|
||||
)
|
||||
|
||||
def embed_input_ids(self, input_ids: torch.Tensor) -> torch.Tensor:
|
||||
hidden_states = self.embed_tokens(input_ids)
|
||||
hidden_states = hidden_states * self.embedding_multiplier_scale
|
||||
return hidden_states
|
||||
|
||||
def forward(
|
||||
self,
|
||||
input_ids: torch.Tensor | None,
|
||||
positions: torch.Tensor,
|
||||
intermediate_tensors: IntermediateTensors | None,
|
||||
inputs_embeds: torch.Tensor | None = None,
|
||||
) -> torch.Tensor | IntermediateTensors:
|
||||
if get_pp_group().is_first_rank:
|
||||
if inputs_embeds is not None:
|
||||
hidden_states = inputs_embeds
|
||||
else:
|
||||
hidden_states = self.embed_input_ids(input_ids)
|
||||
residual = None
|
||||
else:
|
||||
assert intermediate_tensors is not None
|
||||
hidden_states = intermediate_tensors["hidden_states"]
|
||||
residual = intermediate_tensors["residual"]
|
||||
|
||||
for layer in islice(self.layers, self.start_layer, self.end_layer):
|
||||
hidden_states, residual = layer(positions, hidden_states, residual)
|
||||
|
||||
if not get_pp_group().is_last_rank:
|
||||
return IntermediateTensors(
|
||||
{"hidden_states": hidden_states, "residual": residual}
|
||||
)
|
||||
|
||||
hidden_states, _ = self.norm(hidden_states, residual)
|
||||
return hidden_states
|
||||
|
||||
def get_expert_mapping(self) -> list[tuple[str, str, int, str]]:
|
||||
# Map expert parameter names to standard names
|
||||
num_experts = _get_num_experts(self.config)
|
||||
return fused_moe_make_expert_params_mapping(
|
||||
self,
|
||||
ckpt_gate_proj_name=self.ckpt_gate_proj_name,
|
||||
ckpt_down_proj_name=self.ckpt_down_proj_name,
|
||||
ckpt_up_proj_name=self.ckpt_up_proj_name,
|
||||
num_experts=num_experts,
|
||||
)
|
||||
|
||||
def load_weights(self, weights: Iterable[tuple[str, torch.Tensor]]) -> set[str]:
|
||||
stacked_params_mapping = [
|
||||
# (param_name, shard_name, shard_id)
|
||||
("qkv_proj", "q_proj", "q"),
|
||||
("qkv_proj", "k_proj", "k"),
|
||||
("qkv_proj", "v_proj", "v"),
|
||||
("mlp.gate_up_proj", "mlp.gate_proj", 0),
|
||||
("mlp.gate_up_proj", "mlp.up_proj", 1),
|
||||
]
|
||||
|
||||
params_dict = dict(self.named_parameters())
|
||||
loaded_params: set[str] = set()
|
||||
expert_params_mapping = self.get_expert_mapping()
|
||||
for name, loaded_weight in weights:
|
||||
# Apply version-specific weight name remapping
|
||||
for old_pattern, new_pattern in self.weight_name_remapping.items():
|
||||
if old_pattern in name:
|
||||
name = name.replace(old_pattern, new_pattern)
|
||||
|
||||
for param_name, weight_name, shard_id in stacked_params_mapping:
|
||||
if weight_name not in name:
|
||||
continue
|
||||
name = name.replace(weight_name, param_name)
|
||||
# Skip loading extra bias for GPTQ models.
|
||||
if (
|
||||
name.endswith(".bias") or name.endswith("_bias")
|
||||
) and name not in params_dict:
|
||||
continue
|
||||
# Skip layers on other devices.
|
||||
if is_pp_missing_parameter(name, self):
|
||||
continue
|
||||
if name.endswith("scale"):
|
||||
# Remapping the name of FP8 kv-scale.
|
||||
name = maybe_remap_kv_scale_name(name, params_dict)
|
||||
if name is None:
|
||||
continue
|
||||
if name not in params_dict:
|
||||
continue
|
||||
param = params_dict[name]
|
||||
weight_loader = param.weight_loader
|
||||
weight_loader(param, loaded_weight, shard_id)
|
||||
break
|
||||
else:
|
||||
for mapping in expert_params_mapping:
|
||||
param_name, weight_name, expert_id, shard_id = mapping
|
||||
if weight_name not in name:
|
||||
continue
|
||||
name = name.replace(weight_name, param_name)
|
||||
# Skip layers on other devices.
|
||||
if is_pp_missing_parameter(name, self):
|
||||
continue
|
||||
if (
|
||||
name.endswith(".bias") or name.endswith("_bias")
|
||||
) and name not in params_dict:
|
||||
continue
|
||||
param = params_dict[name]
|
||||
weight_loader = param.weight_loader
|
||||
weight_loader(
|
||||
param,
|
||||
loaded_weight,
|
||||
name,
|
||||
shard_id=shard_id,
|
||||
expert_id=expert_id,
|
||||
)
|
||||
break
|
||||
else:
|
||||
# Skip loading extra bias for GPTQ models.
|
||||
if (
|
||||
name.endswith(".bias") or name.endswith("_bias")
|
||||
) and name not in params_dict:
|
||||
continue
|
||||
# Skip layers on other devices.
|
||||
if is_pp_missing_parameter(name, self):
|
||||
continue
|
||||
|
||||
# Remapping the name of FP8 kv-scale.
|
||||
name = maybe_remap_kv_scale_name(name, params_dict)
|
||||
if name is None:
|
||||
continue
|
||||
|
||||
# Handle Grok1-specific norm.scale naming
|
||||
if "norm.scale" in name:
|
||||
name = name.replace("scale", "weight")
|
||||
|
||||
if name not in params_dict:
|
||||
continue
|
||||
param = params_dict[name]
|
||||
weight_loader = getattr(
|
||||
param, "weight_loader", default_weight_loader
|
||||
)
|
||||
weight_loader(param, loaded_weight)
|
||||
loaded_params.add(name)
|
||||
return loaded_params
|
||||
|
||||
|
||||
class GrokBaseForCausalLM(nn.Module, SupportsLoRA, SupportsPP):
|
||||
"""Base class for Grok models with shared logic."""
|
||||
|
||||
fall_back_to_pt_during_load = False
|
||||
|
||||
# Subclasses should override these
|
||||
packed_modules_mapping = {
|
||||
"qkv_proj": [
|
||||
"q_proj",
|
||||
"k_proj",
|
||||
"v_proj",
|
||||
],
|
||||
}
|
||||
|
||||
# Expert weight naming - subclasses override these
|
||||
ckpt_gate_proj_name: str = "linear"
|
||||
ckpt_down_proj_name: str = "linear_1"
|
||||
ckpt_up_proj_name: str = "linear_v"
|
||||
|
||||
def get_weight_name_remapping(self) -> dict[str, str]:
|
||||
"""Return weight name remapping for this version. Override in subclasses."""
|
||||
return {}
|
||||
|
||||
def __init__(self, *, vllm_config: VllmConfig, prefix: str = ""):
|
||||
super().__init__()
|
||||
|
||||
config = vllm_config.model_config.hf_config
|
||||
quant_config = vllm_config.quant_config
|
||||
|
||||
self.config = config
|
||||
self.quant_config = quant_config
|
||||
|
||||
self.model = Grok1Model(
|
||||
vllm_config=vllm_config,
|
||||
prefix=maybe_prefix(prefix, "model"),
|
||||
ckpt_gate_proj_name=self.ckpt_gate_proj_name,
|
||||
ckpt_down_proj_name=self.ckpt_down_proj_name,
|
||||
ckpt_up_proj_name=self.ckpt_up_proj_name,
|
||||
weight_name_remapping=self.get_weight_name_remapping(),
|
||||
)
|
||||
|
||||
self.lm_head = ParallelLMHead(
|
||||
config.vocab_size,
|
||||
config.hidden_size,
|
||||
quant_config=quant_config,
|
||||
prefix=maybe_prefix(prefix, "lm_head"),
|
||||
)
|
||||
|
||||
if self.config.tie_word_embeddings:
|
||||
self.lm_head.weight = self.model.embed_tokens.weight
|
||||
|
||||
self.output_multiplier_scale = getattr(
|
||||
config, "output_multiplier_scale", DEFAULT_OUTPUT_MULTIPLIER_SCALE
|
||||
)
|
||||
self.logits_processor = LogitsProcessor(
|
||||
config.vocab_size,
|
||||
scale=self.output_multiplier_scale,
|
||||
soft_cap=getattr(config, "final_logit_softcapping", None),
|
||||
)
|
||||
|
||||
self.make_empty_intermediate_tensors = (
|
||||
self.model.make_empty_intermediate_tensors
|
||||
)
|
||||
|
||||
def embed_input_ids(self, input_ids: torch.Tensor) -> torch.Tensor:
|
||||
return self.model.embed_input_ids(input_ids)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
input_ids: torch.Tensor | None,
|
||||
positions: torch.Tensor,
|
||||
intermediate_tensors: IntermediateTensors | None = None,
|
||||
inputs_embeds: torch.Tensor | None = None,
|
||||
) -> torch.Tensor | IntermediateTensors:
|
||||
hidden_states = self.model(
|
||||
input_ids, positions, intermediate_tensors, inputs_embeds
|
||||
)
|
||||
return hidden_states
|
||||
|
||||
def compute_logits(
|
||||
self,
|
||||
hidden_states: torch.Tensor,
|
||||
) -> torch.Tensor | None:
|
||||
logits = self.logits_processor(self.lm_head, hidden_states)
|
||||
return logits
|
||||
|
||||
def load_weights(self, weights: Iterable[tuple[str, torch.Tensor]]) -> set[str]:
|
||||
# Skip lm_head when tie_word_embeddings is True
|
||||
skip_prefixes = ["lm_head"] if self.config.tie_word_embeddings else None
|
||||
|
||||
loader = AutoWeightsLoader(
|
||||
self,
|
||||
skip_prefixes=skip_prefixes,
|
||||
)
|
||||
return loader.load_weights(weights)
|
||||
|
||||
def get_expert_mapping(self) -> list[tuple[str, str, int, str]]:
|
||||
return self.model.get_expert_mapping()
|
||||
|
||||
|
||||
class Grok1ForCausalLM(GrokBaseForCausalLM):
|
||||
"""Grok1-specific implementation."""
|
||||
|
||||
# Grok1 expert weight naming
|
||||
ckpt_gate_proj_name = "linear"
|
||||
ckpt_down_proj_name = "linear_1"
|
||||
ckpt_up_proj_name = "linear_v"
|
||||
|
||||
def get_weight_name_remapping(self) -> dict[str, str]:
|
||||
# Grok1 uses standard naming, no remapping needed
|
||||
return {}
|
||||
|
||||
|
||||
class Grok2ForCausalLM(GrokBaseForCausalLM):
|
||||
"""Grok2-specific implementation."""
|
||||
|
||||
# Grok2 has additional packed modules for MLP
|
||||
packed_modules_mapping = {
|
||||
"qkv_proj": [
|
||||
"q_proj",
|
||||
"k_proj",
|
||||
"v_proj",
|
||||
],
|
||||
"gate_up_proj": [
|
||||
"gate_proj",
|
||||
"up_proj",
|
||||
],
|
||||
}
|
||||
|
||||
# Grok2 expert weight naming
|
||||
ckpt_gate_proj_name = "w1"
|
||||
ckpt_down_proj_name = "w2"
|
||||
ckpt_up_proj_name = "w3"
|
||||
|
||||
def get_weight_name_remapping(self) -> dict[str, str]:
|
||||
# Grok2 checkpoint uses different naming conventions
|
||||
return {
|
||||
".self_attn.": ".attn.",
|
||||
".block_sparse_moe.": ".moe_block.",
|
||||
}
|
||||
|
||||
|
||||
# Version dispatch mapping
|
||||
_GROK_VERSIONS: dict[str, type[GrokBaseForCausalLM]] = {
|
||||
"grok1": Grok1ForCausalLM,
|
||||
"grok2": Grok2ForCausalLM,
|
||||
}
|
||||
|
||||
|
||||
class GrokForCausalLM(GrokBaseForCausalLM):
|
||||
"""Factory class that dispatches to version-specific implementation."""
|
||||
|
||||
def __new__(cls, *, vllm_config: VllmConfig, prefix: str = ""):
|
||||
config = vllm_config.model_config.hf_config
|
||||
version = _get_grok_version(config)
|
||||
|
||||
instance_cls = _GROK_VERSIONS.get(version)
|
||||
if instance_cls is None:
|
||||
raise ValueError(f"Unsupported Grok version: {version}")
|
||||
|
||||
# Merge class attributes for LoRA/quantization compatibility
|
||||
cls.packed_modules_mapping = dict(cls.packed_modules_mapping)
|
||||
cls.packed_modules_mapping.update(instance_cls.packed_modules_mapping)
|
||||
|
||||
return instance_cls(vllm_config=vllm_config, prefix=prefix)
|
||||
@@ -124,8 +124,6 @@ _TEXT_GENERATION_MODELS = {
|
||||
"GraniteMoeHybridForCausalLM": ("granitemoehybrid", "GraniteMoeHybridForCausalLM"),
|
||||
"GraniteMoeSharedForCausalLM": ("granitemoeshared", "GraniteMoeSharedForCausalLM"),
|
||||
"GritLM": ("gritlm", "GritLM"),
|
||||
"Grok1ModelForCausalLM": ("grok1", "GrokForCausalLM"),
|
||||
"Grok1ForCausalLM": ("grok1", "GrokForCausalLM"),
|
||||
"HrmTextForCausalLM": ("hrm_text", "HrmTextForCausalLM"),
|
||||
"HunYuanMoEV1ForCausalLM": ("hunyuan_v1", "HunYuanMoEV1ForCausalLM"),
|
||||
"HunYuanDenseV1ForCausalLM": ("hunyuan_v1", "HunYuanDenseV1ForCausalLM"),
|
||||
@@ -731,6 +729,8 @@ _PREVIOUSLY_SUPPORTED_MODELS = {
|
||||
"BaichuanForCausalLM": "0.23.0",
|
||||
"AquilaModel": "0.24.0",
|
||||
"AquilaForCausalLM": "0.24.0",
|
||||
"Grok1ModelForCausalLM": "0.24.0",
|
||||
"Grok1ForCausalLM": "0.24.0",
|
||||
}
|
||||
|
||||
_OOT_SUPPORTED_MODELS = {
|
||||
|
||||
@@ -180,7 +180,6 @@ class MoEMixin(MixtureOfExperts):
|
||||
# (ckpt_gate_proj_name, ckpt_down_proj_name, ckpt_up_proj_name)
|
||||
("gate_proj", "down_proj", "up_proj"), # Most common MoE style
|
||||
("w1", "w2", "w3"), # Granite, Mixtral, Phi MoE style
|
||||
("linear", "linear_1", "linear_v"), # Grok1 style
|
||||
]
|
||||
num_experts = self.model_config.get_num_experts()
|
||||
num_redundant_experts = self.parallel_config.eplb_config.num_redundant_experts
|
||||
@@ -238,8 +237,6 @@ class MoEMixin(MixtureOfExperts):
|
||||
wrapped_arch = self.config.architectures[0].lower()
|
||||
if "gptoss" in wrapped_arch:
|
||||
activation = "swigluoai"
|
||||
elif "grok1" in wrapped_arch:
|
||||
activation = "gelu"
|
||||
|
||||
# Expert mapping for `AutoWeightsLoader`
|
||||
expert_mapping = self.get_expert_mapping()
|
||||
|
||||
@@ -1,90 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
|
||||
from vllm.config import VllmConfig
|
||||
from vllm.entrypoints.chat_utils import (
|
||||
ChatCompletionMessageParam,
|
||||
ConversationMessage,
|
||||
parse_chat_messages,
|
||||
parse_chat_messages_async,
|
||||
)
|
||||
from vllm.logger import init_logger
|
||||
from vllm.tokenizers.grok2 import Grok2Tokenizer
|
||||
from vllm.utils.async_utils import make_async
|
||||
|
||||
from .base import BaseRenderer
|
||||
from .inputs import DictPrompt
|
||||
from .inputs.preprocess import parse_dec_only_prompt
|
||||
from .params import ChatParams
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
|
||||
class Grok2Renderer(BaseRenderer[Grok2Tokenizer]):
|
||||
def __init__(
|
||||
self,
|
||||
config: VllmConfig,
|
||||
tokenizer: Grok2Tokenizer | None,
|
||||
) -> None:
|
||||
super().__init__(config, tokenizer)
|
||||
|
||||
self._apply_chat_template_async = make_async(
|
||||
self._apply_chat_template, executor=self._executor
|
||||
)
|
||||
|
||||
def _apply_chat_template(self, *args, **kwargs):
|
||||
return self.get_tokenizer().apply_chat_template(*args, **kwargs)
|
||||
|
||||
def render_messages(
|
||||
self,
|
||||
messages: list[ChatCompletionMessageParam],
|
||||
params: ChatParams,
|
||||
) -> tuple[list[ConversationMessage], DictPrompt]:
|
||||
conversation, mm_data, mm_uuids = parse_chat_messages(
|
||||
messages,
|
||||
self.model_config,
|
||||
content_format="string",
|
||||
media_io_kwargs=params.media_io_kwargs,
|
||||
mm_processor_kwargs=params.mm_processor_kwargs,
|
||||
)
|
||||
|
||||
prompt_raw = self._apply_chat_template(
|
||||
conversation=conversation,
|
||||
messages=messages,
|
||||
**params.get_apply_chat_template_kwargs(),
|
||||
)
|
||||
|
||||
prompt = parse_dec_only_prompt(prompt_raw)
|
||||
if mm_data is not None:
|
||||
prompt["multi_modal_data"] = mm_data
|
||||
if mm_uuids is not None:
|
||||
prompt["multi_modal_uuids"] = mm_uuids
|
||||
|
||||
return conversation, prompt
|
||||
|
||||
async def render_messages_async(
|
||||
self,
|
||||
messages: list[ChatCompletionMessageParam],
|
||||
params: ChatParams,
|
||||
) -> tuple[list[ConversationMessage], DictPrompt]:
|
||||
conversation, mm_data, mm_uuids = await parse_chat_messages_async(
|
||||
messages,
|
||||
self.model_config,
|
||||
content_format="string",
|
||||
media_io_kwargs=params.media_io_kwargs,
|
||||
mm_processor_kwargs=params.mm_processor_kwargs,
|
||||
)
|
||||
|
||||
prompt_raw = await self._apply_chat_template_async(
|
||||
conversation=conversation,
|
||||
messages=messages,
|
||||
**params.get_apply_chat_template_kwargs(),
|
||||
)
|
||||
|
||||
prompt = parse_dec_only_prompt(prompt_raw)
|
||||
if mm_data is not None:
|
||||
prompt["multi_modal_data"] = mm_data
|
||||
if mm_uuids is not None:
|
||||
prompt["multi_modal_uuids"] = mm_uuids
|
||||
|
||||
return conversation, prompt
|
||||
@@ -22,7 +22,6 @@ logger = init_logger(__name__)
|
||||
_VLLM_RENDERERS = {
|
||||
"deepseek_v32": ("deepseek_v32", "DeepseekV32Renderer"),
|
||||
"deepseek_v4": ("deepseek_v4", "DeepseekV4Renderer"),
|
||||
"grok2": ("grok2", "Grok2Renderer"),
|
||||
"hf": ("hf", "HfRenderer"),
|
||||
"kimi_audio": ("hf", "HfRenderer"),
|
||||
"mistral": ("mistral", "MistralRenderer"),
|
||||
|
||||
@@ -1,452 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
"""Tokenizer for Grok-2 .tok.json format."""
|
||||
|
||||
import functools
|
||||
import json
|
||||
from collections.abc import Collection, Sequence, Set
|
||||
from pathlib import Path
|
||||
from typing import Any, Literal, overload
|
||||
|
||||
from huggingface_hub.utils import (
|
||||
EntryNotFoundError,
|
||||
HfHubHTTPError,
|
||||
RepositoryNotFoundError,
|
||||
RevisionNotFoundError,
|
||||
)
|
||||
from transformers import BatchEncoding
|
||||
from transformers.utils import chat_template_utils as hf_chat_utils
|
||||
|
||||
from vllm.entrypoints.chat_utils import ChatCompletionMessageParam
|
||||
from vllm.logger import init_logger
|
||||
from vllm.transformers_utils.repo_utils import hf_api
|
||||
|
||||
from .protocol import TokenizerLike
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
PAD = "<|pad|>"
|
||||
EOS = "<|eos|>"
|
||||
SEP = "<|separator|>"
|
||||
RESERVED_TOKEN_TEXTS = [f"<|reserved_{i}|>" for i in range(3, 128)]
|
||||
CONTROL_TOKEN_TEXTS = [f"<|control{i}|>" for i in range(1, 705)]
|
||||
DEFAULT_SPECIAL_TOKENS = [PAD, SEP, EOS]
|
||||
DEFAULT_CONTROL_TOKENS = {"pad": PAD, "sep": SEP, "eos": EOS}
|
||||
DEFAULT_CHAT_TEMPLATE = (
|
||||
"{% for message in messages %}"
|
||||
"{% if message['role'] == 'user' %}"
|
||||
"{{ 'Human: ' + message['content'].strip() + '<|separator|>\\n\\n' }}"
|
||||
"{% elif message['role'] == 'system' %}"
|
||||
"{{ 'System: ' + message['content'].strip() + '<|separator|>\\n\\n' }}"
|
||||
"{% elif message['role'] == 'assistant' %}"
|
||||
"{{ 'Assistant: ' + message['content'] + '<|separator|>\\n\\n' }}"
|
||||
"{% endif %}"
|
||||
"{% endfor %}"
|
||||
"{% if add_generation_prompt %}"
|
||||
"{{ 'Assistant:' }}"
|
||||
"{% endif %}"
|
||||
)
|
||||
|
||||
# Default + separate each single digit.
|
||||
PAT_STR_B = (
|
||||
r"""(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\r\n\p{L}\p{N}]?\p{L}+|\p{N}|"""
|
||||
r""" ?[^\s\p{L}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+"""
|
||||
)
|
||||
|
||||
|
||||
def _maybe_load_tokenizer_config(
|
||||
model_path: Path,
|
||||
*,
|
||||
repo_id: str | None,
|
||||
revision: str | None,
|
||||
download_dir: str | None,
|
||||
) -> dict[str, Any]:
|
||||
config_path = model_path / "tokenizer_config.json"
|
||||
if config_path.is_file():
|
||||
with config_path.open("r", encoding="utf-8") as f:
|
||||
return json.load(f)
|
||||
|
||||
if repo_id is None:
|
||||
return {}
|
||||
|
||||
try:
|
||||
config_file = hf_api().hf_hub_download(
|
||||
repo_id=repo_id,
|
||||
filename="tokenizer_config.json",
|
||||
revision=revision,
|
||||
cache_dir=download_dir,
|
||||
)
|
||||
except (RepositoryNotFoundError, RevisionNotFoundError, EntryNotFoundError):
|
||||
# If the repo, revision, or file does not exist, fall back silently.
|
||||
return {}
|
||||
except HfHubHTTPError as exc:
|
||||
logger.warning(
|
||||
"Failed to download tokenizer_config.json from %s. "
|
||||
"This may be due to a network or authentication issue. "
|
||||
"The default chat template will be used. Error: %s",
|
||||
repo_id,
|
||||
exc,
|
||||
)
|
||||
return {}
|
||||
|
||||
try:
|
||||
with Path(config_file).open("r", encoding="utf-8") as f:
|
||||
return json.load(f)
|
||||
except json.JSONDecodeError as exc:
|
||||
logger.warning(
|
||||
"Failed to parse tokenizer_config.json. "
|
||||
"The default chat template will be used. Error: %s",
|
||||
exc,
|
||||
)
|
||||
return {}
|
||||
except OSError as exc:
|
||||
logger.warning(
|
||||
"Failed to open tokenizer_config.json. "
|
||||
"The default chat template will be used. Error: %s",
|
||||
exc,
|
||||
)
|
||||
return {}
|
||||
|
||||
|
||||
def _load_tiktoken_encoding(
|
||||
vocab_file: Path,
|
||||
) -> tuple[Any, dict[str, int]]:
|
||||
try:
|
||||
import tiktoken
|
||||
except ImportError as exc:
|
||||
raise ImportError("Grok-2 tokenizer requires the `tiktoken` package.") from exc
|
||||
|
||||
with vocab_file.open("rb") as f:
|
||||
xtok_dict = json.load(f)
|
||||
|
||||
mergeable_ranks = {
|
||||
bytes(item["bytes"]): item["token"]
|
||||
for item in xtok_dict.get("regular_tokens", [])
|
||||
}
|
||||
special_tokens = {
|
||||
bytes(item["bytes"]).decode("utf-8", errors="replace"): item["token"]
|
||||
for item in xtok_dict.get("special_tokens", [])
|
||||
}
|
||||
|
||||
if xtok_dict.get("word_split") == "V1":
|
||||
pat_str = PAT_STR_B
|
||||
else:
|
||||
raise ValueError(f"Unknown word_split: {xtok_dict.get('word_split')!r}")
|
||||
|
||||
pat_str = xtok_dict.get("pat_str", pat_str)
|
||||
|
||||
kwargs = {
|
||||
"name": str(vocab_file),
|
||||
"pat_str": pat_str,
|
||||
"mergeable_ranks": mergeable_ranks,
|
||||
"special_tokens": special_tokens,
|
||||
}
|
||||
|
||||
if "vocab_size" in xtok_dict:
|
||||
kwargs["explicit_n_vocab"] = xtok_dict["vocab_size"]
|
||||
|
||||
tokenizer = tiktoken.Encoding(**kwargs)
|
||||
|
||||
default_allowed_special: set[str] | None = None
|
||||
if "default_allowed_special" in xtok_dict:
|
||||
default_allowed_special = {
|
||||
bytes(bytes_list).decode("utf-8", errors="replace")
|
||||
for bytes_list in xtok_dict["default_allowed_special"]
|
||||
}
|
||||
|
||||
tokenizer._default_allowed_special = default_allowed_special or set()
|
||||
tokenizer._control_tokens = DEFAULT_CONTROL_TOKENS
|
||||
|
||||
def encode_patched(
|
||||
self,
|
||||
text: str,
|
||||
*,
|
||||
allowed_special: Literal["all"] | Set[str] = set(),
|
||||
disallowed_special: Literal["all"] | Collection[str] = "all",
|
||||
) -> list[int]:
|
||||
del disallowed_special
|
||||
if isinstance(allowed_special, set):
|
||||
allowed_special |= self._default_allowed_special
|
||||
return tiktoken.Encoding.encode(
|
||||
self,
|
||||
text,
|
||||
allowed_special=allowed_special,
|
||||
disallowed_special=(),
|
||||
)
|
||||
|
||||
tokenizer.encode = functools.partial(encode_patched, tokenizer)
|
||||
tokenizer._default_allowed_special |= set(DEFAULT_CONTROL_TOKENS.values())
|
||||
tokenizer._default_allowed_special |= set(
|
||||
CONTROL_TOKEN_TEXTS + RESERVED_TOKEN_TEXTS
|
||||
)
|
||||
|
||||
return tokenizer, special_tokens
|
||||
|
||||
|
||||
class Grok2Tokenizer(TokenizerLike):
|
||||
@classmethod
|
||||
def from_pretrained(
|
||||
cls,
|
||||
path_or_repo_id: str | Path,
|
||||
*args,
|
||||
trust_remote_code: bool = False,
|
||||
revision: str | None = None,
|
||||
download_dir: str | None = None,
|
||||
**kwargs,
|
||||
) -> "Grok2Tokenizer":
|
||||
if args:
|
||||
logger.debug_once("Ignoring extra positional args for Grok2Tokenizer.")
|
||||
|
||||
path = Path(path_or_repo_id)
|
||||
if path.is_file():
|
||||
vocab_file = path
|
||||
model_path = path.parent
|
||||
repo_id = None
|
||||
elif path.is_dir():
|
||||
vocab_file = path / "tokenizer.tok.json"
|
||||
model_path = path
|
||||
repo_id = None
|
||||
else:
|
||||
vocab_file = Path(
|
||||
hf_api().hf_hub_download(
|
||||
repo_id=str(path_or_repo_id),
|
||||
filename="tokenizer.tok.json",
|
||||
revision=revision,
|
||||
cache_dir=download_dir,
|
||||
)
|
||||
)
|
||||
model_path = vocab_file.parent
|
||||
repo_id = str(path_or_repo_id)
|
||||
|
||||
if not vocab_file.is_file():
|
||||
raise FileNotFoundError(f"tokenizer.tok.json not found at {vocab_file}.")
|
||||
|
||||
config = _maybe_load_tokenizer_config(
|
||||
model_path,
|
||||
repo_id=repo_id,
|
||||
revision=revision,
|
||||
download_dir=download_dir,
|
||||
)
|
||||
|
||||
return cls(
|
||||
vocab_file=vocab_file,
|
||||
name_or_path=str(path_or_repo_id),
|
||||
truncation_side=kwargs.get("truncation_side", "left"),
|
||||
chat_template=config.get("chat_template"),
|
||||
init_kwargs=config,
|
||||
)
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
vocab_file: Path,
|
||||
name_or_path: str,
|
||||
truncation_side: str,
|
||||
chat_template: str | None,
|
||||
init_kwargs: dict[str, Any] | None = None,
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.name_or_path = name_or_path
|
||||
self._truncation_side = truncation_side
|
||||
self.init_kwargs = init_kwargs or {}
|
||||
self._chat_template = chat_template or DEFAULT_CHAT_TEMPLATE
|
||||
|
||||
self._tokenizer, self._special_tokens = _load_tiktoken_encoding(vocab_file)
|
||||
|
||||
self._token_to_id: dict[str, int] = {}
|
||||
self._id_to_token: dict[int, str] = {}
|
||||
for token, token_id in self._tokenizer._mergeable_ranks.items():
|
||||
token_str = token.decode("utf-8", errors="replace")
|
||||
self._token_to_id[token_str] = token_id
|
||||
self._id_to_token[token_id] = token_str
|
||||
|
||||
for token, token_id in self._special_tokens.items():
|
||||
self._token_to_id[token] = token_id
|
||||
self._id_to_token[token_id] = token
|
||||
|
||||
bos_token_id = self._special_tokens.get(SEP)
|
||||
if bos_token_id is None:
|
||||
bos_token_id = self._special_tokens.get(PAD)
|
||||
if bos_token_id is None:
|
||||
bos_token_id = self._special_tokens.get(EOS)
|
||||
if bos_token_id is None:
|
||||
bos_token_id = 0
|
||||
self._bos_token_id = bos_token_id
|
||||
|
||||
self._eos_token_id = self._special_tokens.get(EOS, self._bos_token_id)
|
||||
self._pad_token_id = self._special_tokens.get(PAD, self._eos_token_id)
|
||||
self._unk_token_id = self._pad_token_id
|
||||
|
||||
self._max_chars_per_token = max(len(tok) for tok in self._token_to_id)
|
||||
|
||||
def num_special_tokens_to_add(self) -> int:
|
||||
return 0
|
||||
|
||||
@property
|
||||
def all_special_tokens(self) -> list[str]:
|
||||
return list(self._special_tokens.keys())
|
||||
|
||||
@property
|
||||
def all_special_ids(self) -> list[int]:
|
||||
return list(self._special_tokens.values())
|
||||
|
||||
@property
|
||||
def bos_token_id(self) -> int:
|
||||
return self._bos_token_id
|
||||
|
||||
@property
|
||||
def eos_token_id(self) -> int:
|
||||
return self._eos_token_id
|
||||
|
||||
@property
|
||||
def pad_token_id(self) -> int:
|
||||
return self._pad_token_id
|
||||
|
||||
@property
|
||||
def is_fast(self) -> bool:
|
||||
return False
|
||||
|
||||
@property
|
||||
def vocab_size(self) -> int:
|
||||
return self._tokenizer.n_vocab
|
||||
|
||||
@property
|
||||
def max_token_id(self) -> int:
|
||||
return self._tokenizer.n_vocab - 1
|
||||
|
||||
@property
|
||||
def max_chars_per_token(self) -> int:
|
||||
return self._max_chars_per_token
|
||||
|
||||
@property
|
||||
def truncation_side(self) -> str:
|
||||
return self._truncation_side
|
||||
|
||||
def get_vocab(self) -> dict[str, int]:
|
||||
return dict(self._token_to_id)
|
||||
|
||||
def get_added_vocab(self) -> dict[str, int]:
|
||||
return dict(self._special_tokens)
|
||||
|
||||
def _maybe_truncate(self, tokens: list[int], max_length: int | None) -> list[int]:
|
||||
if max_length is None or len(tokens) <= max_length:
|
||||
return tokens
|
||||
if self.truncation_side == "left":
|
||||
return tokens[-max_length:]
|
||||
return tokens[:max_length]
|
||||
|
||||
def encode(
|
||||
self,
|
||||
text: str,
|
||||
truncation: bool | None = None,
|
||||
max_length: int | None = None,
|
||||
add_special_tokens: bool = True,
|
||||
) -> list[int]:
|
||||
del add_special_tokens
|
||||
tokens = self._tokenizer.encode(text)
|
||||
if truncation:
|
||||
tokens = self._maybe_truncate(tokens, max_length)
|
||||
return tokens
|
||||
|
||||
def decode(
|
||||
self, ids: Sequence[int] | int, skip_special_tokens: bool = False
|
||||
) -> str:
|
||||
if isinstance(ids, int):
|
||||
ids = [ids]
|
||||
if skip_special_tokens:
|
||||
ids = [
|
||||
token_id
|
||||
for token_id in ids
|
||||
if token_id not in self._special_tokens.values()
|
||||
]
|
||||
return self._tokenizer.decode(ids)
|
||||
|
||||
@overload
|
||||
def convert_tokens_to_ids(self, tokens: str) -> int: ...
|
||||
|
||||
@overload
|
||||
def convert_tokens_to_ids(self, tokens: list[str]) -> list[int]: ...
|
||||
|
||||
def convert_tokens_to_ids(self, tokens: str | list[str]) -> int | list[int]:
|
||||
if isinstance(tokens, str):
|
||||
return self._token_to_id.get(tokens, self._unk_token_id)
|
||||
return [self._token_to_id.get(token, self._unk_token_id) for token in tokens]
|
||||
|
||||
def convert_ids_to_tokens(
|
||||
self, ids: Sequence[int], skip_special_tokens: bool = False
|
||||
) -> list[str]:
|
||||
tokens = []
|
||||
for token_id in ids:
|
||||
if skip_special_tokens and token_id in self._special_tokens.values():
|
||||
continue
|
||||
tokens.append(self._id_to_token.get(token_id, "<|unk|>"))
|
||||
return tokens
|
||||
|
||||
def convert_tokens_to_string(self, tokens: list[str]) -> str:
|
||||
token_ids = self.convert_tokens_to_ids(tokens)
|
||||
return self.decode(token_ids, skip_special_tokens=False)
|
||||
|
||||
def __call__(
|
||||
self,
|
||||
text: str | list[str],
|
||||
text_pair: str | None = None,
|
||||
add_special_tokens: bool = True,
|
||||
truncation: bool = False,
|
||||
max_length: int | None = None,
|
||||
) -> BatchEncoding:
|
||||
if text_pair is not None:
|
||||
raise NotImplementedError("text_pair is not supported for Grok2Tokenizer.")
|
||||
|
||||
if isinstance(text, list):
|
||||
input_ids_batch: list[list[int]] = [
|
||||
self.encode(
|
||||
item,
|
||||
truncation=truncation,
|
||||
max_length=max_length,
|
||||
add_special_tokens=add_special_tokens,
|
||||
)
|
||||
for item in text
|
||||
]
|
||||
attention_mask_batch = [[1] * len(ids) for ids in input_ids_batch]
|
||||
return BatchEncoding(
|
||||
{"input_ids": input_ids_batch, "attention_mask": attention_mask_batch}
|
||||
)
|
||||
|
||||
input_ids = self.encode(
|
||||
text,
|
||||
truncation=truncation,
|
||||
max_length=max_length,
|
||||
add_special_tokens=add_special_tokens,
|
||||
)
|
||||
attention_mask = [1] * len(input_ids)
|
||||
return BatchEncoding({"input_ids": input_ids, "attention_mask": attention_mask})
|
||||
|
||||
def get_chat_template(
|
||||
self, chat_template: str | None, tools: list[dict[str, Any]] | None = None
|
||||
) -> str | None:
|
||||
del tools
|
||||
return chat_template or self._chat_template
|
||||
|
||||
def apply_chat_template(
|
||||
self,
|
||||
messages: list[ChatCompletionMessageParam],
|
||||
tools: list[dict[str, Any]] | None = None,
|
||||
chat_template: str | None = None,
|
||||
tokenize: bool = False,
|
||||
**kwargs,
|
||||
) -> str | list[int]:
|
||||
template = self.get_chat_template(chat_template, tools=tools)
|
||||
if template is None:
|
||||
raise ValueError(
|
||||
"No chat template available. Provide `chat_template` explicitly."
|
||||
)
|
||||
kwargs["return_dict"] = False
|
||||
prompt = hf_chat_utils.apply_chat_template(
|
||||
conversation=messages,
|
||||
chat_template=template,
|
||||
tools=tools,
|
||||
**kwargs,
|
||||
)
|
||||
if tokenize:
|
||||
return self.encode(prompt, add_special_tokens=False)
|
||||
return prompt
|
||||
@@ -36,7 +36,6 @@ _MODEL_TYPES_WITH_INCORRECT_TOKENIZER_CLASS: set[str] = {"step3_vl", "step3p7"}
|
||||
_VLLM_TOKENIZERS = {
|
||||
"deepseek_v32": ("deepseek_v32", "DeepseekV32Tokenizer"),
|
||||
"deepseek_v4": ("deepseek_v4", "DeepseekV4Tokenizer"),
|
||||
"grok2": ("grok2", "Grok2Tokenizer"),
|
||||
"hf": ("hf", "CachedHfTokenizer"),
|
||||
"kimi_audio": ("kimi_audio", "KimiAudioTokenizer"),
|
||||
"mistral": ("mistral", "MistralTokenizer"),
|
||||
|
||||
Reference in New Issue
Block a user