forked from Karylab-cklius/vllm
Compare commits
89
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6257a61c3d | ||
|
|
9fde043f54 | ||
|
|
24dd2aec81 | ||
|
|
3ee9eea928 | ||
|
|
5bce653e09 | ||
|
|
5ad11172b7 | ||
|
|
f70caef48b | ||
|
|
8d8ec38361 | ||
|
|
b1c6dba558 | ||
|
|
598d51153a | ||
|
|
095adf1fdc | ||
|
|
51ee564e56 | ||
|
|
373eb314af | ||
|
|
641cb59592 | ||
|
|
07f9baf756 | ||
|
|
7a90eb98ab | ||
|
|
8f4c69b222 | ||
|
|
8b79971bb9 | ||
|
|
f676808ba0 | ||
|
|
98e4726a14 | ||
|
|
740f379fae | ||
|
|
40cc2e8327 | ||
|
|
ba22152096 | ||
|
|
90ce3a09be | ||
|
|
26c754d847 | ||
|
|
3d7f357ebf | ||
|
|
736f1a5907 | ||
|
|
344609ab17 | ||
|
|
d039c17114 | ||
|
|
cdab28319f | ||
|
|
2fa10566e3 | ||
|
|
fb265fc8fb | ||
|
|
8f0e75e16b | ||
|
|
98ba9b9583 | ||
|
|
990c2a0187 | ||
|
|
e433634c78 | ||
|
|
16f8110935 | ||
|
|
d9c1767cd4 | ||
|
|
e9cc1fd093 | ||
|
|
f1073c050c | ||
|
|
394edc8108 | ||
|
|
69715823df | ||
|
|
6569df6a3e | ||
|
|
f2aaf59151 | ||
|
|
95a248faed | ||
|
|
d2ec433e37 | ||
|
|
78a04c208d | ||
|
|
b71218107f | ||
|
|
cc1d020d01 | ||
|
|
8974ed89cd | ||
|
|
fb2faceacd | ||
|
|
b6cc46ec3b | ||
|
|
fa4321de3d | ||
|
|
9226613043 | ||
|
|
34b560b725 | ||
|
|
91b5647300 | ||
|
|
4a6bf3c77f | ||
|
|
d2afe39647 | ||
|
|
2a9113f998 | ||
|
|
0cd6f767e3 | ||
|
|
f1445f6dbd | ||
|
|
1d354c694e | ||
|
|
2f21224527 | ||
|
|
fa1fa968c4 | ||
|
|
6eac8e0070 | ||
|
|
1a308c449c | ||
|
|
e7c9df9449 | ||
|
|
26eb87204d | ||
|
|
4c3c17d43b | ||
|
|
f329ce405b | ||
|
|
07516fda67 | ||
|
|
67ff0ae30f | ||
|
|
ab3b6d97aa | ||
|
|
fb5291b35b | ||
|
|
d6d39c111e | ||
|
|
379950191f | ||
|
|
576bf75d0e | ||
|
|
f006e5a24c | ||
|
|
f63dca6838 | ||
|
|
8651f043b8 | ||
|
|
3775d5fcab | ||
|
|
d7192cfccf | ||
|
|
978de83353 | ||
|
|
a14f57a3ac | ||
|
|
18f658bb31 | ||
|
|
400a9c386d | ||
|
|
bbdcbe4686 | ||
|
|
4875b4456b | ||
|
|
cc17a5e0d1 |
@@ -76,6 +76,30 @@ steps:
|
||||
pytest -v -s v1/sample/test_logprobs.py &&
|
||||
pytest -v -s v1/sample/test_logprobs_e2e.py'
|
||||
|
||||
- label: Basic Models Tests (Initialization)
|
||||
timeout_in_minutes: 60
|
||||
device: intel_gpu
|
||||
agent_tags:
|
||||
label: production
|
||||
gpu: 1+
|
||||
mem: 24+
|
||||
no_plugin: true
|
||||
working_dir: "."
|
||||
env:
|
||||
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||
REPO: "vllm-ci-test-repo"
|
||||
VLLM_TEST_DEVICE: "xpu"
|
||||
source_file_dependencies:
|
||||
- vllm/
|
||||
- tests/models/test_initialization.py
|
||||
- tests/models/registry.py
|
||||
commands:
|
||||
- >-
|
||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||
'export VLLM_XPU_FUSED_MOE_USE_REF=1 &&
|
||||
cd tests &&
|
||||
pytest -v -s models/test_initialization.py::test_can_initialize_large_subset[Eagle3MiniMaxM2ForCausalLM]'
|
||||
|
||||
- label: XPU CPU Offload
|
||||
timeout_in_minutes: 60
|
||||
device: intel_gpu
|
||||
|
||||
@@ -49,6 +49,30 @@ steps:
|
||||
VLLM_XPU_FUSED_MOE_USE_REF=1 python3 examples/basic/offline_inference/generate.py --model Qwen/Qwen3-30B-A3B-Instruct-2507-FP8 --enforce-eager -tp 2 --max-model-len 8192 &&
|
||||
python3 examples/basic/offline_inference/generate.py --model INCModel/Qwen3-30B-A3B-Instruct-2507-MXFP4-LLMC --enforce-eager -tp 2 --max-model-len 8192
|
||||
'
|
||||
- label: "XPU W8A8 FP8 Linear Examples"
|
||||
depends_on:
|
||||
- image-build-xpu
|
||||
timeout_in_minutes: 60
|
||||
device: intel_gpu
|
||||
agent_tags:
|
||||
label: production
|
||||
gpu: 1+
|
||||
mem: 24+
|
||||
no_plugin: true
|
||||
env:
|
||||
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||
REPO: "vllm-ci-test-repo"
|
||||
VLLM_TEST_DEVICE: "xpu"
|
||||
source_file_dependencies:
|
||||
- vllm/
|
||||
- .buildkite/intel_jobs/test-intel.yaml
|
||||
commands:
|
||||
- >-
|
||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||
'python3 examples/basic/offline_inference/generate.py --linear-backend xpu --model RedHatAI/Meta-Llama-3.1-8B-Instruct-FP8 --enforce-eager --max-model-len 4096 &&
|
||||
python3 examples/basic/offline_inference/generate.py --linear-backend xpu --model neuralmagic/Llama-3.2-1B-Instruct-FP8-dynamic --enforce-eager --max-model-len 4096 &&
|
||||
python3 examples/basic/offline_inference/generate.py --linear-backend xpu --model meta-llama/Llama-3.2-1B-Instruct --quantization fp8 --enforce-eager --max-model-len 4096
|
||||
'
|
||||
- label: "XPU V1 test"
|
||||
depends_on:
|
||||
- image-build-xpu
|
||||
@@ -120,4 +144,27 @@ steps:
|
||||
- >-
|
||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||
'cd tests &&
|
||||
pytest -v -s quantization/test_auto_round.py'
|
||||
pytest -v -s quantization/test_auto_round.py'
|
||||
- label: "XPU compressed tensors FP8 test"
|
||||
depends_on:
|
||||
- image-build-xpu
|
||||
timeout_in_minutes: 60
|
||||
device: intel_gpu
|
||||
agent_tags:
|
||||
label: production
|
||||
gpu: 1+
|
||||
mem: 16+
|
||||
no_plugin: true
|
||||
env:
|
||||
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||
REPO: "vllm-ci-test-repo"
|
||||
VLLM_TEST_DEVICE: "xpu"
|
||||
source_file_dependencies:
|
||||
- vllm/
|
||||
- tests/quantization/test_compressed_tensors.py
|
||||
- .buildkite/intel_jobs/test-intel.yaml
|
||||
commands:
|
||||
- >-
|
||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||
'cd tests &&
|
||||
pytest -v -s quantization/test_compressed_tensors.py::test_compressed_tensors_fp8'
|
||||
@@ -448,9 +448,11 @@ checkout="${BUILDKITE_BUILD_CHECKOUT_PATH:-}"
|
||||
if [[ -z "${checkout}" || ! -d "${checkout}" ]]; then
|
||||
checkout="."
|
||||
fi
|
||||
if git -C "${checkout}" rev-parse --is-inside-work-tree >/dev/null 2>&1; then
|
||||
# Pass safe.directory per-command (-c) because buildkite runs will always fail
|
||||
# the next check on git 2.35.2+ due to mixed uses of root and buildkite-agent/uids.
|
||||
if git -c "safe.directory=${checkout}" -C "${checkout}" rev-parse --is-inside-work-tree >/dev/null 2>&1; then
|
||||
vllm_standalone_merge_base="$(
|
||||
git -C "${checkout}" merge-base HEAD origin/main 2>/dev/null || true
|
||||
git -c "safe.directory=${checkout}" -C "${checkout}" merge-base HEAD origin/main 2>/dev/null || true
|
||||
)"
|
||||
fi
|
||||
if [[ -z "${vllm_standalone_merge_base}" ]]; then
|
||||
@@ -549,6 +551,7 @@ else
|
||||
fi
|
||||
|
||||
docker run \
|
||||
-t -i \
|
||||
--device /dev/kfd $BUILDKITE_AGENT_META_DATA_RENDER_DEVICES \
|
||||
$RDMA_FLAGS \
|
||||
--network=host \
|
||||
|
||||
@@ -38,7 +38,8 @@ function cpu_tests() {
|
||||
pytest -x -v -s tests/kernels/attention/test_cpu_attn.py
|
||||
pytest -x -v -s tests/kernels/core/test_cpu_activation.py
|
||||
pytest -x -v -s tests/kernels/moe/test_cpu_fused_moe.py
|
||||
pytest -x -v -s tests/kernels/mamba/cpu/test_cpu_gdn_ops.py"
|
||||
pytest -x -v -s tests/kernels/mamba/cpu/test_cpu_gdn_ops.py
|
||||
pytest -x -v -s tests/kernels/moe/test_cpu_int4_moe.py"
|
||||
|
||||
# skip tests requiring model downloads if HF_TOKEN is not set
|
||||
# due to rate-limits
|
||||
|
||||
@@ -109,7 +109,9 @@ run_nodes() {
|
||||
if [ "$node" -ne 0 ]; then
|
||||
docker exec -d "node$node" /bin/bash -c "cd $WORKING_DIR ; ${COMMANDS[$node]}"
|
||||
else
|
||||
docker exec "node$node" /bin/bash -c "cd $WORKING_DIR ; ${COMMANDS[$node]}"
|
||||
# Allocate a TTY (-t -i) for the foreground head node so its output
|
||||
# keeps ANSI color in the Buildkite log (see run-amd-test.sh).
|
||||
docker exec -t -i "node$node" /bin/bash -c "cd $WORKING_DIR ; ${COMMANDS[$node]}"
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
+251
-9
@@ -472,7 +472,7 @@ steps:
|
||||
commands:
|
||||
- TARGET_TEST_SUITE=MI300 pytest basic_correctness/ -v -s -m 'distributed(num_gpus=2)'
|
||||
- CUDA_VISIBLE_DEVICES=0,1 pytest -v -s model_executor/model_loader/test_sharded_state_loader.py -m '(not slow_test)'
|
||||
- pytest models/test_transformers.py -v -s -m 'distributed(num_gpus=2)'
|
||||
- pytest models/transformers/test_backend.py -v -s -m 'distributed(num_gpus=2)'
|
||||
- pytest models/language -v -s -m 'distributed(num_gpus=2)'
|
||||
- pytest models/multimodal -v -s -m 'distributed(num_gpus=2)' --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_phi4siglip.py
|
||||
- pytest models/multimodal/generation/test_phi4siglip.py -v -s -m 'distributed(num_gpus=2)'
|
||||
@@ -1025,6 +1025,23 @@ steps:
|
||||
commands:
|
||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small.txt
|
||||
|
||||
- label: MRCR Eval Small Models # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_1
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/model_executor/models/
|
||||
- vllm/model_executor/model_loader/
|
||||
- vllm/v1/attention/backends/
|
||||
- vllm/v1/attention/selector.py
|
||||
- vllm/_aiter_ops.py
|
||||
- vllm/platforms/rocm.py
|
||||
- tests/evals/mrcr/
|
||||
commands:
|
||||
- pytest -s -v evals/mrcr/test_mrcr_correctness.py --config-list-file=evals/mrcr/configs/models-small.txt
|
||||
|
||||
- label: LM Eval Small Models (MI300) # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
@@ -1283,6 +1300,21 @@ steps:
|
||||
|
||||
#---------------------------------------------------------- mi300 · kernels ----------------------------------------------------------#
|
||||
|
||||
- label: vLLM IR Tests # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_1
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/"
|
||||
source_file_dependencies:
|
||||
- vllm/ir
|
||||
- vllm/kernels
|
||||
- vllm/_aiter_ops.py
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- pytest -v -s tests/ir
|
||||
- pytest -v -s tests/kernels/ir
|
||||
|
||||
- label: Kernels Attention Test %N # TBD
|
||||
timeout_in_minutes: 100
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
@@ -1317,6 +1349,21 @@ steps:
|
||||
commands:
|
||||
- pytest -v -s kernels/core --ignore=kernels/core/test_minimax_reduce_rms.py kernels/test_concat_mla_q.py kernels/test_top_k_per_row.py
|
||||
|
||||
- label: Kernels KDA Test # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_1
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/model_executor/layers/fla/ops/kda.py
|
||||
- vllm/model_executor/layers/fla/ops/chunk_delta_h.py
|
||||
- vllm/model_executor/layers/fla/ops/l2norm.py
|
||||
- tests/kernels/test_kda.py
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- pytest -v -s kernels/test_kda.py
|
||||
|
||||
- label: Kernels MoE Test %N # TBD
|
||||
timeout_in_minutes: 95
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
@@ -1429,13 +1476,135 @@ steps:
|
||||
- pytest -v -s model_executor -m '(not slow_test)'
|
||||
- pytest -v -s entrypoints/openai/completion/test_tensorizer_entrypoint.py
|
||||
|
||||
#---------------------------------------------------- mi300 · model_runner_v2 -------------------------------------------------------#
|
||||
|
||||
- label: Model Runner V2 Core Tests # TBD
|
||||
timeout_in_minutes: 60
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_1
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/v1/worker/gpu/
|
||||
- vllm/v1/worker/gpu_worker.py
|
||||
- vllm/v1/core/sched/
|
||||
- vllm/v1/attention/
|
||||
- tests/v1/engine/test_llm_engine.py
|
||||
- tests/v1/e2e/
|
||||
- tests/entrypoints/llm/test_struct_output_generate.py
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- set -x
|
||||
- export VLLM_USE_V2_MODEL_RUNNER=1
|
||||
- pytest -v -s v1/engine/test_llm_engine.py -k "not test_engine_metrics"
|
||||
- ENFORCE_EAGER=1 pytest -v -s v1/e2e/general/test_async_scheduling.py -k "not ngram"
|
||||
- pytest -v -s v1/e2e/general/test_context_length.py
|
||||
- pytest -v -s v1/e2e/general/test_min_tokens.py
|
||||
- pytest -v -s entrypoints/llm/test_struct_output_generate.py -k "xgrammar and not speculative_config6 and not speculative_config7 and not speculative_config8 and not speculative_config0"
|
||||
|
||||
- label: Model Runner V2 Examples # TBD
|
||||
timeout_in_minutes: 60
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_1
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/examples"
|
||||
source_file_dependencies:
|
||||
- vllm/v1/worker/gpu/
|
||||
- vllm/v1/core/sched/
|
||||
- vllm/v1/worker/gpu_worker.py
|
||||
- examples/basic/offline_inference/
|
||||
- examples/generate/multimodal/
|
||||
- examples/features/
|
||||
- examples/pooling/embed/vision_embedding_offline.py
|
||||
- examples/features/tensorize_vllm_model.py
|
||||
- examples/deployment/llm_engine_example.py
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- set -x
|
||||
- export VLLM_USE_V2_MODEL_RUNNER=1
|
||||
- pip install tensorizer
|
||||
- python3 basic/offline_inference/chat.py
|
||||
- python3 basic/offline_inference/generate.py --model facebook/opt-125m
|
||||
- python3 generate/multimodal/audio_language_offline.py --seed 0
|
||||
- python3 generate/multimodal/vision_language_offline.py --seed 0
|
||||
- python3 generate/multimodal/vision_language_multi_image_offline.py --seed 0
|
||||
- python3 generate/multimodal/encoder_decoder_multimodal_offline.py --model-type whisper --seed 0
|
||||
- python3 pooling/embed/vision_embedding_offline.py --seed 0
|
||||
- python3 features/automatic_prefix_caching/prefix_caching_offline.py
|
||||
- python3 deployment/llm_engine_example.py
|
||||
- python3 features/tensorize_vllm_model.py --model facebook/opt-125m serialize --serialized-directory /tmp/ --suffix v1 && python3 features/tensorize_vllm_model.py --model facebook/opt-125m deserialize --path-to-tensors /tmp/vllm/facebook/opt-125m/v1/model.tensors
|
||||
- python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 2048
|
||||
- python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle3 --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 1536
|
||||
|
||||
- label: Model Runner V2 Distributed (2 GPUs) # TBD
|
||||
timeout_in_minutes: 60
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_2
|
||||
num_gpus: 2
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/v1/worker/gpu/
|
||||
- vllm/v1/worker/gpu_worker.py
|
||||
- tests/basic_correctness/test_basic_correctness.py
|
||||
- tests/v1/distributed/test_async_llm_dp.py
|
||||
- tests/v1/distributed/test_eagle_dp.py
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- set -x
|
||||
- export VLLM_USE_V2_MODEL_RUNNER=1
|
||||
- TARGET_TEST_SUITE=MI300 pytest -v -s basic_correctness/test_basic_correctness.py -m 'distributed(num_gpus=2)' -k "not ray and not True"
|
||||
- TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_async_llm_dp.py -k "not ray"
|
||||
- TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_eagle_dp.py
|
||||
|
||||
- label: Model Runner V2 Pipeline Parallelism (4 GPUs) # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_4
|
||||
num_gpus: 4
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/v1/worker/gpu/
|
||||
- vllm/v1/worker/gpu_worker.py
|
||||
- tests/distributed/test_pipeline_parallel.py
|
||||
- tests/distributed/test_pp_cudagraph.py
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- set -x
|
||||
- export VLLM_USE_V2_MODEL_RUNNER=1
|
||||
- pytest -v -s distributed/test_pipeline_parallel.py -k "not ray and not Jamba"
|
||||
- pytest -v -s distributed/test_pp_cudagraph.py -k "not ray"
|
||||
|
||||
- label: Model Runner V2 Spec Decode # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_1
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/v1/worker/gpu/
|
||||
- vllm/v1/worker/gpu_worker.py
|
||||
- tests/v1/spec_decode/test_max_len.py
|
||||
- tests/v1/spec_decode/test_rejection_sampler_utils.py
|
||||
- tests/v1/spec_decode/test_synthetic_rejection_sampler_utils.py
|
||||
- tests/v1/e2e/spec_decode/test_spec_decode.py
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- set -x
|
||||
- export VLLM_USE_V2_MODEL_RUNNER=1
|
||||
- pytest -v -s v1/spec_decode/test_max_len.py -k "eagle or mtp"
|
||||
- pytest -v -s v1/spec_decode/test_rejection_sampler_utils.py
|
||||
- pytest -v -s v1/spec_decode/test_synthetic_rejection_sampler_utils.py
|
||||
- pytest -v -s v1/e2e/spec_decode/test_spec_decode.py -k "eagle or mtp"
|
||||
|
||||
#------------------------------------------------------ mi300 · models / basic -------------------------------------------------------#
|
||||
|
||||
- label: Basic Models Tests (Extra Initialization) %N # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_1
|
||||
parallelism: 2
|
||||
parallelism: 6
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/model_executor/models/
|
||||
@@ -1468,10 +1637,10 @@ steps:
|
||||
source_file_dependencies:
|
||||
- vllm/
|
||||
- tests/models/test_terratorch.py
|
||||
- tests/models/test_transformers.py
|
||||
- tests/models/transformers/test_backend.py
|
||||
- tests/models/test_registry.py
|
||||
commands:
|
||||
- pytest -v -s models/test_terratorch.py models/test_transformers.py models/test_registry.py
|
||||
- pytest -v -s models/test_terratorch.py models/transformers/test_backend.py models/test_registry.py
|
||||
|
||||
#----------------------------------------------------- mi300 · models / language -----------------------------------------------------#
|
||||
|
||||
@@ -1691,7 +1860,7 @@ steps:
|
||||
- examples/
|
||||
commands:
|
||||
- pip install --upgrade git+https://github.com/huggingface/transformers
|
||||
- pytest -v -s tests/models/test_transformers.py
|
||||
- pytest -v -s tests/models/transformers/test_backend.py
|
||||
- pytest -v -s tests/models/multimodal/test_mapping.py
|
||||
- python3 examples/basic/offline_inference/chat.py
|
||||
- python3 examples/generate/multimodal/vision_language_offline.py --model-type qwen2_5_vl
|
||||
@@ -1745,6 +1914,22 @@ steps:
|
||||
- pytest -v -s plugins_tests/test_oot_registration_offline.py # it needs a clean process
|
||||
- pytest -v -s plugins_tests/lora_resolvers # unit tests for in-tree lora resolver plugins
|
||||
|
||||
- label: GGUF Plugin # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_1
|
||||
optional: true
|
||||
soft_fail: true
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/model_executor/layers/quantization
|
||||
- tests/plugins_tests/test_gguf_plugin.py
|
||||
- tests/plugins_tests/gguf
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- pip install "vllm-gguf-plugin >= 0.0.2"
|
||||
- pytest -v -s plugins_tests/gguf
|
||||
|
||||
#------------------------------------------------------- mi300 · quantization --------------------------------------------------------#
|
||||
|
||||
- label: Quantization # TBD
|
||||
@@ -2063,7 +2248,7 @@ steps:
|
||||
- pytest -v -s v1/worker
|
||||
- pytest -v -s -m 'not cpu_test' v1/kv_connector/unit
|
||||
- pytest -v -s -m 'not cpu_test' v1/metrics
|
||||
- pip install -U git+https://github.com/robertgshaw2-redhat/lm-evaluation-harness.git@streaming-api
|
||||
- pip install -U git+https://github.com/vllm-project/lm-evaluation-harness.git@streaming-api
|
||||
# - export HSA_NO_SCRATCH_RECLAIM=1
|
||||
- pytest -v -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
|
||||
|
||||
@@ -2254,6 +2439,59 @@ steps:
|
||||
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||
- HYBRID_SSM=1 ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||
|
||||
- label: Hybrid SSM NixlConnector PD prefix cache test (2 GPUs) # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_2
|
||||
num_gpus: 2
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/distributed/kv_transfer/kv_connector/v1/nixl/
|
||||
- vllm/v1/core/sched/
|
||||
- vllm/v1/core/kv_cache_coordinator.py
|
||||
- tests/v1/kv_connector/nixl_integration/
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||
- ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/run_mamba_prefix_cache_test.sh
|
||||
|
||||
- label: MultiConnector (Nixl+Offloading) PD accuracy (2 GPUs) # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_2
|
||||
num_gpus: 2
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/distributed/kv_transfer/kv_connector/v1/nixl/
|
||||
- vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py
|
||||
- vllm/distributed/kv_transfer/kv_connector/v1/offloading_connector.py
|
||||
- vllm/distributed/kv_transfer/kv_connector/v1/offloading/
|
||||
- tests/v1/kv_connector/nixl_integration/
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||
- ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/run_multi_connector_accuracy_test.sh
|
||||
|
||||
- label: MultiConnector (Nixl+Offloading) PD edge cases (2 GPUs) # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
agent_pool: mi300_2
|
||||
num_gpus: 2
|
||||
optional: true
|
||||
working_dir: "/vllm-workspace/tests"
|
||||
source_file_dependencies:
|
||||
- vllm/distributed/kv_transfer/kv_connector/v1/nixl/
|
||||
- vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py
|
||||
- vllm/distributed/kv_transfer/kv_connector/v1/offloading_connector.py
|
||||
- vllm/distributed/kv_transfer/kv_connector/v1/offloading/
|
||||
- tests/v1/kv_connector/nixl_integration/
|
||||
- vllm/platforms/rocm.py
|
||||
commands:
|
||||
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||
- ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/run_multi_connector_edge_case_test.sh
|
||||
|
||||
- label: V1 e2e (4 GPUs) # TBD
|
||||
timeout_in_minutes: 180
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
|
||||
@@ -2544,15 +2782,19 @@ steps:
|
||||
- vllm/envs.py
|
||||
- examples/offline_inference/data_parallel.py
|
||||
- tests/distributed/test_context_parallel.py
|
||||
- tests/distributed/test_rocm_aiter_custom_ar.py
|
||||
- tests/distributed/test_rocm_quick_reduce.py
|
||||
- tests/distributed/test_quick_all_reduce.py
|
||||
- tests/v1/e2e/general/test_rocm_aiter_custom_ar.py
|
||||
- tests/v1/distributed/test_dbo.py
|
||||
- tests/utils.py
|
||||
commands:
|
||||
- pytest -v -s tests/distributed/test_context_parallel.py
|
||||
- pytest -v -s tests/v1/distributed/test_dbo.py
|
||||
- pytest -v -s tests/distributed/test_rocm_aiter_custom_ar.py
|
||||
- pytest -v -s tests/v1/e2e/general/test_rocm_aiter_custom_ar.py
|
||||
- pytest -v -s tests/distributed/test_rocm_quick_reduce.py
|
||||
- pytest -v -s tests/distributed/test_quick_all_reduce.py
|
||||
- pytest -v -s tests/v1/distributed/test_dbo.py
|
||||
|
||||
#-------------------------------------------------------- mi355 · entrypoints --------------------------------------------------------#
|
||||
|
||||
@@ -2770,7 +3012,7 @@ steps:
|
||||
#--------------------------------------------------------- mi355 · examples ----------------------------------------------------------#
|
||||
|
||||
- label: Examples # TBD
|
||||
timeout_in_minutes: 45
|
||||
timeout_in_minutes: 100
|
||||
mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
|
||||
agent_pool: mi355_1
|
||||
working_dir: "/vllm-workspace/examples"
|
||||
@@ -3133,7 +3375,7 @@ steps:
|
||||
- pytest -v -s v1/worker
|
||||
- pytest -v -s -m 'not cpu_test' v1/kv_connector/unit
|
||||
- pytest -v -s -m 'not cpu_test' v1/metrics
|
||||
- pip install -U git+https://github.com/robertgshaw2-redhat/lm-evaluation-harness.git@streaming-api
|
||||
- pip install -U git+https://github.com/vllm-project/lm-evaluation-harness.git@streaming-api
|
||||
- pytest -v -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
|
||||
|
||||
- label: V1 Sample + Logits # TBD
|
||||
|
||||
@@ -103,7 +103,7 @@ steps:
|
||||
- pytest -v -s -m 'not cpu_test' v1/kv_connector/unit
|
||||
- pytest -v -s -m 'not cpu_test' v1/metrics
|
||||
# Integration test for streaming correctness (requires special branch).
|
||||
- pip install -U git+https://github.com/robertgshaw2-redhat/lm-evaluation-harness.git@streaming-api
|
||||
- pip install -U git+https://github.com/vllm-project/lm-evaluation-harness.git@streaming-api
|
||||
- pytest -v -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
|
||||
mirror:
|
||||
amd:
|
||||
|
||||
@@ -36,10 +36,10 @@ steps:
|
||||
source_file_dependencies:
|
||||
- vllm/
|
||||
- tests/models/test_terratorch.py
|
||||
- tests/models/test_transformers.py
|
||||
- tests/models/transformers/test_backend.py
|
||||
- tests/models/test_registry.py
|
||||
commands:
|
||||
- pytest -v -s models/test_terratorch.py models/test_transformers.py models/test_registry.py
|
||||
- pytest -v -s models/test_terratorch.py models/transformers/test_backend.py models/test_registry.py
|
||||
mirror:
|
||||
amd:
|
||||
device: mi325_1
|
||||
@@ -55,6 +55,7 @@ steps:
|
||||
- vllm/
|
||||
- tests/models/test_utils.py
|
||||
- tests/models/test_vision.py
|
||||
- tests/models/transformers/fusers/
|
||||
device: cpu-small
|
||||
commands:
|
||||
- pytest -v -s models/test_utils.py models/test_vision.py
|
||||
- pytest -v -s models/test_utils.py models/test_vision.py models/transformers/fusers/
|
||||
|
||||
@@ -17,7 +17,7 @@ steps:
|
||||
- TARGET_TEST_SUITE=L4 pytest basic_correctness/ -v -s -m 'distributed(num_gpus=2)'
|
||||
- CUDA_VISIBLE_DEVICES=0,1 pytest -v -s model_executor/model_loader/test_sharded_state_loader.py -m '(not slow_test)'
|
||||
# Avoid importing model tests that cause CUDA reinitialization error
|
||||
- pytest models/test_transformers.py -v -s -m 'distributed(num_gpus=2)'
|
||||
- pytest models/transformers/test_backend.py -v -s -m 'distributed(num_gpus=2)'
|
||||
- pytest models/language -v -s -m 'distributed(num_gpus=2)'
|
||||
- pytest models/multimodal/generation/test_phi4siglip.py -v -s -m 'distributed(num_gpus=2)'
|
||||
- pytest models/multimodal -v -s -m 'distributed(num_gpus=2)' --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_phi4siglip.py
|
||||
|
||||
@@ -27,6 +27,7 @@ steps:
|
||||
- tests/models/multimodal
|
||||
commands:
|
||||
- pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "qwen3 or gemma"
|
||||
- pytest -v -s models/multimodal/generation/test_mm_prefix_lm.py -m core_model
|
||||
- pytest -v -s models/multimodal/generation/test_qwen2_5_vl.py -m core_model
|
||||
mirror:
|
||||
amd:
|
||||
@@ -58,7 +59,7 @@ steps:
|
||||
- vllm/
|
||||
- tests/models/multimodal
|
||||
commands:
|
||||
- pytest -v -s models/multimodal -m core_model --ignore models/multimodal/generation/test_common.py --ignore models/multimodal/generation/test_ultravox.py --ignore models/multimodal/generation/test_qwen2_5_vl.py --ignore models/multimodal/generation/test_qwen2_vl.py --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_memory_leak.py --ignore models/multimodal/generation/test_vit_cudagraph.py --ignore models/multimodal/processing
|
||||
- pytest -v -s models/multimodal -m core_model --ignore models/multimodal/generation/test_common.py --ignore models/multimodal/generation/test_ultravox.py --ignore models/multimodal/generation/test_qwen2_5_vl.py --ignore models/multimodal/generation/test_qwen2_vl.py --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_mm_prefix_lm.py --ignore models/multimodal/generation/test_memory_leak.py --ignore models/multimodal/generation/test_vit_cudagraph.py --ignore models/multimodal/processing
|
||||
- pytest -v -s models/multimodal/generation/test_vit_cudagraph.py -m core_model
|
||||
- pytest models/multimodal/generation/test_memory_leak.py -m core_model
|
||||
- cd .. && VLLM_WORKER_MULTIPROC_METHOD=spawn pytest -v -s tests/models/multimodal/generation/test_whisper.py -m core_model # Otherwise, mp_method="spawn" doesn't work
|
||||
|
||||
@@ -15,26 +15,24 @@ steps:
|
||||
- tests/utils.py
|
||||
- tests/benchmarks/test_serve_cli.py
|
||||
- tests/entrypoints/openai/chat_completion/test_chat_completion.py
|
||||
- tests/entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py
|
||||
# - tests/entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py
|
||||
|
||||
# - tests/entrypoints/openai/completion/test_prompt_validation.py
|
||||
- tests/entrypoints/openai/completion/test_shutdown.py
|
||||
- tests/entrypoints/openai/test_return_token_ids.py
|
||||
- tests/entrypoints/openai/test_uds.py
|
||||
# - tests/entrypoints/openai/test_return_token_ids.py
|
||||
# - tests/entrypoints/openai/test_uds.py
|
||||
- tests/v1/sample/test_logprobs_e2e.py
|
||||
commands:
|
||||
- export VLLM_USE_RUST_FRONTEND=1
|
||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- pytest -v -s benchmarks/test_serve_cli.py -k "not insecure"
|
||||
- pytest -v -s benchmarks/test_serve_cli.py -k "not insecure and not (test_bench_serve and not test_bench_serve_chat)"
|
||||
- pytest -v -s entrypoints/openai/chat_completion/test_chat_completion.py -k "not test_invalid_json_schema and not test_invalid_regex"
|
||||
- pytest -v -s entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py -k "not multiple"
|
||||
# - pytest -v -s entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py -k "not invalid"
|
||||
|
||||
# - pytest -v -s entrypoints/openai/completion/test_prompt_validation.py -k "not prompt_embeds"
|
||||
- pytest -v -s entrypoints/openai/completion/test_shutdown.py -k "not engine_failure and not test_abort_timeout_exits_quickly"
|
||||
# test_comparison streams differently: Rust emits a separate first (prompt_token_ids) chunk and
|
||||
# finish chunk without logprobs, while the test reads `logprobs.tokens` on every chunk.
|
||||
- pytest -v -s entrypoints/openai/test_return_token_ids.py -k "not test_comparison"
|
||||
- pytest -v -s entrypoints/openai/test_uds.py
|
||||
# - pytest -v -s entrypoints/openai/test_return_token_ids.py
|
||||
# - pytest -v -s entrypoints/openai/test_uds.py
|
||||
- pytest -v -s v1/sample/test_logprobs_e2e.py -k "test_prompt_logprobs_e2e_server"
|
||||
|
||||
- label: Rust Frontend Serve/Admin Coverage
|
||||
@@ -47,24 +45,19 @@ steps:
|
||||
- vllm/entrypoints/serve/
|
||||
- vllm/v1/engine/
|
||||
- tests/utils.py
|
||||
- tests/entrypoints/serve/dev/rpc/test_collective_rpc.py
|
||||
# - tests/entrypoints/serve/dev/rpc/test_collective_rpc.py
|
||||
- tests/entrypoints/scale_out/token_in_token_out/test_serving_tokens.py
|
||||
- tests/entrypoints/serve/instrumentator/test_basic.py
|
||||
- tests/entrypoints/serve/instrumentator/test_metrics.py
|
||||
- tests/entrypoints/serve/dev/test_sleep.py
|
||||
- tests/entrypoints/serve/tokenize/test_tokenization.py
|
||||
# - tests/entrypoints/serve/dev/test_sleep.py
|
||||
commands:
|
||||
- export VLLM_USE_RUST_FRONTEND=1
|
||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- pytest -v -s entrypoints/serve/dev/rpc/test_collective_rpc.py
|
||||
# server_load can be flaky under the Rust frontend; keep it excluded for now.
|
||||
- pytest -v -s entrypoints/serve/instrumentator/test_basic.py -k "not server_load"
|
||||
# test_generate_logprobs expects Python-style top_logprobs truncation (dedup sampled + cap at max(k, 1)).
|
||||
- pytest -v -s entrypoints/scale_out/token_in_token_out/test_serving_tokens.py -k "not lora and not test_generate_logprobs and not stop_string_workflow"
|
||||
# - pytest -v -s entrypoints/serve/dev/rpc/test_collective_rpc.py
|
||||
- pytest -v -s entrypoints/serve/instrumentator/test_basic.py -k "not show_version and not server_load"
|
||||
- pytest -v -s entrypoints/scale_out/token_in_token_out/test_serving_tokens.py -k "not stream and not lora and not test_generate_logprobs and not stop_string_workflow"
|
||||
- pytest -v -s entrypoints/serve/instrumentator/test_metrics.py -k "text and not show and not run_batch and not test_metrics_counts and not test_metrics_exist"
|
||||
- pytest -v -s entrypoints/serve/dev/test_sleep.py
|
||||
# /tokenizer_info is not implemented in the Rust frontend (the CLI flag is accepted as a no-op).
|
||||
- pytest -v -s entrypoints/serve/tokenize/test_tokenization.py -k "not tokenizer_info"
|
||||
# - pytest -v -s entrypoints/serve/dev/test_sleep.py
|
||||
|
||||
- label: Rust Frontend Core Correctness
|
||||
timeout_in_minutes: 30
|
||||
@@ -92,7 +85,7 @@ steps:
|
||||
commands:
|
||||
- export VLLM_USE_RUST_FRONTEND=1
|
||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- pytest -v -s tool_use --ignore=tool_use/mistral --models llama3.2
|
||||
- pytest -v -s tool_use --ignore=tool_use/mistral --models llama3.2 -k "not test_response_format_with_tool_choice_required and not test_parallel_tool_calls_false and not test_tool_call_and_choice"
|
||||
|
||||
- label: Rust Frontend Distributed
|
||||
timeout_in_minutes: 30
|
||||
|
||||
+1
-1
@@ -119,7 +119,7 @@
|
||||
|
||||
# Transformers modeling backend
|
||||
/vllm/model_executor/models/transformers @hmellor
|
||||
/tests/models/test_transformers.py @hmellor
|
||||
/tests/models/transformers @hmellor
|
||||
|
||||
# Docs
|
||||
/docs/mkdocs @hmellor
|
||||
|
||||
@@ -400,7 +400,6 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
||||
"csrc/libtorch_stable/topk.cu"
|
||||
"csrc/libtorch_stable/mamba/selective_scan_fwd.cu"
|
||||
"csrc/libtorch_stable/cache_kernels.cu"
|
||||
"csrc/libtorch_stable/cache_kernels.cu"
|
||||
"csrc/libtorch_stable/cache_kernels_fused.cu"
|
||||
"csrc/libtorch_stable/custom_all_reduce.cu"
|
||||
"csrc/libtorch_stable/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu")
|
||||
|
||||
@@ -132,8 +132,10 @@ def benchmark_function(
|
||||
reset_memory_stats()
|
||||
|
||||
# Benchmark
|
||||
start_events = [torch.Event(enable_timing=True) for _ in range(benchmark_iters)]
|
||||
end_events = [torch.Event(enable_timing=True) for _ in range(benchmark_iters)]
|
||||
start_events = [
|
||||
torch.cuda.Event(enable_timing=True) for _ in range(benchmark_iters)
|
||||
]
|
||||
end_events = [torch.cuda.Event(enable_timing=True) for _ in range(benchmark_iters)]
|
||||
|
||||
for i in range(benchmark_iters):
|
||||
logits_copy = logits.clone()
|
||||
|
||||
@@ -134,8 +134,8 @@ def benchmark_config(
|
||||
torch.accelerator.synchronize()
|
||||
|
||||
# Benchmark
|
||||
start = torch.Event(enable_timing=True)
|
||||
end = torch.Event(enable_timing=True)
|
||||
start = torch.cuda.Event(enable_timing=True)
|
||||
end = torch.cuda.Event(enable_timing=True)
|
||||
start.record()
|
||||
for _ in range(num_iters):
|
||||
with override_config(config):
|
||||
|
||||
@@ -170,8 +170,8 @@ def benchmark_config(
|
||||
graph.replay()
|
||||
torch.accelerator.synchronize()
|
||||
|
||||
start = torch.Event(enable_timing=True)
|
||||
end = torch.Event(enable_timing=True)
|
||||
start = torch.cuda.Event(enable_timing=True)
|
||||
end = torch.cuda.Event(enable_timing=True)
|
||||
latencies: list[float] = []
|
||||
for _ in range(num_iters):
|
||||
start.record()
|
||||
|
||||
@@ -15,6 +15,7 @@ endif()
|
||||
#
|
||||
set(ENABLE_X86_ISA $ENV{VLLM_CPU_X86})
|
||||
set(ENABLE_ARM_BF16 $ENV{VLLM_CPU_ARM_BF16})
|
||||
set(ENABLE_RVV_BF16 $ENV{VLLM_CPU_RVV_BF16})
|
||||
|
||||
include_directories("${CMAKE_SOURCE_DIR}/csrc")
|
||||
|
||||
@@ -110,6 +111,13 @@ else()
|
||||
set(ARM_BF16_FOUND ON)
|
||||
message(STATUS "ARM BF16 support enabled via VLLM_CPU_ARM_BF16 environment variable")
|
||||
endif()
|
||||
# Some kernels (e.g. Bianbu on Spacemit X100) do not report zvfbfmin
|
||||
# in /proc/cpuinfo despite hardware support. VLLM_CPU_RVV_BF16=1
|
||||
# overrides the detection result.
|
||||
if (ENABLE_RVV_BF16)
|
||||
set(RVV_BF16_FOUND ON)
|
||||
message(STATUS "RVV BF16 support enabled via VLLM_CPU_RVV_BF16 environment variable")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (CMAKE_SYSTEM_PROCESSOR MATCHES "x86_64|amd64" OR ENABLE_X86_ISA)
|
||||
@@ -178,7 +186,10 @@ elseif (CMAKE_SYSTEM_PROCESSOR MATCHES "riscv64")
|
||||
# Override with -DVLLM_RVV_VLEN=128 or -DVLLM_RVV_VLEN=256 for RVV.
|
||||
if(NOT DEFINED VLLM_RVV_VLEN)
|
||||
# Auto-detect: find the largest zvl<N>b in /proc/cpuinfo isa line.
|
||||
if(EXISTS /proc/cpuinfo)
|
||||
# Skip when cross-compiling — /proc/cpuinfo describes the build host.
|
||||
if(CMAKE_CROSSCOMPILING)
|
||||
message(STATUS "Cross-compiling: skipping VLEN auto-detection from /proc/cpuinfo")
|
||||
elseif(EXISTS /proc/cpuinfo)
|
||||
file(READ /proc/cpuinfo _cpuinfo)
|
||||
set(_best 0)
|
||||
foreach(_n IN ITEMS 128 256 512 1024)
|
||||
@@ -186,6 +197,13 @@ elseif (CMAKE_SYSTEM_PROCESSOR MATCHES "riscv64")
|
||||
set(_best ${_n})
|
||||
endif()
|
||||
endforeach()
|
||||
# Only VLEN=128 and VLEN=256 are supported by the RVV kernels.
|
||||
if(_best GREATER 256)
|
||||
message(WARNING
|
||||
"Detected VLEN=${_best} but only 128/256 are supported; "
|
||||
"clamping to 256")
|
||||
set(_best 256)
|
||||
endif()
|
||||
if(_best GREATER 0)
|
||||
set(VLLM_RVV_VLEN ${_best})
|
||||
endif()
|
||||
@@ -195,9 +213,9 @@ elseif (CMAKE_SYSTEM_PROCESSOR MATCHES "riscv64")
|
||||
if(NOT DEFINED VLLM_RVV_VLEN AND (RVV_FP16_FOUND OR RVV_BF16_FOUND))
|
||||
message(FATAL_ERROR
|
||||
"RISC-V RVV is available but VLEN could not be auto-detected. "
|
||||
"Please specify VLEN explicitly:\n"
|
||||
" -DVLLM_RVV_VLEN=128 (for VLEN=128 hardware)\n"
|
||||
" -DVLLM_RVV_VLEN=256 (for VLEN=256 hardware, e.g. Spacemit X100)")
|
||||
"Please specify VLEN explicitly via CMAKE_ARGS:\n"
|
||||
" CMAKE_ARGS='-DVLLM_RVV_VLEN=128' (for VLEN=128 hardware)\n"
|
||||
" CMAKE_ARGS='-DVLLM_RVV_VLEN=256' (for VLEN=256 hardware, e.g. Spacemit X100)")
|
||||
endif()
|
||||
endif()
|
||||
if(VLLM_RVV_VLEN AND VLLM_RVV_VLEN GREATER 0)
|
||||
@@ -209,7 +227,7 @@ elseif (CMAKE_SYSTEM_PROCESSOR MATCHES "riscv64")
|
||||
message(STATUS "BF16 extension detected")
|
||||
set(MARCH_FLAGS -march=rv64gcv_zvfh_zfbfmin_zvfbfmin_zvl${VLLM_RVV_VLEN}b -mrvv-vector-bits=zvl -mabi=lp64d)
|
||||
elseif(RVV_FP16_FOUND)
|
||||
message(WARNING "BF16 functionality is not available")
|
||||
message(WARNING "BF16 functionality is not available.")
|
||||
set(MARCH_FLAGS -march=rv64gcv_zvfh_zvl${VLLM_RVV_VLEN}b -mrvv-vector-bits=zvl -mabi=lp64d)
|
||||
else()
|
||||
message(STATUS "compile riscv with scalar (no FP16/BF16)")
|
||||
|
||||
@@ -13,7 +13,8 @@ static inline cpu_attention::Fp8KVCacheDataType parse_fp8_kv_dtype(
|
||||
|
||||
bool cpu_attn_has_isa(const std::string& isa) {
|
||||
if (isa == "rvv") {
|
||||
#if defined(__riscv) && defined(__riscv_v_min_vlen) && __riscv_v_min_vlen == 128
|
||||
#if defined(__riscv) && defined(__riscv_v_min_vlen) && \
|
||||
(__riscv_v_min_vlen == 128 || __riscv_v_min_vlen == 256)
|
||||
return true;
|
||||
#else
|
||||
return false;
|
||||
|
||||
+56
-14
@@ -417,8 +417,10 @@ class AttentionScheduler {
|
||||
has_decode_request = has_decode_request || (q_token_num == 1);
|
||||
decode_only_batch = decode_only_batch && (q_token_num == 1);
|
||||
}
|
||||
int32_t q_head_per_kv = input.num_heads_q / input.num_heads_kv;
|
||||
const bool supports_gqa = q_head_per_kv <= max_num_q_per_iter;
|
||||
const int32_t original_q_head_per_kv =
|
||||
input.num_heads_q / input.num_heads_kv;
|
||||
int32_t q_head_per_kv = original_q_head_per_kv;
|
||||
const bool supports_gqa = original_q_head_per_kv <= max_num_q_per_iter;
|
||||
const bool use_gqa_fast_path = supports_gqa && decode_only_batch;
|
||||
const bool use_gqa_scratchpad = supports_gqa && has_decode_request;
|
||||
if (!use_gqa_scratchpad) {
|
||||
@@ -671,22 +673,62 @@ class AttentionScheduler {
|
||||
metadata_ptr->effective_thread_num = effective_thread_num;
|
||||
|
||||
{
|
||||
// when q_tile_size = max_num_q_per_iter, requires max
|
||||
// attention_scratchpad_size
|
||||
AttentionScratchPad sc(0, *metadata_ptr, 0x0);
|
||||
int64_t n = AttentionScheduler::calcu_tile_size_with_constant_q(
|
||||
cache_size, input.head_dim, input.elem_size, input.q_buffer_elem_size,
|
||||
input.logits_buffer_elem_size, input.output_buffer_elem_size,
|
||||
max_num_q_per_iter, kv_len_alignment, max_num_q_per_iter, true);
|
||||
sc.update(input.head_dim, input.q_buffer_elem_size,
|
||||
input.logits_buffer_elem_size, input.output_buffer_elem_size,
|
||||
max_num_q_per_iter, max_num_q_per_iter, n);
|
||||
int64_t max_attention_scratchpad_size = 0;
|
||||
|
||||
for (const AttentionWorkItemGroup& item : workitems) {
|
||||
const bool curr_use_gqa =
|
||||
use_gqa_fast_path || (supports_gqa && item.q_token_num == 1);
|
||||
const int32_t curr_q_heads_per_kv =
|
||||
curr_use_gqa ? original_q_head_per_kv : 1;
|
||||
const int32_t curr_default_q_tile_token_num =
|
||||
default_tile_size / curr_q_heads_per_kv;
|
||||
|
||||
for (int32_t q_token_offset = 0; q_token_offset < item.q_token_num;
|
||||
q_token_offset += curr_default_q_tile_token_num) {
|
||||
const int32_t actual_q_token_num = std::min(
|
||||
curr_default_q_tile_token_num, item.q_token_num - q_token_offset);
|
||||
const int32_t q_head_tile_size =
|
||||
actual_q_token_num * curr_q_heads_per_kv;
|
||||
const int32_t rounded_q_head_tile_size =
|
||||
((q_head_tile_size + max_num_q_per_iter - 1) /
|
||||
max_num_q_per_iter) *
|
||||
max_num_q_per_iter;
|
||||
|
||||
const int64_t n = AttentionScheduler::calcu_tile_size_with_constant_q(
|
||||
cache_size, input.head_dim, input.elem_size,
|
||||
input.q_buffer_elem_size, input.logits_buffer_elem_size,
|
||||
input.output_buffer_elem_size, max_num_q_per_iter,
|
||||
kv_len_alignment, rounded_q_head_tile_size,
|
||||
rounded_q_head_tile_size <= max_num_q_per_iter);
|
||||
|
||||
sc.update(input.head_dim, input.q_buffer_elem_size,
|
||||
input.logits_buffer_elem_size,
|
||||
input.output_buffer_elem_size, max_num_q_per_iter,
|
||||
rounded_q_head_tile_size, n);
|
||||
|
||||
max_attention_scratchpad_size = std::max(
|
||||
max_attention_scratchpad_size, sc.get_thread_scratchpad_size());
|
||||
}
|
||||
}
|
||||
|
||||
metadata_ptr->attention_scratchpad_size_per_thread =
|
||||
((sc.get_thread_scratchpad_size() + 63) / 64) * 64;
|
||||
((max_attention_scratchpad_size + 63) / 64) * 64;
|
||||
|
||||
int32_t max_reduction_q_head_tile_size = 0;
|
||||
for (const ReductionWorkItemGroup& item : reduce_workitems) {
|
||||
const bool curr_use_gqa =
|
||||
use_gqa_fast_path || (supports_gqa && item.q_token_id_num == 1);
|
||||
const int32_t curr_q_heads_per_kv =
|
||||
curr_use_gqa ? original_q_head_per_kv : 1;
|
||||
|
||||
max_reduction_q_head_tile_size =
|
||||
std::max(max_reduction_q_head_tile_size,
|
||||
item.q_token_id_num * curr_q_heads_per_kv);
|
||||
}
|
||||
|
||||
sc.update(0, metadata_ptr->reduction_split_num, input.head_dim,
|
||||
q_head_per_kv * split_kv_q_token_num_threshold,
|
||||
input.output_buffer_elem_size);
|
||||
max_reduction_q_head_tile_size, input.output_buffer_elem_size);
|
||||
metadata_ptr->reduction_scratchpad_size_per_kv_head =
|
||||
((sc.get_reduction_scratchpad_size() + 63) / 64) * 64;
|
||||
}
|
||||
|
||||
@@ -214,11 +214,18 @@ struct BF16Vec32 : public Vec<BF16Vec32> {
|
||||
|
||||
explicit BF16Vec32(const BF16Vec8& v) {
|
||||
fixed_u16x8_t u16_val = bf16_to_u16(v.reg);
|
||||
fixed_u16x32_t u16_combined =
|
||||
RVVI4(__riscv_vcreate_v_u16, LMUL_128, _u16, LMUL_512)(
|
||||
u16_val, u16_val, u16_val, u16_val);
|
||||
reg = RVVI4(__riscv_vreinterpret_v_u16, LMUL_512, _bf16,
|
||||
LMUL_512)(u16_combined);
|
||||
// Widen LMUL_128 → LMUL_256 so vslideup operands share a type.
|
||||
// At VLEN=256 this is mf2→m1 (both integer); at VLEN=128 it is m1→m2.
|
||||
fixed_u16x16_t ext =
|
||||
RVVI4(__riscv_vlmul_ext_v_u16, LMUL_128, _u16, LMUL_256)(u16_val);
|
||||
// Build 16-element half: place the 8 elements at offsets 0 and 8.
|
||||
fixed_u16x16_t half = RVVI(__riscv_vmv_v_x_u16, LMUL_256)(0, 16);
|
||||
half = RVVI(__riscv_vslideup_vx_u16, LMUL_256)(half, ext, 0, 8);
|
||||
half = RVVI(__riscv_vslideup_vx_u16, LMUL_256)(half, ext, 8, 16);
|
||||
// Double to LMUL_512 (m1→m2 at VLEN=256, m2→m4 at VLEN=128).
|
||||
fixed_u16x32_t dst =
|
||||
RVVI4(__riscv_vcreate_v_u16, LMUL_256, _u16, LMUL_512)(half, half);
|
||||
reg = RVVI4(__riscv_vreinterpret_v_u16, LMUL_512, _bf16, LMUL_512)(dst);
|
||||
};
|
||||
|
||||
void save(void* ptr) const {
|
||||
@@ -623,17 +630,29 @@ struct FP32Vec16 : public Vec<FP32Vec16> {
|
||||
data.reg, data.reg)) {};
|
||||
explicit FP32Vec16(const FP32Vec16& data) : reg(data.reg) {};
|
||||
explicit FP32Vec16(int64_t value, const FP32Vec16& lut) {
|
||||
const uint64_t q_values = static_cast<uint64_t>(value);
|
||||
auto packed = RVVI(__riscv_vmv_v_x_u64, LMUL_1024)(q_values, VEC_ELEM_NUM);
|
||||
auto lane_ids = RVVI(__riscv_vid_v_u64, LMUL_1024)(VEC_ELEM_NUM);
|
||||
auto shifts =
|
||||
RVVI(__riscv_vsll_vx_u64, LMUL_1024)(lane_ids, 2, VEC_ELEM_NUM);
|
||||
auto shifted =
|
||||
RVVI(__riscv_vsrl_vv_u64, LMUL_1024)(packed, shifts, VEC_ELEM_NUM);
|
||||
auto idx64 =
|
||||
RVVI(__riscv_vand_vx_u64, LMUL_1024)(shifted, 0xF, VEC_ELEM_NUM);
|
||||
auto idx32 = RVVI(__riscv_vnsrl_wx_u32, LMUL_512)(idx64, 0, VEC_ELEM_NUM);
|
||||
reg = RVVI(__riscv_vrgather_vv_f32, LMUL_512)(lut.reg, idx32, VEC_ELEM_NUM);
|
||||
// Split into two 32-bit halves to avoid u64 @ LMUL_1024 (m8 on
|
||||
// VLEN=128 / m4 on VLEN=256), which causes heavy register spilling.
|
||||
constexpr int HALF = VEC_ELEM_NUM / 2;
|
||||
const auto q = static_cast<uint64_t>(value);
|
||||
const uint32_t lo = static_cast<uint32_t>(q);
|
||||
const uint32_t hi = static_cast<uint32_t>(q >> 32);
|
||||
|
||||
auto lane_ids = RVVI(__riscv_vid_v_u32, LMUL_256)(HALF);
|
||||
auto shifts = RVVI(__riscv_vsll_vx_u32, LMUL_256)(lane_ids, 2, HALF);
|
||||
|
||||
auto packed_lo = RVVI(__riscv_vmv_v_x_u32, LMUL_256)(lo, HALF);
|
||||
auto idx_lo = RVVI(__riscv_vand_vx_u32, LMUL_256)(
|
||||
RVVI(__riscv_vsrl_vv_u32, LMUL_256)(packed_lo, shifts, HALF), 0xF,
|
||||
HALF);
|
||||
|
||||
auto packed_hi = RVVI(__riscv_vmv_v_x_u32, LMUL_256)(hi, HALF);
|
||||
auto idx_hi = RVVI(__riscv_vand_vx_u32, LMUL_256)(
|
||||
RVVI(__riscv_vsrl_vv_u32, LMUL_256)(packed_hi, shifts, HALF), 0xF,
|
||||
HALF);
|
||||
|
||||
auto idx =
|
||||
RVVI4(__riscv_vcreate_v_u32, LMUL_256, _u32, LMUL_512)(idx_lo, idx_hi);
|
||||
reg = RVVI(__riscv_vrgather_vv_f32, LMUL_512)(lut.reg, idx, VEC_ELEM_NUM);
|
||||
}
|
||||
explicit FP32Vec16(const FP16Vec16& v);
|
||||
|
||||
|
||||
@@ -278,7 +278,8 @@ TORCH_LIBRARY_EXPAND(TORCH_EXTENSION_NAME, ops) {
|
||||
ops.def(
|
||||
"dynamic_4bit_int_moe("
|
||||
"Tensor x, Tensor topk_ids, Tensor topk_weights,"
|
||||
"Tensor w13_packed, Tensor w2_packed, int H, int I, int I2,"
|
||||
"Tensor w13_packed, Tensor w2_packed,"
|
||||
"int hidden_size, int intermediate_size,"
|
||||
"int group_size, bool apply_router_weight_on_input, int activation_kind"
|
||||
") -> Tensor");
|
||||
|
||||
|
||||
@@ -29,25 +29,37 @@ enum ActivationKind : int64_t {
|
||||
|
||||
torch::Tensor dynamic_4bit_int_moe_cpu(
|
||||
torch::Tensor x, torch::Tensor topk_ids, torch::Tensor topk_weights,
|
||||
torch::Tensor w13_packed, torch::Tensor w2_packed, int64_t H, int64_t I,
|
||||
int64_t I2, int64_t group_size, bool apply_router_weight_on_input,
|
||||
int64_t activation_kind) {
|
||||
torch::Tensor w13_packed, torch::Tensor w2_packed, int64_t hidden_size,
|
||||
int64_t intermediate_size, int64_t group_size,
|
||||
bool apply_router_weight_on_input, int64_t activation_kind) {
|
||||
TORCH_CHECK(x.dim() == 2, "x must be 2D");
|
||||
TORCH_CHECK(topk_ids.dim() == 2 && topk_weights.dim() == 2,
|
||||
"topk tensors must be [T, K]");
|
||||
TORCH_CHECK(
|
||||
w13_packed.size(0) == w2_packed.size(0),
|
||||
"w13_packed and w2_packed must have same number of experts in dim 0");
|
||||
TORCH_CHECK(I2 == 2 * I, "I2 must equal 2*I");
|
||||
|
||||
const int64_t T = x.size(0);
|
||||
const int64_t K = topk_ids.size(1);
|
||||
const int64_t E = w13_packed.size(0);
|
||||
const int64_t N = T * K;
|
||||
const int64_t w13_out_features = 2 * intermediate_size;
|
||||
|
||||
auto x_c = x.contiguous();
|
||||
// _dyn_quant_matmul_4bit kernel natively supports these pre-quant activation
|
||||
// dtypes:
|
||||
// - fp32: with channelwise and groupwise
|
||||
// - bf16: with channelwise -> upcast to fp32 for groupwise
|
||||
// - fp16: not supported -> upcast to fp32 for groupwise & channelwise
|
||||
const auto output_dtype = x_c.scalar_type();
|
||||
const bool should_cast_input =
|
||||
((group_size != -1) && output_dtype == at::kBFloat16) ||
|
||||
output_dtype == at::kHalf;
|
||||
if (should_cast_input) {
|
||||
x_c = x_c.to(at::kFloat);
|
||||
}
|
||||
auto ids_c = topk_ids.contiguous();
|
||||
auto gates_c = topk_weights.to(at::kFloat).contiguous();
|
||||
auto gates_c = topk_weights.to(x_c.scalar_type()).contiguous();
|
||||
|
||||
// bucketing tokens -> experts
|
||||
c10::SmallVector<int64_t, 64> counts(
|
||||
@@ -63,35 +75,42 @@ torch::Tensor dynamic_4bit_int_moe_cpu(
|
||||
c10::SmallVector<int64_t, 65> offsets(E + 1, 0); // ( E +1 )
|
||||
for (int64_t e = 0; e < E; ++e) offsets[e + 1] = offsets[e] + counts[e];
|
||||
|
||||
// expert_tokens = [tokens indices for expert 0, ...]
|
||||
// expert_gates = [router weights for tokens assigned to expert 0, ...]
|
||||
auto expert_tokens = at::empty({offsets[E]}, ids_c.options());
|
||||
auto expert_gates = at::empty({offsets[E]}, gates_c.options());
|
||||
{
|
||||
c10::SmallVector<int64_t, 64> cursor(E, 0);
|
||||
const auto* ids_ptr = ids_c.data_ptr<int64_t>();
|
||||
const auto* gts_ptr = gates_c.data_ptr<float>();
|
||||
auto* tok_ptr = expert_tokens.data_ptr<int64_t>();
|
||||
auto* gate_ptr = expert_gates.data_ptr<float>();
|
||||
AT_DISPATCH_FLOATING_TYPES_AND2(
|
||||
at::ScalarType::BFloat16, at::ScalarType::Half, gates_c.scalar_type(),
|
||||
"bucket_expert_tokens_and_gates", [&] {
|
||||
const auto* ids_ptr = ids_c.data_ptr<int64_t>();
|
||||
const auto* gts_ptr = gates_c.data_ptr<scalar_t>();
|
||||
auto* tok_ptr = expert_tokens.data_ptr<int64_t>();
|
||||
auto* gate_ptr = expert_gates.data_ptr<scalar_t>();
|
||||
|
||||
for (int64_t t = 0; t < T; ++t) {
|
||||
const int64_t base = t * K;
|
||||
for (int64_t k = 0; k < K; ++k) {
|
||||
const int64_t idx = base + k;
|
||||
const int64_t e = ids_ptr[idx];
|
||||
const int64_t p = offsets[e] + (cursor[e]++);
|
||||
tok_ptr[p] = t;
|
||||
gate_ptr[p] = gts_ptr[idx];
|
||||
}
|
||||
}
|
||||
for (int64_t t = 0; t < T; ++t) {
|
||||
const int64_t base = t * K;
|
||||
for (int64_t k = 0; k < K; ++k) {
|
||||
const int64_t idx = base + k;
|
||||
const int64_t e = ids_ptr[idx];
|
||||
const int64_t p = offsets[e] + (cursor[e]++);
|
||||
tok_ptr[p] = t;
|
||||
gate_ptr[p] = gts_ptr[idx];
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
const int64_t g_eff_13 = (group_size != -1) ? group_size : H;
|
||||
const int64_t g_eff_2 = (group_size != -1) ? group_size : I;
|
||||
const int64_t g_eff_13 = (group_size != -1) ? group_size : hidden_size;
|
||||
const int64_t g_eff_2 = (group_size != -1) ? group_size : intermediate_size;
|
||||
|
||||
// X_all [num_tokens * K, hidden_size]
|
||||
auto X_all = x_c.index_select(/*dim=*/0, expert_tokens);
|
||||
if (apply_router_weight_on_input) {
|
||||
X_all = X_all.mul(expert_gates.unsqueeze(1));
|
||||
}
|
||||
auto Y_all = at::empty({offsets[E], H}, x_c.options());
|
||||
auto Y_all = at::empty({offsets[E], hidden_size}, x_c.options());
|
||||
|
||||
at::parallel_for(0, offsets[E], 0, [&](int64_t idx_begin, int64_t idx_end) {
|
||||
c10::InferenceMode guard;
|
||||
@@ -109,11 +128,13 @@ torch::Tensor dynamic_4bit_int_moe_cpu(
|
||||
auto w2_e = w2_packed.select(/*dim=*/0, e);
|
||||
|
||||
// W13
|
||||
auto y13 =
|
||||
mm(x_e, w13_e, g_eff_13, /*in_features=*/H, /*out_features=*/I2);
|
||||
auto y13 = mm(x_e, w13_e, g_eff_13, /*in_features=*/hidden_size,
|
||||
/*out_features=*/w13_out_features);
|
||||
|
||||
auto g_part = y13.narrow(/*dim=*/1, /*start=*/0, /*length=*/I);
|
||||
auto u_part = y13.narrow(/*dim=*/1, /*start=*/I, /*length=*/I);
|
||||
auto g_part =
|
||||
y13.narrow(/*dim=*/1, /*start=*/0, /*length=*/intermediate_size);
|
||||
auto u_part = y13.narrow(/*dim=*/1, /*start=*/intermediate_size,
|
||||
/*length=*/intermediate_size);
|
||||
|
||||
torch::Tensor act;
|
||||
if (activation_kind == ActivationKind::SwiGLUOAI) { // SwiGLUOAI
|
||||
@@ -128,7 +149,8 @@ torch::Tensor dynamic_4bit_int_moe_cpu(
|
||||
}
|
||||
|
||||
// W2
|
||||
auto y = mm(act, w2_e, g_eff_2, /*in_features=*/I, /*out_features=*/H);
|
||||
auto y = mm(act, w2_e, g_eff_2, /*in_features=*/intermediate_size,
|
||||
/*out_features=*/hidden_size);
|
||||
|
||||
// Store per-expert result
|
||||
Y_all.narrow(/*dim=*/0, /*start=*/start, /*length=*/te).copy_(y);
|
||||
@@ -138,8 +160,11 @@ torch::Tensor dynamic_4bit_int_moe_cpu(
|
||||
if (!apply_router_weight_on_input) {
|
||||
Y_all = Y_all.mul(expert_gates.unsqueeze(1));
|
||||
}
|
||||
if (Y_all.scalar_type() != output_dtype) {
|
||||
Y_all = Y_all.to(output_dtype);
|
||||
}
|
||||
|
||||
auto out = at::zeros({T, H}, x.options());
|
||||
auto out = at::zeros({T, hidden_size}, x.options());
|
||||
out =
|
||||
at::index_add(out, /*dim=*/0, /*index=*/expert_tokens, /*source=*/Y_all);
|
||||
|
||||
|
||||
+3
-3
@@ -53,9 +53,9 @@ void dynamic_scaled_int8_quant(torch::Tensor& out, torch::Tensor const& input,
|
||||
|
||||
torch::Tensor dynamic_4bit_int_moe_cpu(
|
||||
torch::Tensor x, torch::Tensor topk_ids, torch::Tensor topk_weights,
|
||||
torch::Tensor w13_packed, torch::Tensor w2_packed, int64_t H, int64_t I,
|
||||
int64_t I2, int64_t group_size, bool apply_router_weight_on_input,
|
||||
int64_t activation_kind);
|
||||
torch::Tensor w13_packed, torch::Tensor w2_packed, int64_t hidden_size,
|
||||
int64_t intermediate_size, int64_t group_size,
|
||||
bool apply_router_weight_on_input, int64_t activation_kind);
|
||||
|
||||
using fptr_t = int64_t;
|
||||
#ifdef USE_ROCM
|
||||
|
||||
+10
-7
@@ -25,7 +25,6 @@ FROM ubuntu:22.04 AS base-common
|
||||
WORKDIR /workspace
|
||||
|
||||
ARG PYTHON_VERSION=3.12
|
||||
ARG PIP_EXTRA_INDEX_URL="https://download.pytorch.org/whl/cpu"
|
||||
|
||||
ARG max_jobs=32
|
||||
ENV MAX_JOBS=${max_jobs}
|
||||
@@ -53,8 +52,6 @@ ENV PATH="$VIRTUAL_ENV/bin:$PATH"
|
||||
ENV UV_HTTP_TIMEOUT=500
|
||||
|
||||
# Install Python dependencies
|
||||
ENV PIP_EXTRA_INDEX_URL=${PIP_EXTRA_INDEX_URL}
|
||||
ENV UV_EXTRA_INDEX_URL=${PIP_EXTRA_INDEX_URL}
|
||||
ENV UV_INDEX_STRATEGY="unsafe-best-match"
|
||||
ENV UV_LINK_MODE="copy"
|
||||
|
||||
@@ -64,7 +61,7 @@ COPY requirements/cpu.txt requirements/cpu.txt
|
||||
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv pip install --upgrade pip && \
|
||||
uv pip install -r requirements/cpu.txt
|
||||
uv pip install -r requirements/cpu.txt --torch-backend cpu
|
||||
|
||||
ARG TARGETARCH
|
||||
ENV TARGETARCH=${TARGETARCH}
|
||||
@@ -149,7 +146,7 @@ RUN if [ "$TARGETARCH" = "arm64" ] && [ "$VLLM_CPU_X86" != "0" ]; then \
|
||||
COPY requirements/build/cpu.txt requirements/build/cpu.txt
|
||||
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv pip install -r requirements/build/cpu.txt
|
||||
uv pip install -r requirements/build/cpu.txt --torch-backend cpu
|
||||
|
||||
COPY . .
|
||||
|
||||
@@ -205,7 +202,7 @@ RUN case "$(uname -m)" in \
|
||||
esac
|
||||
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv pip install -r requirements/test/cpu.txt
|
||||
uv pip install -r requirements/test/cpu.txt --torch-backend cpu
|
||||
|
||||
######################### DEV IMAGE #########################
|
||||
FROM vllm-build AS vllm-dev
|
||||
@@ -231,7 +228,7 @@ COPY --from=vllm-test-deps /vllm-workspace/requirements/test/cpu.txt requirement
|
||||
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv pip install -r requirements/lint.txt && \
|
||||
uv pip install -r requirements/test/cpu.txt && \
|
||||
uv pip install -r requirements/test/cpu.txt --torch-backend cpu && \
|
||||
pre-commit install --hook-type pre-commit --hook-type commit-msg
|
||||
|
||||
ENTRYPOINT ["bash"]
|
||||
@@ -301,6 +298,12 @@ LABEL ai.vllm.build.cpu-x86="${VLLM_CPU_X86:-false}"
|
||||
LABEL ai.vllm.build.cpu-arm-bf16="${VLLM_CPU_ARM_BF16:-false}"
|
||||
LABEL ai.vllm.build.python-version="${PYTHON_VERSION:-3.12}"
|
||||
|
||||
# Copy the examples directory (including the chat/tool templates) so it is
|
||||
# present in the released image, as the CUDA image ships it too. The vllm-test
|
||||
# stage above adds examples/ for testing only, so without this the published
|
||||
# vllm-openai-cpu image would not ship examples/*.jinja.
|
||||
COPY examples examples
|
||||
|
||||
ENTRYPOINT ["vllm", "serve"]
|
||||
|
||||
|
||||
|
||||
@@ -249,7 +249,7 @@ RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
NUMBA_WHL_FILE=$(ls /tmp/numba-wheels/*.whl) && \
|
||||
OPENCV_WHL_FILE=$(ls /tmp/opencv-wheels/*.whl) && \
|
||||
GUIDANCE_WHL_FILE=$(ls /tmp/guidance-wheels/*.whl) && \
|
||||
uv pip install -v \
|
||||
uv pip install -v \
|
||||
$ARROW_WHL_FILE \
|
||||
$VISION_WHL_FILE \
|
||||
$HF_XET_WHL_FILE \
|
||||
@@ -257,6 +257,7 @@ RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
$NUMBA_WHL_FILE \
|
||||
$OPENCV_WHL_FILE \
|
||||
$GUIDANCE_WHL_FILE \
|
||||
--torch-backend cpu \
|
||||
--index-strategy unsafe-best-match \
|
||||
-r requirements/build/cpu.txt \
|
||||
-r requirements/cpu.txt
|
||||
|
||||
@@ -1337,7 +1337,7 @@ Serve and benchmark VLM2Vec:
|
||||
# Run this in another process
|
||||
vllm serve TIGER-Lab/VLM2Vec-Full --runner pooling \
|
||||
--trust-remote-code \
|
||||
--chat-template examples/template_vlm2vec_phi3v.jinja
|
||||
--chat-template examples/pooling/embed/template/vlm2vec_phi3v.jinja
|
||||
|
||||
# Run these one by one after the server is up
|
||||
# download dataset
|
||||
|
||||
@@ -276,8 +276,9 @@ By default vLLM uses the standard Hugging Face `tokenizers` library to power
|
||||
the fast tokenizer. For BPE tokenizers (Qwen, Llama, DeepSeek, GPT-OSS, etc.)
|
||||
you can switch to the [fastokens](https://github.com/crusoecloud/fastokens)
|
||||
Rust backend, a drop-in replacement that's substantially faster on
|
||||
encode/decode and on streaming detokenization. Enable it by setting
|
||||
`VLLM_USE_FASTOKENS=1`:
|
||||
encode/decode and on streaming detokenization. `VLLM_USE_FASTOKENS` is
|
||||
available in vLLM v0.23.0 and later. If your installed vLLM version does not
|
||||
recognize the environment variable, upgrade vLLM before enabling the override:
|
||||
|
||||
```console
|
||||
VLLM_USE_FASTOKENS=1 vllm serve Qwen/Qwen3-8B
|
||||
|
||||
@@ -168,7 +168,7 @@ Priority is **1 = highest** (tried first).
|
||||
| `FLASH_ATTN` | FA4* | fp16, bf16 | `auto`, `float16`, `bfloat16` | %16 | Any | ✅ | ✅ | ❌ | ✅ | All | ≥10.0 |
|
||||
| `FLASH_ATTN_DIFFKV` | | fp16, bf16 | `auto` | Any | Any | ❌ | ❌ | ❌ | ✅ | Decoder | Any |
|
||||
| `FLEX_ATTENTION` | | fp16, bf16, fp32 | `auto`, `float16`, `bfloat16` | %16 | Any | ❌ | ✅ | ✅ | ❌ | Decoder, Encoder Only | Any |
|
||||
| `HPC_ATTN` | | fp16, bf16 | `auto`, `fp8_e4m3` | 64 | 128 | ❌ | ❌ | ❌ | ❌ | Decoder | ≥9.0 |
|
||||
| `HPC_ATTN` | | fp16, bf16 | `auto`, `bfloat16`, `fp8_e4m3` | 64 | 128 | ❌ | ❌ | ❌ | ❌ | Decoder | ≥9.0 |
|
||||
| `ROCM_AITER_FA` | | fp16, bf16 | `auto`, `float16`, `bfloat16`, `fp8`, `fp8_e4m3`, `fp8_e5m2` | 16, 32 | 64, 128, 256 | ✅ | ✅ | ❌ | ❌ | Decoder | N/A |
|
||||
| `ROCM_AITER_UNIFIED_ATTN` | | bf16 | `auto`, `bfloat16`, `fp8`, `fp8_e4m3` | %16 | Any | ✅ | ❌ | ✅ | ❌ | All | N/A |
|
||||
| `ROCM_ATTN` | | fp16, bf16, fp32 | `auto`, `float16`, `bfloat16`, `fp8`, `fp8_e4m3`, `fp8_e5m2` | %16 | 32, 64, 80, 96, 128, 160, 192, 224, 256 | ❌ | ✅ | ✅ | ❌ | Decoder, Encoder, Encoder Only | N/A |
|
||||
|
||||
@@ -75,14 +75,11 @@ vllm serve Intel/DeepSeek-R1-0528-Qwen3-8B-int4-AutoRound \
|
||||
--max-model-len 4096
|
||||
```
|
||||
|
||||
!!! note
|
||||
To deploy `wNa16` models on Intel GPU/CPU, please add `--enforce-eager` for now.
|
||||
|
||||
## Evaluating the Quantized Model with vLLM
|
||||
|
||||
```bash
|
||||
lm_eval --model vllm \
|
||||
--model_args pretrained="Intel/DeepSeek-R1-0528-Qwen3-8B-int4-AutoRound,max_model_len=8192,max_num_batched_tokens=32768,max_num_seqs=128,gpu_memory_utilization=0.8,dtype=bfloat16,max_gen_toks=2048,enforce_eager=True" \
|
||||
--model_args pretrained="Intel/DeepSeek-R1-0528-Qwen3-8B-int4-AutoRound,max_model_len=8192,max_num_batched_tokens=32768,max_num_seqs=128,gpu_memory_utilization=0.8,dtype=bfloat16,max_gen_toks=2048" \
|
||||
--tasks gsm8k \
|
||||
--num_fewshot 5 \
|
||||
--batch_size 128
|
||||
|
||||
@@ -62,6 +62,8 @@ weight name. Unset fields fall back to the `--quantization` shorthand's
|
||||
defaults, or for already-quantized checkpoints to whatever the checkpoint
|
||||
declares.
|
||||
|
||||
On XPU, non-block FP8 scaled-mm linear layers default to W8A16; setting `--linear-backend xpu` forces W8A8. Use `--linear-backend xpu_woq` to explicitly select weight-only quantization (W8A16).
|
||||
|
||||
The CLI accepts the same shape as JSON or as dotted keys:
|
||||
|
||||
```bash
|
||||
|
||||
@@ -49,6 +49,32 @@ You can configure how the quantization scales are computed in vLLM using three d
|
||||
- `kv_cache_dtype="fp8_e4m3"`: Supported on CUDA 11.8+ and ROCm (AMD GPUs)
|
||||
- `kv_cache_dtype="fp8_e5m2"`: Supported on CUDA 11.8+
|
||||
|
||||
### Skipping Specific Layers from KV-Cache Quantization
|
||||
|
||||
Some attention layer types (e.g. sliding-window) are more sensitive to KV-cache quantization. The `--kv-cache-dtype-skip-layers` flag leaves the specified layers at the model's native dtype while keeping the rest of the layers under the chosen quantized dtype. The flag accepts either layer indices or layer-type names:
|
||||
|
||||
```bash
|
||||
# Skip every sliding-window attention layer.
|
||||
vllm serve <model> \
|
||||
--kv-cache-dtype fp8 \
|
||||
--kv-cache-dtype-skip-layers sliding_window
|
||||
|
||||
# Skip specific layer indices.
|
||||
vllm serve <model> \
|
||||
--kv-cache-dtype fp8 \
|
||||
--kv-cache-dtype-skip-layers 0 1 23
|
||||
```
|
||||
|
||||
Programmatic usage:
|
||||
|
||||
```python
|
||||
llm = LLM(
|
||||
model="meta-llama/Llama-3.1-8B-Instruct",
|
||||
kv_cache_dtype="fp8",
|
||||
kv_cache_dtype_skip_layers=["sliding_window"],
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Examples
|
||||
|
||||
@@ -71,8 +71,5 @@ VLLM_USE_V2_MODEL_RUNNER=0 vllm serve meta-llama/Llama-3.1-8B-Instruct \
|
||||
|
||||
## Limitations
|
||||
|
||||
* only tested with Eagle and Eagle-3. Other SD methods may or may not work out of the box
|
||||
* only usable with Model Runner V1
|
||||
* not compatible with full cuda graph so we force piece-wise cuda graph with this feature
|
||||
|
||||
We are working on enabling it on MRv2 with full cuda graph support.
|
||||
* Tested with Eagle, Eagle-3, and DFlash. Other SD methods may or may not work out of the box
|
||||
* Full Cudagraph only works with Model Runner V2. MRv1 only supports piece-wise cuda graph with this feature
|
||||
|
||||
@@ -35,15 +35,10 @@ After installation of XCode and the Command Line Tools, which include Apple Clan
|
||||
```bash
|
||||
git clone https://github.com/vllm-project/vllm.git
|
||||
cd vllm
|
||||
uv pip install -r requirements/cpu.txt --index-strategy unsafe-best-match
|
||||
uv pip install -r requirements/cpu.txt
|
||||
uv pip install -e .
|
||||
```
|
||||
|
||||
!!! tip
|
||||
The `--index-strategy unsafe-best-match` flag is needed to resolve dependencies across multiple package indexes (PyTorch CPU index and PyPI). Without this flag, you may encounter `typing-extensions` version conflicts.
|
||||
|
||||
The term "unsafe" refers to the package resolution strategy, not security. By default, `uv` only searches the first index where a package is found to prevent dependency confusion attacks. This flag allows `uv` to search all configured indexes to find the best compatible versions. Since both PyTorch and PyPI are trusted package sources, using this strategy is safe and appropriate for vLLM installation.
|
||||
|
||||
!!! note
|
||||
On macOS the `VLLM_TARGET_DEVICE` is automatically set to `cpu`, which is currently the only supported device.
|
||||
|
||||
|
||||
@@ -48,10 +48,10 @@ Execute the following commands to build and install vLLM from source.
|
||||
|
||||
```bash
|
||||
uv pip install -v \
|
||||
--extra-index-url https://download.pytorch.org/whl/cpu \
|
||||
--torch-backend auto \
|
||||
-r requirements/build/cpu.txt \
|
||||
-r requirements/cpu.txt \
|
||||
--torch-backend cpu \
|
||||
--index-strategy unsafe-best-match && \
|
||||
VLLM_TARGET_DEVICE=cpu python setup.py bdist_wheel && \
|
||||
uv pip install dist/*.whl
|
||||
```
|
||||
|
||||
@@ -2,8 +2,7 @@
|
||||
|
||||
!!! note
|
||||
We currently support pooling models primarily for convenience. This is not guaranteed to provide any performance
|
||||
improvements over using Hugging Face Transformers or Sentence Transformers directly.
|
||||
|
||||
improvements over using Hugging Face Transformers or Sentence Transformers directly.
|
||||
We plan to optimize pooling models in vLLM. Please comment on <https://github.com/vllm-project/vllm/issues/21796> if you have any suggestions!
|
||||
|
||||
## What are pooling models?
|
||||
@@ -63,7 +62,7 @@ please refer to [IO Processor Plugins](../../design/io_processor_plugins.md).
|
||||
|
||||
!!! note
|
||||
Within classification tasks, there is a specialized subcategory: Cross-encoder (aka reranker) models. These models
|
||||
are a subset of classification models that accept two prompts as input and output num_labels equal to 1.
|
||||
are a subset of classification models that accept two prompts as input and output num_labels equal to 1.
|
||||
|
||||
### Pooling Types
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@ These models are what we list in [supported text models](#list-of-text-only-lang
|
||||
|
||||
### Transformers
|
||||
|
||||
vLLM also supports model implementations that are available in Transformers. You should expect the performance of a Transformers model implementation used in vLLM to be within <5% of the performance of a dedicated vLLM model implementation. We call this feature the "Transformers modeling backend".
|
||||
vLLM also supports model implementations that are available in Transformers. We call this feature the "Transformers modeling backend". The performance of models loaded with the Transformers modeling backend should be identical to a dedicated vLLM model implementation.
|
||||
|
||||
Currently, the Transformers modeling backend works for the following:
|
||||
|
||||
@@ -140,7 +140,7 @@ Here is what happens in the background when this model is loaded:
|
||||
|
||||
That's it!
|
||||
|
||||
For your model to be compatible with vLLM's tensor parallel and/or pipeline parallel features, you must add `base_model_tp_plan` and/or `base_model_pp_plan` to your model's config class:
|
||||
For your model to be compatible with vLLM's tensor parallel and/or pipeline parallel features, you may need to add `base_model_tp_plan` and/or `base_model_pp_plan` to your model's config class:
|
||||
|
||||
<details class="code">
|
||||
<summary>configuration_my_model.py</summary>
|
||||
@@ -168,9 +168,11 @@ class MyConfig(PretrainedConfig):
|
||||
</details>
|
||||
|
||||
- `base_model_tp_plan` is a `dict` that maps fully qualified layer name patterns to tensor parallel styles (currently only `"colwise"` and `"rowwise"` are supported).
|
||||
- vLLM infers the tensor parallel style of standard attention (`q`/`k`/`v`/`o_proj`) and gated-MLP/experts (`gate`/`up`/`down_proj`) projections if it can fuse them, so these may not need to be listed. `base_model_tp_plan` is only _required_ for layers that do not follow these patterns; any linear that is neither fused nor named in the plan is replicated.
|
||||
- `base_model_pp_plan` is a `dict` that maps direct child layer names to `tuple`s of `list`s of `str`s:
|
||||
- You only need to do this for layers which are not present on all pipeline stages
|
||||
- vLLM assumes that there will be only one `nn.ModuleList`, which is distributed across the pipeline stages
|
||||
- When no `base_model_pp_plan` is provided, the Transformers modelling backend infers the split from the text model's sole `nn.ModuleList`, keeping the parameter-bearing modules around it (input embeddings, final norm) on the first/last stage (depending on declaration order) and parameter-free modules (e.g. rotary embeddings) on every stage
|
||||
- The `list` in the first element of the `tuple` contains the names of the input arguments
|
||||
- The `list` in the last element of the `tuple` contains the names of the variables the layer outputs to in your modeling code
|
||||
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
# ruff: noqa: E501
|
||||
|
||||
"""
|
||||
Example online usage of the Jina Reranker v3 score and rerank APIs with a task
|
||||
instruction.
|
||||
|
||||
Run `vllm serve jinaai/jina-reranker-v3 --runner pooling` to start up the
|
||||
server in vLLM.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
def post_http_request(prompt: dict, api_url: str) -> requests.Response:
|
||||
headers = {"User-Agent": "Test Client"}
|
||||
response = requests.post(api_url, headers=headers, json=prompt)
|
||||
return response
|
||||
|
||||
|
||||
def print_response(name: str, prompt: dict, response: requests.Response) -> None:
|
||||
print(f"\n{name} request:")
|
||||
print(json.dumps(prompt, indent=2))
|
||||
print(f"\n{name} response:")
|
||||
print(json.dumps(response.json(), indent=2))
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--host", type=str, default="localhost")
|
||||
parser.add_argument("--port", type=int, default=8000)
|
||||
parser.add_argument("--model", type=str, default="jinaai/jina-reranker-v3")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main(args):
|
||||
score_url = f"http://{args.host}:{args.port}/score"
|
||||
rerank_url = f"http://{args.host}:{args.port}/rerank"
|
||||
model_name = args.model
|
||||
|
||||
query = "Which passage is about sports?"
|
||||
documents = [
|
||||
"Basketball is played by two teams on a court.",
|
||||
"Green tea contains antioxidants and may support metabolism.",
|
||||
]
|
||||
instruction = "Rank passages about sports higher than passages about nutrition."
|
||||
|
||||
score_prompt = {
|
||||
"model": model_name,
|
||||
"queries": query,
|
||||
"documents": documents,
|
||||
"instruction": instruction,
|
||||
}
|
||||
score_response = post_http_request(prompt=score_prompt, api_url=score_url)
|
||||
print_response("Score", score_prompt, score_response)
|
||||
|
||||
rerank_prompt = {
|
||||
"model": model_name,
|
||||
"query": query,
|
||||
"documents": documents,
|
||||
"instruction": instruction,
|
||||
}
|
||||
rerank_response = post_http_request(prompt=rerank_prompt, api_url=rerank_url)
|
||||
print_response("Rerank", rerank_prompt, rerank_response)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
args = parse_args()
|
||||
main(args)
|
||||
@@ -1,4 +1,3 @@
|
||||
--extra-index-url https://download.pytorch.org/whl/cpu
|
||||
cmake>=3.26.1
|
||||
ninja
|
||||
packaging>=24.2
|
||||
|
||||
@@ -1,4 +1,3 @@
|
||||
--extra-index-url https://download.pytorch.org/whl/cpu
|
||||
# Common dependencies
|
||||
-r common.txt
|
||||
|
||||
|
||||
@@ -25,7 +25,7 @@ nvidia-cutlass-dsl[cu13]==4.5.2
|
||||
quack-kernels>=0.3.3
|
||||
|
||||
# Tokenspeed_MLA for faster mla with spec decode
|
||||
tokenspeed-mla==0.1.2
|
||||
tokenspeed-mla==0.1.2; platform_system == "Linux"
|
||||
|
||||
# Humming kernels for quantization gemm
|
||||
humming-kernels[cu13]==0.1.6
|
||||
|
||||
@@ -112,9 +112,10 @@ charset-normalizer==3.4.0
|
||||
# via requests
|
||||
chz==0.3.0
|
||||
# via gpt-oss
|
||||
click==8.1.7
|
||||
click==8.4.2
|
||||
# via
|
||||
# black
|
||||
# huggingface-hub
|
||||
# jiwer
|
||||
# nltk
|
||||
# ray
|
||||
@@ -309,7 +310,7 @@ h2==4.3.0
|
||||
# via httpx
|
||||
harfile==0.5.0
|
||||
# via schemathesis
|
||||
hf-xet==1.4.3
|
||||
hf-xet==1.5.1
|
||||
# via huggingface-hub
|
||||
hiredis==3.0.0
|
||||
# via tensorizer
|
||||
@@ -335,7 +336,7 @@ httpx==0.27.2
|
||||
# schemathesis
|
||||
httpx-sse==0.4.3
|
||||
# via mcp
|
||||
huggingface-hub==1.10.2
|
||||
huggingface-hub==1.22.0
|
||||
# via
|
||||
# accelerate
|
||||
# datasets
|
||||
@@ -1182,7 +1183,6 @@ typer==0.26.8
|
||||
# fastapi-cli
|
||||
# fastapi-cloud-cli
|
||||
# fastsafetensors
|
||||
# huggingface-hub
|
||||
# perceptron
|
||||
# transformers
|
||||
typing-extensions==4.15.0
|
||||
|
||||
@@ -117,9 +117,10 @@ charset-normalizer==3.4.0
|
||||
# via requests
|
||||
chz==0.3.0
|
||||
# via gpt-oss
|
||||
click==8.1.7
|
||||
click==8.4.2
|
||||
# via
|
||||
# black
|
||||
# huggingface-hub
|
||||
# jiwer
|
||||
# nltk
|
||||
# ray
|
||||
@@ -330,7 +331,7 @@ h2==4.3.0
|
||||
# via httpx
|
||||
harfile==0.5.0
|
||||
# via schemathesis
|
||||
hf-xet==1.4.3
|
||||
hf-xet==1.5.1
|
||||
# via huggingface-hub
|
||||
hiredis==3.0.0
|
||||
# via tensorizer
|
||||
@@ -356,7 +357,7 @@ httpx==0.27.2
|
||||
# schemathesis
|
||||
httpx-sse==0.4.3
|
||||
# via mcp
|
||||
huggingface-hub==1.10.2
|
||||
huggingface-hub==1.22.0
|
||||
# via
|
||||
# accelerate
|
||||
# datasets
|
||||
@@ -1285,7 +1286,6 @@ typer==0.26.8
|
||||
# fastapi-cli
|
||||
# fastapi-cloud-cli
|
||||
# fastsafetensors
|
||||
# huggingface-hub
|
||||
# perceptron
|
||||
# transformers
|
||||
typing-extensions==4.15.0
|
||||
|
||||
@@ -116,9 +116,10 @@ choreographer==1.2.1
|
||||
# via kaleido
|
||||
chz==0.4.0
|
||||
# via gpt-oss
|
||||
click==8.3.1
|
||||
click==8.4.2
|
||||
# via
|
||||
# black
|
||||
# huggingface-hub
|
||||
# jiwer
|
||||
# nltk
|
||||
# ray
|
||||
@@ -323,7 +324,7 @@ h2==4.3.0
|
||||
# via httpx
|
||||
harfile==0.5.0
|
||||
# via schemathesis
|
||||
hf-xet==1.4.3
|
||||
hf-xet==1.5.1
|
||||
# via huggingface-hub
|
||||
hiredis==3.3.1
|
||||
# via tensorizer
|
||||
@@ -349,7 +350,7 @@ httpx==0.27.2
|
||||
# schemathesis
|
||||
httpx-sse==0.4.3
|
||||
# via mcp
|
||||
huggingface-hub==1.10.2
|
||||
huggingface-hub==1.22.0
|
||||
# via
|
||||
# accelerate
|
||||
# datasets
|
||||
@@ -1244,7 +1245,6 @@ typer==0.24.1
|
||||
# fastapi-cli
|
||||
# fastapi-cloud-cli
|
||||
# fastsafetensors
|
||||
# huggingface-hub
|
||||
# perceptron
|
||||
# transformers
|
||||
typing-extensions==4.15.0
|
||||
|
||||
@@ -83,8 +83,9 @@ charset-normalizer==3.4.6
|
||||
# via requests
|
||||
chz==0.4.0
|
||||
# via gpt-oss
|
||||
click==8.3.1
|
||||
click==8.4.2
|
||||
# via
|
||||
# huggingface-hub
|
||||
# jiwer
|
||||
# nltk
|
||||
# rich-toolkit
|
||||
@@ -206,7 +207,7 @@ h11==0.16.0
|
||||
# uvicorn
|
||||
harfile==0.4.0
|
||||
# via schemathesis
|
||||
hf-xet==1.4.3
|
||||
hf-xet==1.5.1
|
||||
# via huggingface-hub
|
||||
html2text==2025.4.15
|
||||
# via gpt-oss
|
||||
@@ -227,7 +228,7 @@ httpx==0.28.1
|
||||
# schemathesis
|
||||
httpx-sse==0.4.3
|
||||
# via mcp
|
||||
huggingface-hub==1.10.2
|
||||
huggingface-hub==1.22.0
|
||||
# via
|
||||
# accelerate
|
||||
# datasets
|
||||
@@ -959,7 +960,6 @@ typer==0.24.1
|
||||
# via
|
||||
# fastapi-cli
|
||||
# fastapi-cloud-cli
|
||||
# huggingface-hub
|
||||
# transformers
|
||||
typing-extensions==4.15.0
|
||||
# via
|
||||
|
||||
@@ -16,5 +16,5 @@ torch==2.12.0
|
||||
torchaudio
|
||||
torchvision
|
||||
|
||||
auto_round_lib>=0.13.3
|
||||
auto_round_lib>=0.14.0
|
||||
vllm_xpu_kernels @ https://github.com/vllm-project/vllm-xpu-kernels/releases/download/v0.1.10.1/vllm_xpu_kernels-0.1.10.1-cp38-abi3-manylinux_2_28_x86_64.whl
|
||||
|
||||
Generated
+28
-13
@@ -489,9 +489,9 @@ checksum = "8f1fe948ff07f4bd06c30984e69f5b4899c516a3ef74f34df92a2df2ab535495"
|
||||
|
||||
[[package]]
|
||||
name = "bytes"
|
||||
version = "1.11.1"
|
||||
version = "1.12.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1e748733b7cbc798e1434b6ac524f0c1ff2ab456fe201501e6497c8417a4fc33"
|
||||
checksum = "8ae3f5d315924270530207e2a68396c3cc547f6dca3fbdca317cfb1a51edb593"
|
||||
dependencies = [
|
||||
"serde",
|
||||
]
|
||||
@@ -2112,6 +2112,16 @@ version = "0.2.183"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b5b646652bf6661599e1da8901b3b9522896f01e736bad5f723fe7a3a27f899d"
|
||||
|
||||
[[package]]
|
||||
name = "libloading"
|
||||
version = "0.8.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d7c4b02199fee7c5d21a5ae7d8cfa79a6ef5bb2fc834d6e9058e89c825efdc55"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"windows-link",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "libm"
|
||||
version = "0.2.16"
|
||||
@@ -2156,21 +2166,27 @@ checksum = "11d3d7f243d5c5a8b9bb5d6dd2b1602c0cb0b9db1621bafc7ed66e35ff9fe092"
|
||||
|
||||
[[package]]
|
||||
name = "llm-multimodal"
|
||||
version = "1.5.0"
|
||||
source = "git+https://github.com/vllm-project/llm-multimodal?rev=046b669bd1c4faa2a7e05344d8cbf7b2befb37d5#046b669bd1c4faa2a7e05344d8cbf7b2befb37d5"
|
||||
version = "1.7.1"
|
||||
source = "git+https://github.com/smg-project/llm-multimodal?rev=7d74582aeaf0e4086a44964382655d22f1af0686#7d74582aeaf0e4086a44964382655d22f1af0686"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"base64 0.22.1",
|
||||
"blake3",
|
||||
"bytes",
|
||||
"fast_image_resize",
|
||||
"hf-hub",
|
||||
"image",
|
||||
"libloading",
|
||||
"ndarray 0.17.2",
|
||||
"once_cell",
|
||||
"reqwest",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"serde_with",
|
||||
"tempfile",
|
||||
"thiserror 2.0.18",
|
||||
"tokio",
|
||||
"tracing",
|
||||
"url",
|
||||
]
|
||||
|
||||
@@ -2365,9 +2381,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "mio"
|
||||
version = "1.1.1"
|
||||
version = "1.2.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a69bcab0ad47271a0234d9422b131806bf3968021e5dc9328caf2d4cd58557fc"
|
||||
checksum = "02bd0af71c67b473010cbbc60715ee815645a4dc942899111f494b4b737d6fda"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"wasi",
|
||||
@@ -2532,9 +2548,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "once_cell"
|
||||
version = "1.21.3"
|
||||
version = "1.21.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "42f5e15c9953c5e4ccceeb2e7382a716482c34515315f7b03532b8b4e8393d2d"
|
||||
checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
||||
|
||||
[[package]]
|
||||
name = "once_cell_polyfill"
|
||||
@@ -4452,9 +4468,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "tokio"
|
||||
version = "1.50.0"
|
||||
version = "1.52.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "27ad5e34374e03cfffefc301becb44e9dc3c17584f414349ebe29ed26661822d"
|
||||
checksum = "8fc7f01b389ac15039e4dc9531aa973a135d7a4135281b12d7c1bc79fd57fffe"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"libc",
|
||||
@@ -4469,9 +4485,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "tokio-macros"
|
||||
version = "2.6.1"
|
||||
version = "2.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5c55a2eff8b69ce66c84f85e1da1c233edc36ceb85a2058d11b0d6a3c7e7569c"
|
||||
checksum = "385a6cb71ab9ab790c5fe8d67f1645e6c450a7ce006a33de03daa956cf70a496"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -5268,7 +5284,6 @@ dependencies = [
|
||||
"clap",
|
||||
"educe",
|
||||
"expect-test",
|
||||
"form_urlencoded",
|
||||
"futures",
|
||||
"http-body",
|
||||
"hyper",
|
||||
|
||||
+9
-3
@@ -31,7 +31,7 @@ axum = "0.8.8"
|
||||
base64 = "0.22.1"
|
||||
bytemuck = { version = "1.25.0", features = ["extern_crate_alloc"] }
|
||||
byteorder = "1.5.0"
|
||||
bytes = "1.11.1"
|
||||
bytes = "1.12.0"
|
||||
clap = { version = "4.5.38", features = ["derive", "env"] }
|
||||
criterion = "0.5.1"
|
||||
easy-ext = "1.0.3"
|
||||
@@ -39,7 +39,6 @@ educe = "0.6.0"
|
||||
enum-as-inner = "0.7.0"
|
||||
expect-test = "1.5.1"
|
||||
fastokens = { version = "0.2.1", default-features = false }
|
||||
form_urlencoded = "1.2.2"
|
||||
futures = "0.3.31"
|
||||
half = { version = "2.7.1", features = ["bytemuck"] }
|
||||
hex = "0.4.3"
|
||||
@@ -54,7 +53,7 @@ hyper-util = { version = "0.1.20", features = [
|
||||
indexmap = "2.13.0"
|
||||
itertools = "0.14.0"
|
||||
libc = "0.2.177"
|
||||
llm-multimodal = { git = "https://github.com/vllm-project/llm-multimodal", rev = "046b669bd1c4faa2a7e05344d8cbf7b2befb37d5" }
|
||||
llm-multimodal = { git = "https://github.com/smg-project/llm-multimodal", rev = "7d74582aeaf0e4086a44964382655d22f1af0686" }
|
||||
mimalloc = "0.1.52"
|
||||
minijinja = { version = "2.0", features = ["unstable_machinery", "json", "builtins", "loader", "loop_controls", "preserve_order"] }
|
||||
minijinja-contrib = { version = "2.0", features = ["pycompat"] }
|
||||
@@ -145,6 +144,13 @@ too_many_arguments = "allow"
|
||||
[profile.dev]
|
||||
panic = "abort"
|
||||
|
||||
# Speed up cold tokenizer construction in tests.
|
||||
[profile.dev.package]
|
||||
fastokens = { opt-level = 3 }
|
||||
regex-automata = { opt-level = 3 }
|
||||
serde_json = { opt-level = 3 }
|
||||
tokenizers = { opt-level = 3 }
|
||||
|
||||
[profile.release]
|
||||
lto = "thin"
|
||||
panic = "abort"
|
||||
|
||||
@@ -303,7 +303,7 @@ mod tests {
|
||||
.unwrap()
|
||||
.join("preprocessor_config.json");
|
||||
write_json(&preprocessor_config_path, r#"{"size":[672,672]}"#);
|
||||
files.preprocessor_config_path = Some(preprocessor_config_path);
|
||||
files.preprocessor_config_path = Some(preprocessor_config_path.clone());
|
||||
|
||||
let backend = HfChatBackend::from_resolved_model_files(
|
||||
files.clone(),
|
||||
@@ -321,6 +321,9 @@ mod tests {
|
||||
|
||||
assert!(backend.multimodal_model_info().is_none());
|
||||
|
||||
let invalid_preprocessor_config = r#"{"size":[672,672]"#;
|
||||
write_json(&preprocessor_config_path, invalid_preprocessor_config);
|
||||
|
||||
let error = HfChatBackend::from_resolved_model_files(
|
||||
files,
|
||||
"test-model".to_string(),
|
||||
|
||||
@@ -16,10 +16,11 @@ use std::sync::{Arc, LazyLock};
|
||||
|
||||
use itertools::izip;
|
||||
use llm_multimodal::{
|
||||
AsyncMultiModalTracker, FieldLayout, ImagePreProcessor, ImageProcessorRegistry, MediaConnector,
|
||||
MediaConnectorConfig, MediaContentPart, Modality, ModelMetadata, ModelProcessorSpec,
|
||||
ModelRegistry, PreProcessorConfig, PreprocessedImages, PromptReplacement, TokenResolver,
|
||||
TrackedMedia,
|
||||
AsyncMultiModalTracker, FieldLayout, MediaConnector, MediaConnectorConfig, MediaContentPart,
|
||||
Modality, ModelMetadata, ModelProcessorSpec, ModelRegistry, PreProcessorConfig,
|
||||
PreprocessedEncoderInputs as PreprocessedImages, PromptReplacement, Tokenizer as TokenResolver,
|
||||
TrackedMedia, VisionPreProcessor as ImagePreProcessor,
|
||||
VisionProcessorRegistry as ImageProcessorRegistry,
|
||||
};
|
||||
use tracing::warn;
|
||||
use vllm_engine_core_client::protocol::dtype::ModelDtype;
|
||||
@@ -555,6 +556,10 @@ impl TokenResolver for TokenizerResolver {
|
||||
fn id_to_token(&self, id: u32) -> Option<String> {
|
||||
self.0.id_to_token(id)
|
||||
}
|
||||
|
||||
fn encode_text(&self, text: &str) -> Option<Vec<u32>> {
|
||||
self.0.encode(text, false).ok()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use std::collections::HashMap;
|
||||
|
||||
use half::{bf16, f16};
|
||||
use llm_multimodal::{ModelSpecificValue, PreprocessedImages};
|
||||
use llm_multimodal::{ModelSpecificValue, PreprocessedEncoderInputs as PreprocessedImages};
|
||||
use vllm_engine_core_client::protocol::dtype::ModelDtype;
|
||||
use vllm_engine_core_client::protocol::multimodal::MmKwargValue as ProtocolKwargValue;
|
||||
use vllm_engine_core_client::protocol::tensor::{ShapeExt as _, WireTensor};
|
||||
@@ -31,14 +31,14 @@ pub(super) fn collect_tensors(
|
||||
float_dtype: ModelDtype,
|
||||
) -> Result<HashMap<String, KwargValue>> {
|
||||
let PreprocessedImages {
|
||||
pixel_values,
|
||||
encoder_input,
|
||||
model_specific,
|
||||
..
|
||||
} = preprocessed;
|
||||
|
||||
let pixel_values = {
|
||||
let shape = pixel_values.shape().to_vec();
|
||||
let data = pixel_values.into_iter().collect();
|
||||
let shape = encoder_input.shape().to_vec();
|
||||
let data = encoder_input.into_iter().collect();
|
||||
KwargValue::from_f32_tensor(data, shape, float_dtype)?
|
||||
};
|
||||
|
||||
|
||||
@@ -214,14 +214,17 @@ macro_rules! roundtrip_tests {
|
||||
($($case:ident => [$($(#[$fixture_attr:meta])* $fixture:ident),* $(,)?]),+ $(,)?) => {
|
||||
paste::paste! {
|
||||
$(
|
||||
$(
|
||||
#[tokio::test]
|
||||
$(#[$fixture_attr])*
|
||||
#[file_serial([<hf_ $case>])]
|
||||
async fn [<roundtrip_ $case _ $fixture>]() -> Result<()> {
|
||||
[<run_roundtrip_ $fixture>](RoundtripCase::$case()).await
|
||||
}
|
||||
)*
|
||||
#[tokio::test]
|
||||
#[file_serial([<hf_ $case>])]
|
||||
async fn [<roundtrip_ $case>]() -> Result<()> {
|
||||
let case = RoundtripCase::$case();
|
||||
let backends = load_roundtrip_backends(&case).await?;
|
||||
$(
|
||||
$(#[$fixture_attr])*
|
||||
[<run_roundtrip_ $fixture>](&case, &backends).await?;
|
||||
)*
|
||||
Ok(())
|
||||
}
|
||||
)+
|
||||
}
|
||||
};
|
||||
@@ -241,18 +244,21 @@ roundtrip_tests! {
|
||||
}
|
||||
|
||||
/// Run the fixed reasoning+content fixture for one model/parser case.
|
||||
async fn run_roundtrip_reasoning_and_content(case: RoundtripCase) -> Result<()> {
|
||||
async fn run_roundtrip_reasoning_and_content(
|
||||
case: &RoundtripCase,
|
||||
backends: &vllm_chat::LoadedModelBackends,
|
||||
) -> Result<()> {
|
||||
for thinking in case.thinking_behavior.fixtures() {
|
||||
run_roundtrip_reasoning_and_content_inner(case.clone(), thinking).await?;
|
||||
run_roundtrip_reasoning_and_content_inner(case, backends, thinking).await?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn run_roundtrip_reasoning_and_content_inner(
|
||||
case: RoundtripCase,
|
||||
case: &RoundtripCase,
|
||||
backends: &vllm_chat::LoadedModelBackends,
|
||||
thinking: Option<bool>,
|
||||
) -> Result<()> {
|
||||
let backends = load_roundtrip_backends(&case).await?;
|
||||
let request = roundtrip_request(
|
||||
"roundtrip-reasoning-content",
|
||||
vec![ChatMessage::text(ChatRole::User, "What is 2 + 2?")],
|
||||
@@ -275,7 +281,7 @@ async fn run_roundtrip_reasoning_and_content_inner(
|
||||
});
|
||||
AssistantMessage { content }
|
||||
};
|
||||
let result = run_roundtrip(&case, &backends, &request, assistant).await?;
|
||||
let result = run_roundtrip(case, backends, &request, assistant).await?;
|
||||
|
||||
assert_eq!(
|
||||
result.parsed_message.reasoning().as_deref().map(str::trim),
|
||||
@@ -293,8 +299,10 @@ async fn run_roundtrip_reasoning_and_content_inner(
|
||||
}
|
||||
|
||||
/// Run the fixed reasoning+multiple-tools fixture for one model/parser case.
|
||||
async fn run_roundtrip_tool_call_mix(case: RoundtripCase) -> Result<()> {
|
||||
let backends = load_roundtrip_backends(&case).await?;
|
||||
async fn run_roundtrip_tool_call_mix(
|
||||
case: &RoundtripCase,
|
||||
backends: &vllm_chat::LoadedModelBackends,
|
||||
) -> Result<()> {
|
||||
let request = roundtrip_request(
|
||||
"roundtrip-reasoning-tools",
|
||||
vec![ChatMessage::text(
|
||||
@@ -308,8 +316,8 @@ async fn run_roundtrip_tool_call_mix(case: RoundtripCase) -> Result<()> {
|
||||
let expected_text = "I will call the tools.";
|
||||
|
||||
let result = run_roundtrip(
|
||||
&case,
|
||||
&backends,
|
||||
case,
|
||||
backends,
|
||||
&request,
|
||||
AssistantMessage {
|
||||
content: vec![
|
||||
@@ -353,12 +361,12 @@ async fn run_roundtrip_tool_call_mix(case: RoundtripCase) -> Result<()> {
|
||||
assert_eq!(tool_calls[0].name, "get_weather");
|
||||
assert_eq!(
|
||||
tool_calls[0].arguments,
|
||||
expected_arguments(&case, r#"{"location": "Shanghai"}"#)?,
|
||||
expected_arguments(case, r#"{"location": "Shanghai"}"#)?,
|
||||
);
|
||||
assert_eq!(tool_calls[1].name, "add");
|
||||
assert_eq!(
|
||||
tool_calls[1].arguments,
|
||||
expected_arguments(&case, r#"{"y": 1.0, "x": 2, "items": ["left", "right"]}"#)?,
|
||||
expected_arguments(case, r#"{"y": 1.0, "x": 2, "items": ["left", "right"]}"#)?,
|
||||
);
|
||||
|
||||
assert_eq!(
|
||||
|
||||
@@ -456,17 +456,13 @@ pub struct ServerUnsupportedArgs {
|
||||
|
||||
/// Enable the `/tokenizer_info` endpoint. May expose chat
|
||||
/// templates and other tokenizer configuration.
|
||||
///
|
||||
/// Accepted as a no-op: the Rust frontend serves `/tokenize` and
|
||||
/// `/detokenize`, but does not implement `/tokenizer_info` yet.
|
||||
#[arg(
|
||||
long,
|
||||
visible_alias = "no-enable-tokenizer-info-endpoint",
|
||||
default_missing_value = "true",
|
||||
num_args = 0..=1,
|
||||
hide = true
|
||||
num_args = 0..=1
|
||||
)]
|
||||
pub enable_tokenizer_info_endpoint: Option<Noop>,
|
||||
pub enable_tokenizer_info_endpoint: Option<Unsupported>,
|
||||
|
||||
/// If set to True, log model outputs (generations).
|
||||
/// Requires `--enable-log-requests`. As with `--enable-log-requests`,
|
||||
|
||||
@@ -373,12 +373,6 @@ impl EngineCoreClient {
|
||||
}
|
||||
|
||||
/// Return the engine-side indices connected to this client.
|
||||
///
|
||||
/// # Panics
|
||||
///
|
||||
/// Panics if any connected engine uses an opaque identity that does not
|
||||
/// encode an index. Use [`Self::known_engine_indices`] for a lossy,
|
||||
/// non-panicking variant.
|
||||
pub fn engine_indices(&self) -> Vec<u32> {
|
||||
self.engines
|
||||
.iter()
|
||||
@@ -386,16 +380,6 @@ impl EngineCoreClient {
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Return the engine-side indices connected to this client, skipping
|
||||
/// engines with opaque identities that do not encode an index (e.g. mock
|
||||
/// engines in tests).
|
||||
pub fn known_engine_indices(&self) -> Vec<u32> {
|
||||
self.engines
|
||||
.iter()
|
||||
.filter_map(|engine| engine.engine_id.engine_index())
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Return the engine identities of all engines connected to this client.
|
||||
pub fn engine_identities(&self) -> Vec<&[u8]> {
|
||||
self.engines.iter().map(|engine| &*engine.engine_id).collect()
|
||||
@@ -528,6 +512,7 @@ impl EngineCoreClient {
|
||||
|
||||
Ok(EngineCoreOutputStream::new(
|
||||
request_id,
|
||||
engine_id.engine_index().unwrap_or(0),
|
||||
self.abort_tx.clone(),
|
||||
rx,
|
||||
))
|
||||
|
||||
@@ -15,7 +15,7 @@ use crate::client::state::{OutputReceiver, RequestRegistry, UtilityReceiver, Uti
|
||||
use crate::client::stream::EngineCoreStreamOutput;
|
||||
use crate::client::{AbortCause, AbortRequest};
|
||||
use crate::error::{client_closed, dispatcher_closed, unexpected_dispatcher_output};
|
||||
use crate::metrics::{LoraInfoExporter, record_scheduler_stats};
|
||||
use crate::metrics::{LoraInfoExporter, SchedulerStatsRecorder};
|
||||
use crate::protocol::encode_msgpack;
|
||||
use crate::protocol::output::{EngineCoreOutput, EngineCoreOutputs};
|
||||
use crate::protocol::request::EngineCoreRequestType;
|
||||
@@ -29,6 +29,7 @@ pub(crate) struct ClientInner {
|
||||
/// The runtime handle used for sending messages to the engine.
|
||||
handle: Handle,
|
||||
model_name: String,
|
||||
scheduler_stats_recorder: SchedulerStatsRecorder,
|
||||
request_reg: Mutex<RequestRegistry>,
|
||||
utility_reg: Mutex<UtilityRegistry>,
|
||||
health_error: ArcSwapOption<Error>,
|
||||
@@ -43,10 +44,13 @@ impl ClientInner {
|
||||
model_name: String,
|
||||
engines: &[ConnectedEngine],
|
||||
) -> Self {
|
||||
let scheduler_stats_recorder =
|
||||
SchedulerStatsRecorder::new(&METRICS.scheduler, &model_name, engines);
|
||||
Self {
|
||||
input_send,
|
||||
handle,
|
||||
model_name,
|
||||
scheduler_stats_recorder,
|
||||
request_reg: Mutex::new(RequestRegistry::new(engines)),
|
||||
utility_reg: Mutex::new(UtilityRegistry::default()),
|
||||
health_error: ArcSwapOption::empty(),
|
||||
@@ -389,12 +393,7 @@ pub(crate) async fn run_output_dispatcher_loop(
|
||||
"dropping scheduler stats for unknown engine"
|
||||
);
|
||||
}
|
||||
record_scheduler_stats(
|
||||
&METRICS.scheduler,
|
||||
inner.model_name(),
|
||||
batch.engine_index,
|
||||
scheduler_stats,
|
||||
);
|
||||
inner.scheduler_stats_recorder.record(batch.engine_index, scheduler_stats);
|
||||
}
|
||||
|
||||
// The engine's scheduler stats never carry adapter names;
|
||||
|
||||
@@ -45,6 +45,7 @@ impl Deref for EngineCoreStreamOutput {
|
||||
/// `finish_reason` is non-`None`.
|
||||
pub struct EngineCoreOutputStream {
|
||||
request_id: String,
|
||||
engine_index: u32,
|
||||
abort_tx: mpsc::UnboundedSender<AbortRequest>,
|
||||
state: State,
|
||||
rx: OutputReceiver,
|
||||
@@ -53,11 +54,13 @@ pub struct EngineCoreOutputStream {
|
||||
impl EngineCoreOutputStream {
|
||||
pub(crate) fn new(
|
||||
request_id: String,
|
||||
engine_index: u32,
|
||||
abort_tx: mpsc::UnboundedSender<AbortRequest>,
|
||||
rx: OutputReceiver,
|
||||
) -> Self {
|
||||
Self {
|
||||
request_id,
|
||||
engine_index,
|
||||
abort_tx,
|
||||
state: State::Running,
|
||||
rx,
|
||||
@@ -68,6 +71,11 @@ impl EngineCoreOutputStream {
|
||||
pub fn request_id(&self) -> &str {
|
||||
&self.request_id
|
||||
}
|
||||
|
||||
/// Return the index of the engine that owns this request.
|
||||
pub fn engine_index(&self) -> u32 {
|
||||
self.engine_index
|
||||
}
|
||||
}
|
||||
|
||||
impl Stream for EngineCoreOutputStream {
|
||||
|
||||
@@ -1,90 +1,190 @@
|
||||
use std::collections::BTreeMap;
|
||||
use std::collections::BTreeSet;
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
|
||||
use vllm_metrics::{
|
||||
EngineLabels, EnginePositionLabels, LoraAdapterNames, LoraInfoLabels, SchedulerMetrics,
|
||||
EngineLabels, EnginePositionLabels, F64Gauge, Family, HistogramMetric, LoraAdapterNames,
|
||||
LoraInfoLabels, SchedulerLogStatsAccumulator, SchedulerMetrics, U64Counter, U64Gauge,
|
||||
WaitingReasonLabels,
|
||||
};
|
||||
|
||||
use crate::protocol::stats::SchedulerStats;
|
||||
use crate::transport::ConnectedEngine;
|
||||
|
||||
const WAITING_REASON_CAPACITY: &str = "capacity";
|
||||
const WAITING_REASON_DEFERRED: &str = "deferred";
|
||||
|
||||
/// Record the scheduler-stats-backed metrics for one engine at one point in
|
||||
/// time.
|
||||
pub(crate) fn record_scheduler_stats(
|
||||
metrics: &SchedulerMetrics,
|
||||
model_name: impl Into<String>,
|
||||
engine: u32,
|
||||
stats: &SchedulerStats,
|
||||
) {
|
||||
let model_name = model_name.into();
|
||||
let labels = EngineLabels {
|
||||
model_name: model_name.clone(),
|
||||
engine,
|
||||
};
|
||||
/// Cached scheduler-stats metric handles for all engines connected to one
|
||||
/// frontend client.
|
||||
pub(crate) struct SchedulerStatsRecorder {
|
||||
engines: BTreeMap<u32, SchedulerStatsHandles>,
|
||||
}
|
||||
|
||||
/// Per-engine cached metric handles used while recording `SchedulerStats`.
|
||||
struct SchedulerStatsHandles {
|
||||
// Base labels reused for dynamic child labels.
|
||||
labels: EngineLabels,
|
||||
|
||||
// Scheduler state gauges.
|
||||
metrics.scheduler_running.get_or_create(&labels).set(stats.num_running_reqs);
|
||||
metrics
|
||||
.scheduler_waiting
|
||||
.get_or_create(&labels)
|
||||
.set(stats.num_waiting_reqs + stats.num_skipped_waiting_reqs);
|
||||
metrics
|
||||
.scheduler_waiting_by_reason
|
||||
.get_or_create(&WaitingReasonLabels {
|
||||
model_name: model_name.clone(),
|
||||
engine,
|
||||
reason: WAITING_REASON_CAPACITY,
|
||||
})
|
||||
.set(stats.num_waiting_reqs);
|
||||
metrics
|
||||
.scheduler_waiting_by_reason
|
||||
.get_or_create(&WaitingReasonLabels {
|
||||
model_name: model_name.clone(),
|
||||
engine,
|
||||
reason: WAITING_REASON_DEFERRED,
|
||||
})
|
||||
.set(stats.num_skipped_waiting_reqs);
|
||||
metrics.kv_cache_usage.get_or_create(&labels).set(stats.kv_cache_usage);
|
||||
scheduler_running: U64Gauge,
|
||||
scheduler_waiting: U64Gauge,
|
||||
scheduler_waiting_capacity: U64Gauge,
|
||||
scheduler_waiting_deferred: U64Gauge,
|
||||
kv_cache_usage: F64Gauge,
|
||||
|
||||
// Prefix-cache counters, including the connector-backed external cache path.
|
||||
metrics
|
||||
.prefix_cache_queries
|
||||
.get_or_create(&labels)
|
||||
.inc_by(stats.prefix_cache_stats.base.queries);
|
||||
metrics
|
||||
.prefix_cache_hits
|
||||
.get_or_create(&labels)
|
||||
.inc_by(stats.prefix_cache_stats.base.hits);
|
||||
prefix_cache_queries: U64Counter,
|
||||
prefix_cache_hits: U64Counter,
|
||||
external_prefix_cache_queries: U64Counter,
|
||||
external_prefix_cache_hits: U64Counter,
|
||||
|
||||
// Speculative decoding counters.
|
||||
spec_decode_num_drafts: U64Counter,
|
||||
spec_decode_num_draft_tokens: U64Counter,
|
||||
spec_decode_num_accepted_tokens: U64Counter,
|
||||
spec_decode_num_accepted_tokens_per_pos: Family<EnginePositionLabels, U64Counter>,
|
||||
|
||||
// Per-engine performance / MFU counters.
|
||||
estimated_flops_per_gpu: U64Counter,
|
||||
estimated_read_bytes_per_gpu: U64Counter,
|
||||
estimated_write_bytes_per_gpu: U64Counter,
|
||||
|
||||
// Sampled KV-cache residency histograms.
|
||||
kv_block_lifetime_seconds: HistogramMetric,
|
||||
kv_block_idle_before_evict_seconds: HistogramMetric,
|
||||
kv_block_reuse_gap_seconds: HistogramMetric,
|
||||
|
||||
// Non-Prometheus interval accumulator for periodic text-log helpers.
|
||||
log_stats: SchedulerLogStatsAccumulator,
|
||||
}
|
||||
|
||||
impl SchedulerStatsRecorder {
|
||||
/// Resolve the fixed-label metric handles for the connected engines.
|
||||
pub(crate) fn new(
|
||||
metrics: &SchedulerMetrics,
|
||||
model_name: &str,
|
||||
engines: &[ConnectedEngine],
|
||||
) -> Self {
|
||||
let engines = engines
|
||||
.iter()
|
||||
.filter_map(|engine| {
|
||||
let engine = engine.engine_id.engine_index()?;
|
||||
Some((
|
||||
engine,
|
||||
resolve_scheduler_stats_handles(metrics, model_name, engine),
|
||||
))
|
||||
})
|
||||
.collect();
|
||||
|
||||
Self { engines }
|
||||
}
|
||||
|
||||
/// Record one scheduler-stats payload for the given engine index.
|
||||
pub(crate) fn record(&self, engine_index: u32, stats: &SchedulerStats) {
|
||||
if let Some(handles) = self.engines.get(&engine_index) {
|
||||
record_scheduler_stats_with_handles(handles, stats);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Resolve all fixed-label scheduler metrics for one engine.
|
||||
fn resolve_scheduler_stats_handles(
|
||||
metrics: &SchedulerMetrics,
|
||||
model_name: &str,
|
||||
engine: u32,
|
||||
) -> SchedulerStatsHandles {
|
||||
let labels = EngineLabels {
|
||||
model_name: model_name.to_string(),
|
||||
engine,
|
||||
};
|
||||
let capacity = WaitingReasonLabels {
|
||||
model_name: model_name.to_string(),
|
||||
engine,
|
||||
reason: WAITING_REASON_CAPACITY,
|
||||
};
|
||||
let deferred = WaitingReasonLabels {
|
||||
model_name: model_name.to_string(),
|
||||
engine,
|
||||
reason: WAITING_REASON_DEFERRED,
|
||||
};
|
||||
|
||||
SchedulerStatsHandles {
|
||||
scheduler_running: metrics.scheduler_running.get_or_create_owned(&labels),
|
||||
scheduler_waiting: metrics.scheduler_waiting.get_or_create_owned(&labels),
|
||||
scheduler_waiting_capacity: metrics
|
||||
.scheduler_waiting_by_reason
|
||||
.get_or_create_owned(&capacity),
|
||||
scheduler_waiting_deferred: metrics
|
||||
.scheduler_waiting_by_reason
|
||||
.get_or_create_owned(&deferred),
|
||||
kv_cache_usage: metrics.kv_cache_usage.get_or_create_owned(&labels),
|
||||
prefix_cache_queries: metrics.prefix_cache_queries.get_or_create_owned(&labels),
|
||||
prefix_cache_hits: metrics.prefix_cache_hits.get_or_create_owned(&labels),
|
||||
external_prefix_cache_queries: metrics
|
||||
.external_prefix_cache_queries
|
||||
.get_or_create_owned(&labels),
|
||||
external_prefix_cache_hits: metrics.external_prefix_cache_hits.get_or_create_owned(&labels),
|
||||
spec_decode_num_drafts: metrics.spec_decode_num_drafts.get_or_create_owned(&labels),
|
||||
spec_decode_num_draft_tokens: metrics
|
||||
.spec_decode_num_draft_tokens
|
||||
.get_or_create_owned(&labels),
|
||||
spec_decode_num_accepted_tokens: metrics
|
||||
.spec_decode_num_accepted_tokens
|
||||
.get_or_create_owned(&labels),
|
||||
spec_decode_num_accepted_tokens_per_pos: metrics
|
||||
.spec_decode_num_accepted_tokens_per_pos
|
||||
.clone(),
|
||||
log_stats: metrics.log_stats.get_or_create_owned(&labels),
|
||||
estimated_flops_per_gpu: metrics.estimated_flops_per_gpu.get_or_create_owned(&labels),
|
||||
estimated_read_bytes_per_gpu: metrics
|
||||
.estimated_read_bytes_per_gpu
|
||||
.get_or_create_owned(&labels),
|
||||
estimated_write_bytes_per_gpu: metrics
|
||||
.estimated_write_bytes_per_gpu
|
||||
.get_or_create_owned(&labels),
|
||||
kv_block_lifetime_seconds: metrics.kv_block_lifetime_seconds.get_or_create_owned(&labels),
|
||||
kv_block_idle_before_evict_seconds: metrics
|
||||
.kv_block_idle_before_evict_seconds
|
||||
.get_or_create_owned(&labels),
|
||||
kv_block_reuse_gap_seconds: metrics.kv_block_reuse_gap_seconds.get_or_create_owned(&labels),
|
||||
labels,
|
||||
}
|
||||
}
|
||||
|
||||
/// Record scheduler-stats values through pre-resolved metric handles.
|
||||
fn record_scheduler_stats_with_handles(handles: &SchedulerStatsHandles, stats: &SchedulerStats) {
|
||||
// Scheduler state gauges.
|
||||
handles.scheduler_running.set(stats.num_running_reqs);
|
||||
handles
|
||||
.scheduler_waiting
|
||||
.set(stats.num_waiting_reqs + stats.num_skipped_waiting_reqs);
|
||||
handles.scheduler_waiting_capacity.set(stats.num_waiting_reqs);
|
||||
handles.scheduler_waiting_deferred.set(stats.num_skipped_waiting_reqs);
|
||||
handles.kv_cache_usage.set(stats.kv_cache_usage);
|
||||
|
||||
// Prefix-cache counters, including the connector-backed external cache path.
|
||||
handles.prefix_cache_queries.inc_by(stats.prefix_cache_stats.base.queries);
|
||||
handles.prefix_cache_hits.inc_by(stats.prefix_cache_stats.base.hits);
|
||||
|
||||
if let Some(connector_prefix_cache_stats) = &stats.connector_prefix_cache_stats {
|
||||
metrics
|
||||
handles
|
||||
.external_prefix_cache_queries
|
||||
.get_or_create(&labels)
|
||||
.inc_by(connector_prefix_cache_stats.base.queries);
|
||||
metrics
|
||||
handles
|
||||
.external_prefix_cache_hits
|
||||
.get_or_create(&labels)
|
||||
.inc_by(connector_prefix_cache_stats.base.hits);
|
||||
}
|
||||
|
||||
// Speculative decoding counters.
|
||||
if let Some(spec_decoding_stats) = &stats.spec_decoding_stats {
|
||||
metrics
|
||||
.spec_decode_num_drafts
|
||||
.get_or_create(&labels)
|
||||
.inc_by(spec_decoding_stats.num_drafts);
|
||||
metrics
|
||||
handles.spec_decode_num_drafts.inc_by(spec_decoding_stats.num_drafts);
|
||||
handles
|
||||
.spec_decode_num_draft_tokens
|
||||
.get_or_create(&labels)
|
||||
.inc_by(spec_decoding_stats.num_draft_tokens);
|
||||
metrics
|
||||
handles
|
||||
.spec_decode_num_accepted_tokens
|
||||
.get_or_create(&labels)
|
||||
.inc_by(spec_decoding_stats.num_accepted_tokens);
|
||||
metrics.log_stats.get_or_create(&labels).observe_spec_decode(
|
||||
handles.log_stats.observe_spec_decode(
|
||||
spec_decoding_stats.num_drafts,
|
||||
&spec_decoding_stats.num_accepted_tokens_per_pos,
|
||||
);
|
||||
@@ -92,11 +192,11 @@ pub(crate) fn record_scheduler_stats(
|
||||
for (position, accepted_tokens) in
|
||||
spec_decoding_stats.num_accepted_tokens_per_pos.iter().copied().enumerate()
|
||||
{
|
||||
metrics
|
||||
handles
|
||||
.spec_decode_num_accepted_tokens_per_pos
|
||||
.get_or_create(&EnginePositionLabels {
|
||||
model_name: model_name.clone(),
|
||||
engine,
|
||||
model_name: handles.labels.model_name.clone(),
|
||||
engine: handles.labels.engine,
|
||||
position: position as u32,
|
||||
})
|
||||
.inc_by(accepted_tokens);
|
||||
@@ -109,22 +209,13 @@ pub(crate) fn record_scheduler_stats(
|
||||
|| perf_stats.num_read_bytes_per_gpu != 0
|
||||
|| perf_stats.num_write_bytes_per_gpu != 0)
|
||||
{
|
||||
metrics
|
||||
.estimated_flops_per_gpu
|
||||
.get_or_create(&labels)
|
||||
.inc_by(perf_stats.num_flops_per_gpu);
|
||||
metrics
|
||||
.estimated_read_bytes_per_gpu
|
||||
.get_or_create(&labels)
|
||||
.inc_by(perf_stats.num_read_bytes_per_gpu);
|
||||
metrics
|
||||
.estimated_write_bytes_per_gpu
|
||||
.get_or_create(&labels)
|
||||
.inc_by(perf_stats.num_write_bytes_per_gpu);
|
||||
handles.estimated_flops_per_gpu.inc_by(perf_stats.num_flops_per_gpu);
|
||||
handles.estimated_read_bytes_per_gpu.inc_by(perf_stats.num_read_bytes_per_gpu);
|
||||
handles.estimated_write_bytes_per_gpu.inc_by(perf_stats.num_write_bytes_per_gpu);
|
||||
}
|
||||
|
||||
if let Some(cudagraph_stats) = &stats.cudagraph_stats {
|
||||
metrics.log_stats.get_or_create(&labels).observe_cudagraph(
|
||||
handles.log_stats.observe_cudagraph(
|
||||
cudagraph_stats.num_unpadded_tokens,
|
||||
cudagraph_stats.num_padded_tokens,
|
||||
cudagraph_stats.num_paddings,
|
||||
@@ -134,16 +225,11 @@ pub(crate) fn record_scheduler_stats(
|
||||
|
||||
// Sampled KV-cache residency histograms.
|
||||
if !stats.kv_cache_eviction_events.is_empty() {
|
||||
let kv_block_lifetime_seconds = metrics.kv_block_lifetime_seconds.get_or_create(&labels);
|
||||
let kv_block_idle_before_evict_seconds =
|
||||
metrics.kv_block_idle_before_evict_seconds.get_or_create(&labels);
|
||||
let kv_block_reuse_gap_seconds = metrics.kv_block_reuse_gap_seconds.get_or_create(&labels);
|
||||
|
||||
for event in &stats.kv_cache_eviction_events {
|
||||
kv_block_lifetime_seconds.observe(event.lifetime_seconds);
|
||||
kv_block_idle_before_evict_seconds.observe(event.idle_seconds);
|
||||
handles.kv_block_lifetime_seconds.observe(event.lifetime_seconds);
|
||||
handles.kv_block_idle_before_evict_seconds.observe(event.idle_seconds);
|
||||
for reuse_gap_seconds in &event.reuse_gaps_seconds {
|
||||
kv_block_reuse_gap_seconds.observe(*reuse_gap_seconds);
|
||||
handles.kv_block_reuse_gap_seconds.observe(*reuse_gap_seconds);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+11
-4
@@ -88,14 +88,21 @@ impl Llm {
|
||||
// Record internal engine-core request ID in the current tracing span.
|
||||
Span::current().record("engine_request_id", &internal_request_id);
|
||||
|
||||
let arrival_time = prepared.engine_request.arrival_time;
|
||||
let max_tokens_param =
|
||||
(prepared.engine_request.sampling_params.as_ref()).map(|p| p.max_tokens);
|
||||
let prompt_len = prepared.prompt_token_ids().len() as u32;
|
||||
|
||||
let stream = self.client.call(prepared.engine_request).await?;
|
||||
|
||||
let request_metrics = RequestMetricsTracker::new(
|
||||
self.client.model_name().to_string(),
|
||||
prepared.engine_request.arrival_time,
|
||||
prepared.prompt_token_ids().len() as u32,
|
||||
(prepared.engine_request.sampling_params.as_ref()).map(|p| p.max_tokens),
|
||||
stream.engine_index(),
|
||||
arrival_time,
|
||||
prompt_len,
|
||||
max_tokens_param,
|
||||
1,
|
||||
);
|
||||
let stream = self.client.call(prepared.engine_request).await?;
|
||||
let guard = self.inflight.track(external_request_id, internal_request_id);
|
||||
|
||||
Ok(GenerateOutputStream::new(
|
||||
|
||||
@@ -248,12 +248,7 @@ impl Stream for GenerateOutputStream {
|
||||
};
|
||||
|
||||
let received_at = current_unix_timestamp_secs();
|
||||
self.request_metrics.observe_output(
|
||||
raw.engine_index,
|
||||
raw.timestamp,
|
||||
received_at,
|
||||
&raw.output,
|
||||
);
|
||||
self.request_metrics.observe_output(raw.timestamp, received_at, &raw.output);
|
||||
|
||||
let raw = raw.output;
|
||||
|
||||
|
||||
+150
-142
@@ -5,15 +5,12 @@ use vllm_engine_core_client::protocol::output::{
|
||||
};
|
||||
use vllm_engine_core_client::protocol::stats::PrefillStats;
|
||||
use vllm_metrics::{
|
||||
EngineLabels, FinishedReasonLabels, METRICS, PromptTokenSourceLabels, RequestMetrics,
|
||||
EngineLabels, Family, FinishedReasonLabels, HistogramMetric, METRICS, PromptTokenSourceLabels,
|
||||
U64Counter,
|
||||
};
|
||||
|
||||
use crate::FinishReason;
|
||||
|
||||
fn metrics() -> &'static RequestMetrics {
|
||||
&METRICS.request
|
||||
}
|
||||
|
||||
const PROMPT_TOKEN_SOURCE_LOCAL_COMPUTE: &str = "local_compute";
|
||||
const PROMPT_TOKEN_SOURCE_LOCAL_CACHE_HIT: &str = "local_cache_hit";
|
||||
const PROMPT_TOKEN_SOURCE_EXTERNAL_KV_TRANSFER: &str = "external_kv_transfer";
|
||||
@@ -29,9 +26,11 @@ const PROMPT_TOKEN_SOURCE_EXTERNAL_KV_TRANSFER: &str = "external_kv_transfer";
|
||||
///
|
||||
/// Original Python update flow:
|
||||
/// <https://github.com/vllm-project/vllm/blob/bc2c0c86efb28e77677a3cfb8687e976914a313a/vllm/v1/engine/output_processor.py#L600-L677>
|
||||
#[derive(Debug, Clone)]
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct RequestMetricsTracker {
|
||||
model_name: String,
|
||||
/// Cached request metric handles for this request's model and engine index.
|
||||
handles: RequestMetricHandles,
|
||||
|
||||
arrival_time: f64,
|
||||
prompt_len: u32,
|
||||
max_tokens_param: Option<u32>,
|
||||
@@ -44,7 +43,38 @@ pub(crate) struct RequestMetricsTracker {
|
||||
first_token_latency: f64,
|
||||
num_generation_tokens: u32,
|
||||
latest_num_cached_tokens: u32,
|
||||
last_seen_engine_index: u32,
|
||||
}
|
||||
|
||||
/// Cached request metric handles for one model and engine index.
|
||||
#[derive(Clone)]
|
||||
struct RequestMetricHandles {
|
||||
labels: EngineLabels,
|
||||
|
||||
// Request-derived counters.
|
||||
num_preemptions: U64Counter,
|
||||
prompt_tokens: U64Counter,
|
||||
prompt_tokens_local_compute: U64Counter,
|
||||
prompt_tokens_local_cache_hit: U64Counter,
|
||||
prompt_tokens_external_kv_transfer: U64Counter,
|
||||
prompt_tokens_cached: U64Counter,
|
||||
generation_tokens: U64Counter,
|
||||
|
||||
// Request lifecycle counters and histograms.
|
||||
request_success: Family<FinishedReasonLabels, U64Counter>,
|
||||
request_prompt_tokens: HistogramMetric,
|
||||
request_generation_tokens: HistogramMetric,
|
||||
request_max_num_generation_tokens: HistogramMetric,
|
||||
request_params_max_tokens: HistogramMetric,
|
||||
request_params_n: HistogramMetric,
|
||||
request_prefill_kv_computed_tokens: HistogramMetric,
|
||||
time_to_first_token_seconds: HistogramMetric,
|
||||
inter_token_latency_seconds: HistogramMetric,
|
||||
e2e_request_latency_seconds: HistogramMetric,
|
||||
request_queue_time_seconds: HistogramMetric,
|
||||
request_prefill_time_seconds: HistogramMetric,
|
||||
request_decode_time_seconds: HistogramMetric,
|
||||
request_inference_time_seconds: HistogramMetric,
|
||||
request_time_per_output_token_seconds: HistogramMetric,
|
||||
}
|
||||
|
||||
impl RequestMetricsTracker {
|
||||
@@ -52,13 +82,14 @@ impl RequestMetricsTracker {
|
||||
/// context.
|
||||
pub(crate) fn new(
|
||||
model_name: String,
|
||||
engine_index: u32,
|
||||
arrival_time: f64,
|
||||
prompt_len: u32,
|
||||
max_tokens_param: Option<u32>,
|
||||
n_param: u32,
|
||||
) -> Self {
|
||||
Self {
|
||||
model_name,
|
||||
handles: resolve_request_metric_handles(&model_name, engine_index),
|
||||
arrival_time,
|
||||
prompt_len,
|
||||
max_tokens_param,
|
||||
@@ -71,7 +102,6 @@ impl RequestMetricsTracker {
|
||||
first_token_latency: 0.0,
|
||||
num_generation_tokens: 0,
|
||||
latest_num_cached_tokens: 0,
|
||||
last_seen_engine_index: 0,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -81,23 +111,18 @@ impl RequestMetricsTracker {
|
||||
/// <https://github.com/vllm-project/vllm/blob/bc2c0c86efb28e77677a3cfb8687e976914a313a/vllm/v1/metrics/stats.py#L331-L384>
|
||||
pub(crate) fn observe_output(
|
||||
&mut self,
|
||||
engine_index: u32,
|
||||
batch_timestamp: f64,
|
||||
received_at: f64,
|
||||
output: &EngineCoreOutput,
|
||||
) {
|
||||
self.last_seen_engine_index = engine_index;
|
||||
if let Some(prefill_stats) = &output.prefill_stats {
|
||||
self.latest_num_cached_tokens = prefill_stats.num_cached_tokens;
|
||||
}
|
||||
self.num_generation_tokens += output.new_token_ids.len() as u32;
|
||||
metrics()
|
||||
.generation_tokens
|
||||
.get_or_create(&engine_labels(&self.model_name, engine_index))
|
||||
.inc_by(output.new_token_ids.len() as u64);
|
||||
self.handles.generation_tokens.inc_by(output.new_token_ids.len() as u64);
|
||||
|
||||
if let Some(events) = &output.events {
|
||||
self.observe_events(engine_index, events);
|
||||
self.observe_events(events);
|
||||
}
|
||||
|
||||
// Only outputs that actually carry tokens drive token-timing metrics.
|
||||
@@ -107,22 +132,16 @@ impl RequestMetricsTracker {
|
||||
if !output.new_token_ids.is_empty() {
|
||||
if self.is_prefilling {
|
||||
if let Some(prefill_stats) = &output.prefill_stats {
|
||||
record_prompt_tokens(&self.model_name, engine_index, prefill_stats);
|
||||
self.record_prompt_tokens(prefill_stats);
|
||||
}
|
||||
self.first_token_latency = received_at - self.arrival_time;
|
||||
observe_time_to_first_token_seconds(
|
||||
&self.model_name,
|
||||
engine_index,
|
||||
self.first_token_latency,
|
||||
);
|
||||
self.handles.time_to_first_token_seconds.observe(self.first_token_latency);
|
||||
self.first_token_ts = batch_timestamp;
|
||||
self.is_prefilling = false;
|
||||
} else if self.last_token_ts > 0.0 {
|
||||
observe_inter_token_latency_seconds(
|
||||
&self.model_name,
|
||||
engine_index,
|
||||
batch_timestamp - self.last_token_ts,
|
||||
);
|
||||
self.handles
|
||||
.inter_token_latency_seconds
|
||||
.observe(batch_timestamp - self.last_token_ts);
|
||||
}
|
||||
|
||||
self.last_token_ts = batch_timestamp;
|
||||
@@ -135,7 +154,6 @@ impl RequestMetricsTracker {
|
||||
/// Original Python finished-request stats:
|
||||
/// <https://github.com/vllm-project/vllm/blob/bc2c0c86efb28e77677a3cfb8687e976914a313a/vllm/v1/metrics/stats.py#L222-L237>
|
||||
pub(crate) fn record_finished(&self, received_at: f64, finish_reason: FinishReason) {
|
||||
let labels = engine_labels(&self.model_name, self.last_seen_engine_index);
|
||||
let prefill_kv_computed_tokens =
|
||||
self.prompt_len.saturating_sub(self.latest_num_cached_tokens);
|
||||
let e2e_latency_seconds = received_at - self.arrival_time;
|
||||
@@ -150,57 +168,47 @@ impl RequestMetricsTracker {
|
||||
0.0
|
||||
};
|
||||
|
||||
record_request_success(&self.model_name, self.last_seen_engine_index, finish_reason);
|
||||
metrics()
|
||||
.request_prompt_tokens
|
||||
.get_or_create(&labels)
|
||||
.observe(self.prompt_len as f64);
|
||||
metrics()
|
||||
self.record_request_success(finish_reason);
|
||||
|
||||
self.handles.request_prompt_tokens.observe(self.prompt_len as f64);
|
||||
self.handles
|
||||
.request_generation_tokens
|
||||
.get_or_create(&labels)
|
||||
.observe(self.num_generation_tokens as f64);
|
||||
metrics()
|
||||
self.handles
|
||||
.request_max_num_generation_tokens
|
||||
.get_or_create(&labels)
|
||||
.observe(self.num_generation_tokens as f64);
|
||||
if let Some(max_tokens_param) = self.max_tokens_param {
|
||||
metrics()
|
||||
.request_params_max_tokens
|
||||
.get_or_create(&labels)
|
||||
.observe(max_tokens_param as f64);
|
||||
self.handles.request_params_max_tokens.observe(max_tokens_param as f64);
|
||||
}
|
||||
metrics().request_params_n.get_or_create(&labels).observe(self.n_param as f64);
|
||||
metrics()
|
||||
self.handles.request_params_n.observe(self.n_param as f64);
|
||||
self.handles
|
||||
.request_prefill_kv_computed_tokens
|
||||
.get_or_create(&labels)
|
||||
.observe(prefill_kv_computed_tokens as f64);
|
||||
metrics()
|
||||
.e2e_request_latency_seconds
|
||||
.get_or_create(&labels)
|
||||
.observe(e2e_latency_seconds);
|
||||
metrics()
|
||||
.request_queue_time_seconds
|
||||
.get_or_create(&labels)
|
||||
.observe(queue_time_seconds);
|
||||
metrics()
|
||||
.request_prefill_time_seconds
|
||||
.get_or_create(&labels)
|
||||
.observe(prefill_time_seconds);
|
||||
metrics()
|
||||
.request_decode_time_seconds
|
||||
.get_or_create(&labels)
|
||||
.observe(decode_time_seconds);
|
||||
metrics()
|
||||
.request_inference_time_seconds
|
||||
.get_or_create(&labels)
|
||||
.observe(inference_time_seconds);
|
||||
metrics()
|
||||
self.handles.e2e_request_latency_seconds.observe(e2e_latency_seconds);
|
||||
self.handles.request_queue_time_seconds.observe(queue_time_seconds);
|
||||
self.handles.request_prefill_time_seconds.observe(prefill_time_seconds);
|
||||
self.handles.request_decode_time_seconds.observe(decode_time_seconds);
|
||||
self.handles.request_inference_time_seconds.observe(inference_time_seconds);
|
||||
self.handles
|
||||
.request_time_per_output_token_seconds
|
||||
.get_or_create(&labels)
|
||||
.observe(time_per_output_token_seconds);
|
||||
}
|
||||
|
||||
fn observe_events(&mut self, engine_index: u32, events: &[EngineCoreEvent]) {
|
||||
/// Record prompt token counters through cached metric handles.
|
||||
fn record_prompt_tokens(&self, prefill_stats: &PrefillStats) {
|
||||
let computed = prefill_stats.num_computed_tokens as u64;
|
||||
let local_cache_hit = prefill_stats.num_local_cached_tokens as u64;
|
||||
let external_kv_transfer = prefill_stats.num_external_cached_tokens as u64;
|
||||
|
||||
self.handles.prompt_tokens.inc_by(prefill_stats.num_prompt_tokens as u64);
|
||||
self.handles.prompt_tokens_local_compute.inc_by(computed);
|
||||
self.handles.prompt_tokens_local_cache_hit.inc_by(local_cache_hit);
|
||||
self.handles.prompt_tokens_external_kv_transfer.inc_by(external_kv_transfer);
|
||||
self.handles.prompt_tokens_cached.inc_by(prefill_stats.num_cached_tokens as u64);
|
||||
}
|
||||
|
||||
/// Record request event counters through cached metric handles.
|
||||
fn observe_events(&mut self, events: &[EngineCoreEvent]) {
|
||||
for event in events {
|
||||
match event.r#type {
|
||||
EngineCoreEventType::Queued => {
|
||||
@@ -212,46 +220,86 @@ impl RequestMetricsTracker {
|
||||
}
|
||||
}
|
||||
EngineCoreEventType::Preempted => {
|
||||
metrics()
|
||||
.num_preemptions
|
||||
.get_or_create(&engine_labels(&self.model_name, engine_index))
|
||||
.inc();
|
||||
self.handles.num_preemptions.inc();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn engine_labels(model_name: &str, engine: u32) -> EngineLabels {
|
||||
EngineLabels {
|
||||
model_name: model_name.to_string(),
|
||||
engine,
|
||||
/// Increment the request-success counter for the terminal finish reason.
|
||||
fn record_request_success(&self, finish_reason: FinishReason) {
|
||||
self.handles
|
||||
.request_success
|
||||
.get_or_create(&FinishedReasonLabels {
|
||||
model_name: self.handles.labels.model_name.clone(),
|
||||
engine: self.handles.labels.engine,
|
||||
finished_reason: finish_reason.as_str(),
|
||||
})
|
||||
.inc();
|
||||
}
|
||||
}
|
||||
|
||||
fn observe_time_to_first_token_seconds(model_name: &str, engine: u32, seconds: f64) {
|
||||
metrics()
|
||||
.time_to_first_token_seconds
|
||||
.get_or_create(&engine_labels(model_name, engine))
|
||||
.observe(seconds);
|
||||
}
|
||||
/// Resolve fixed request metric handles for one model and engine index.
|
||||
fn resolve_request_metric_handles(model_name: &str, engine: u32) -> RequestMetricHandles {
|
||||
let metrics = &METRICS.request;
|
||||
let labels = EngineLabels {
|
||||
model_name: model_name.to_string(),
|
||||
engine,
|
||||
};
|
||||
|
||||
fn observe_inter_token_latency_seconds(model_name: &str, engine: u32, seconds: f64) {
|
||||
metrics()
|
||||
.inter_token_latency_seconds
|
||||
.get_or_create(&engine_labels(model_name, engine))
|
||||
.observe(seconds);
|
||||
}
|
||||
|
||||
fn record_request_success(model_name: &str, engine: u32, finish_reason: FinishReason) {
|
||||
metrics()
|
||||
.request_success
|
||||
.get_or_create(&FinishedReasonLabels {
|
||||
model_name: model_name.to_string(),
|
||||
engine,
|
||||
finished_reason: finish_reason.as_str(),
|
||||
})
|
||||
.inc();
|
||||
RequestMetricHandles {
|
||||
num_preemptions: metrics.num_preemptions.get_or_create_owned(&labels),
|
||||
prompt_tokens: metrics.prompt_tokens.get_or_create_owned(&labels),
|
||||
prompt_tokens_local_compute: metrics.prompt_tokens_by_source.get_or_create_owned(
|
||||
&prompt_token_source_labels(model_name, engine, PROMPT_TOKEN_SOURCE_LOCAL_COMPUTE),
|
||||
),
|
||||
prompt_tokens_local_cache_hit: metrics.prompt_tokens_by_source.get_or_create_owned(
|
||||
&prompt_token_source_labels(model_name, engine, PROMPT_TOKEN_SOURCE_LOCAL_CACHE_HIT),
|
||||
),
|
||||
prompt_tokens_external_kv_transfer: metrics.prompt_tokens_by_source.get_or_create_owned(
|
||||
&prompt_token_source_labels(
|
||||
model_name,
|
||||
engine,
|
||||
PROMPT_TOKEN_SOURCE_EXTERNAL_KV_TRANSFER,
|
||||
),
|
||||
),
|
||||
prompt_tokens_cached: metrics.prompt_tokens_cached.get_or_create_owned(&labels),
|
||||
generation_tokens: metrics.generation_tokens.get_or_create_owned(&labels),
|
||||
request_success: metrics.request_success.clone(),
|
||||
request_prompt_tokens: metrics.request_prompt_tokens.get_or_create_owned(&labels),
|
||||
request_generation_tokens: metrics.request_generation_tokens.get_or_create_owned(&labels),
|
||||
request_max_num_generation_tokens: metrics
|
||||
.request_max_num_generation_tokens
|
||||
.get_or_create_owned(&labels),
|
||||
request_params_max_tokens: metrics.request_params_max_tokens.get_or_create_owned(&labels),
|
||||
request_params_n: metrics.request_params_n.get_or_create_owned(&labels),
|
||||
request_prefill_kv_computed_tokens: metrics
|
||||
.request_prefill_kv_computed_tokens
|
||||
.get_or_create_owned(&labels),
|
||||
time_to_first_token_seconds: metrics
|
||||
.time_to_first_token_seconds
|
||||
.get_or_create_owned(&labels),
|
||||
inter_token_latency_seconds: metrics
|
||||
.inter_token_latency_seconds
|
||||
.get_or_create_owned(&labels),
|
||||
e2e_request_latency_seconds: metrics
|
||||
.e2e_request_latency_seconds
|
||||
.get_or_create_owned(&labels),
|
||||
request_queue_time_seconds: metrics.request_queue_time_seconds.get_or_create_owned(&labels),
|
||||
request_prefill_time_seconds: metrics
|
||||
.request_prefill_time_seconds
|
||||
.get_or_create_owned(&labels),
|
||||
request_decode_time_seconds: metrics
|
||||
.request_decode_time_seconds
|
||||
.get_or_create_owned(&labels),
|
||||
request_inference_time_seconds: metrics
|
||||
.request_inference_time_seconds
|
||||
.get_or_create_owned(&labels),
|
||||
request_time_per_output_token_seconds: metrics
|
||||
.request_time_per_output_token_seconds
|
||||
.get_or_create_owned(&labels),
|
||||
labels,
|
||||
}
|
||||
}
|
||||
|
||||
fn prompt_token_source_labels(
|
||||
@@ -266,45 +314,6 @@ fn prompt_token_source_labels(
|
||||
}
|
||||
}
|
||||
|
||||
fn record_prompt_tokens(model_name: &str, engine: u32, prefill_stats: &PrefillStats) {
|
||||
let computed = prefill_stats.num_computed_tokens as u64;
|
||||
let local_cache_hit = prefill_stats.num_local_cached_tokens as u64;
|
||||
let external_kv_transfer = prefill_stats.num_external_cached_tokens as u64;
|
||||
|
||||
metrics()
|
||||
.prompt_tokens
|
||||
.get_or_create(&engine_labels(model_name, engine))
|
||||
.inc_by(prefill_stats.num_prompt_tokens as u64);
|
||||
metrics()
|
||||
.prompt_tokens_by_source
|
||||
.get_or_create(&prompt_token_source_labels(
|
||||
model_name,
|
||||
engine,
|
||||
PROMPT_TOKEN_SOURCE_LOCAL_COMPUTE,
|
||||
))
|
||||
.inc_by(computed);
|
||||
metrics()
|
||||
.prompt_tokens_by_source
|
||||
.get_or_create(&prompt_token_source_labels(
|
||||
model_name,
|
||||
engine,
|
||||
PROMPT_TOKEN_SOURCE_LOCAL_CACHE_HIT,
|
||||
))
|
||||
.inc_by(local_cache_hit);
|
||||
metrics()
|
||||
.prompt_tokens_by_source
|
||||
.get_or_create(&prompt_token_source_labels(
|
||||
model_name,
|
||||
engine,
|
||||
PROMPT_TOKEN_SOURCE_EXTERNAL_KV_TRANSFER,
|
||||
))
|
||||
.inc_by(external_kv_transfer);
|
||||
metrics()
|
||||
.prompt_tokens_cached
|
||||
.get_or_create(&engine_labels(model_name, engine))
|
||||
.inc_by(prefill_stats.num_cached_tokens as u64);
|
||||
}
|
||||
|
||||
fn diff_or_zero(end: f64, start: f64) -> f64 {
|
||||
if end > 0.0 && start > 0.0 && end >= start {
|
||||
end - start
|
||||
@@ -337,10 +346,10 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn tracker_updates_timing_state_across_prefill_decode_and_finish() {
|
||||
let mut tracker = RequestMetricsTracker::new("model".to_string(), 100.0, 64, Some(128), 1);
|
||||
let mut tracker =
|
||||
RequestMetricsTracker::new("model".to_string(), 2, 100.0, 64, Some(128), 1);
|
||||
|
||||
tracker.observe_output(
|
||||
2,
|
||||
10.0,
|
||||
100.2,
|
||||
&vllm_engine_core_client::protocol::output::EngineCoreOutput {
|
||||
@@ -368,7 +377,6 @@ mod tests {
|
||||
},
|
||||
);
|
||||
tracker.observe_output(
|
||||
2,
|
||||
11.5,
|
||||
100.4,
|
||||
&vllm_engine_core_client::protocol::output::EngineCoreOutput {
|
||||
@@ -384,7 +392,7 @@ mod tests {
|
||||
);
|
||||
|
||||
assert!(!tracker.is_prefilling);
|
||||
assert_eq!(tracker.last_seen_engine_index, 2);
|
||||
assert_eq!(tracker.handles.labels.engine, 2);
|
||||
assert_eq!(tracker.num_generation_tokens, 3);
|
||||
assert_eq!(tracker.queued_ts, 8.0);
|
||||
assert_eq!(tracker.scheduled_ts, 9.0);
|
||||
|
||||
@@ -17,7 +17,7 @@ use vllm_engine_core_client::protocol::request::EngineCoreRequest;
|
||||
use vllm_engine_core_client::protocol::sampling::EngineCoreSamplingParams;
|
||||
use vllm_engine_core_client::protocol::stats::PrefillStats;
|
||||
use vllm_engine_core_client::test_utils::{IpcNamespace, spawn_mock_engine_task};
|
||||
use vllm_engine_core_client::{EngineCoreClient, EngineCoreClientConfig};
|
||||
use vllm_engine_core_client::{EngineCoreClient, EngineCoreClientConfig, EngineId};
|
||||
use vllm_llm::{
|
||||
Error, FinishReason, GenerateOutputStreamExt as _, GeneratePromptInfo, GenerateRequest, Llm,
|
||||
};
|
||||
@@ -699,7 +699,7 @@ async fn abort_by_external_id_aborts_all_internal_requests() {
|
||||
async fn generate_records_request_metrics_in_prometheus_output() {
|
||||
let ipc = IpcNamespace::new().unwrap();
|
||||
let handshake_address = ipc.handshake_endpoint();
|
||||
let engine_id = b"engine-metrics".to_vec();
|
||||
let engine_id = EngineId::from_engine_index(4);
|
||||
let model_name = request_metrics_model_name("metrics-model");
|
||||
|
||||
let (shutdown_tx, engine_task) = spawn_mock_engine_task(
|
||||
@@ -832,7 +832,7 @@ async fn generate_records_request_metrics_in_prometheus_output() {
|
||||
async fn dropping_stream_records_abort_terminal_request_metrics() {
|
||||
let ipc = IpcNamespace::new().unwrap();
|
||||
let handshake_address = ipc.handshake_endpoint();
|
||||
let engine_id = b"engine-metrics-drop".to_vec();
|
||||
let engine_id = EngineId::from_engine_index(5);
|
||||
let model_name = request_metrics_model_name("metrics-drop-model");
|
||||
|
||||
let (shutdown_tx, engine_task) = spawn_mock_engine_task(
|
||||
|
||||
@@ -4,7 +4,7 @@ use std::sync::atomic::AtomicU64;
|
||||
|
||||
use prometheus_client::encoding::text::encode;
|
||||
use prometheus_client::metrics::counter::Counter;
|
||||
use prometheus_client::metrics::family::Family;
|
||||
pub use prometheus_client::metrics::family::Family;
|
||||
use prometheus_client::metrics::gauge::Gauge;
|
||||
use prometheus_client::metrics::histogram::Histogram;
|
||||
use prometheus_client::registry::Registry;
|
||||
@@ -23,6 +23,8 @@ pub use scheduler::*;
|
||||
pub type U64Counter = Counter<u64, AtomicU64>;
|
||||
pub type U64Gauge = Gauge<u64, AtomicU64>;
|
||||
pub type F64Gauge = Gauge<f64, AtomicU64>;
|
||||
/// Histogram metric handle cloned out of a Prometheus family.
|
||||
pub type HistogramMetric = Histogram;
|
||||
pub(crate) type HistogramFamily = Family<EngineLabels, Histogram, fn() -> Histogram>;
|
||||
|
||||
/// Shared Prometheus registry for frontend metrics.
|
||||
|
||||
@@ -46,15 +46,6 @@ pub struct WaitingReasonLabels {
|
||||
pub reason: &'static str,
|
||||
}
|
||||
|
||||
/// Labels for `vllm:engine_sleep_state`.
|
||||
#[derive(Clone, Debug, Hash, PartialEq, Eq, EncodeLabelSet)]
|
||||
pub struct SleepStateLabels {
|
||||
pub model_name: String,
|
||||
pub engine: u32,
|
||||
/// One of `awake`, `weights_offloaded`, or `discard_all`.
|
||||
pub sleep_state: &'static str,
|
||||
}
|
||||
|
||||
/// Adapter names encoded as a deterministic comma-joined Prometheus label value.
|
||||
#[derive(Clone, Debug, Hash, PartialEq, Eq)]
|
||||
pub struct LoraAdapterNames(pub BTreeSet<String>);
|
||||
@@ -164,10 +155,6 @@ pub struct SchedulerMetrics {
|
||||
pub scheduler_waiting_by_reason: Family<WaitingReasonLabels, U64Gauge>,
|
||||
pub kv_cache_usage: Family<EngineLabels, F64Gauge>,
|
||||
|
||||
/// `vllm:engine_sleep_state`, driven by the frontend `/sleep` and
|
||||
/// `/wake_up` routes (mirroring Python `record_sleep_state`).
|
||||
pub engine_sleep_state: Family<SleepStateLabels, U64Gauge>,
|
||||
|
||||
/// `vllm:lora_requests_info`. Value is the emit-time unix timestamp in
|
||||
/// seconds.
|
||||
pub lora_info: Family<LoraInfoLabels, F64Gauge>,
|
||||
@@ -234,16 +221,6 @@ impl SchedulerMetrics {
|
||||
kv_cache_usage.clone(),
|
||||
);
|
||||
|
||||
let engine_sleep_state = Family::default();
|
||||
registry.register(
|
||||
"vllm:engine_sleep_state",
|
||||
"Engine sleep state; awake = 0 means engine is sleeping; \
|
||||
awake = 1 means engine is awake; \
|
||||
weights_offloaded = 1 means sleep level 1; \
|
||||
discard_all = 1 means sleep level 2.",
|
||||
engine_sleep_state.clone(),
|
||||
);
|
||||
|
||||
let lora_info = Family::default();
|
||||
registry.register(
|
||||
"vllm:lora_requests_info",
|
||||
@@ -361,7 +338,6 @@ impl SchedulerMetrics {
|
||||
scheduler_waiting,
|
||||
scheduler_waiting_by_reason,
|
||||
kv_cache_usage,
|
||||
engine_sleep_state,
|
||||
lora_info,
|
||||
prefix_cache_queries,
|
||||
prefix_cache_hits,
|
||||
|
||||
@@ -10,7 +10,6 @@ asynk-strim-attr.workspace = true
|
||||
auto_enums.workspace = true
|
||||
axum.workspace = true
|
||||
educe.workspace = true
|
||||
form_urlencoded.workspace = true
|
||||
futures.workspace = true
|
||||
http-body.workspace = true
|
||||
hyper.workspace = true
|
||||
|
||||
@@ -70,10 +70,6 @@ fn build_router_with_options(
|
||||
dev_mode_enabled: bool,
|
||||
runtime_lora_updating_enabled: bool,
|
||||
) -> Router {
|
||||
// Default `vllm:engine_sleep_state` to awake, matching the init-time
|
||||
// default recorded by the Python stat logger.
|
||||
sleep::record_sleep_state(&state, None);
|
||||
|
||||
let mut router = Router::new()
|
||||
// Health & monitoring
|
||||
.route("/health", get(health::health))
|
||||
|
||||
@@ -29,9 +29,7 @@ pub struct GenerateRequest {
|
||||
impl Normalizable for GenerateRequest {}
|
||||
|
||||
/// Mirrors the Python vLLM `GenerateResponseChoice` class.
|
||||
///
|
||||
/// Do not skip serializing `None` fields here: non-streaming response types
|
||||
/// should serialize `None` as explicit `null`.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub(super) struct GenerateResponseChoice {
|
||||
pub index: u32,
|
||||
@@ -60,6 +58,7 @@ pub(super) struct GenerateStreamResponse {
|
||||
}
|
||||
|
||||
/// Mirrors the Python vLLM `GenerateResponse` class.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub(super) struct GenerateResponse {
|
||||
pub request_id: String,
|
||||
@@ -69,6 +68,7 @@ pub(super) struct GenerateResponse {
|
||||
}
|
||||
|
||||
/// Mirrors the Python vLLM `Logprob` class used in prompt-logprobs payloads.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub(super) struct GenerateLogprob {
|
||||
pub logprob: f32,
|
||||
|
||||
@@ -211,7 +211,7 @@ async fn collect_chat_completion(
|
||||
Some(prefix) => Some(format!("{prefix}{}", message.text())),
|
||||
None => Some(message.text()).filter(|t| !t.is_empty()),
|
||||
},
|
||||
tool_calls,
|
||||
tool_calls: Some(tool_calls).filter(|calls| !calls.is_empty()),
|
||||
reasoning: if include_reasoning { reasoning } else { None },
|
||||
},
|
||||
logprobs,
|
||||
|
||||
@@ -830,7 +830,7 @@ mod tests {
|
||||
let message = ChatCompletionMessage {
|
||||
role: AssistantRole,
|
||||
content: Some("answer".to_string()),
|
||||
tool_calls: Vec::new(),
|
||||
tool_calls: None,
|
||||
reasoning: Some("inner".to_string()),
|
||||
};
|
||||
let message_json = serde_json::to_value(message).expect("message serializes");
|
||||
|
||||
@@ -331,9 +331,7 @@ impl Normalizable for ChatCompletionRequest {
|
||||
}
|
||||
|
||||
/// Mirrors the Python vLLM `ChatCompletionResponse` class.
|
||||
///
|
||||
/// Do not skip serializing `None` fields here: non-streaming response types
|
||||
/// should serialize `None` as explicit `null`.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub(super) struct ChatCompletionResponse {
|
||||
pub id: String,
|
||||
@@ -349,6 +347,7 @@ pub(super) struct ChatCompletionResponse {
|
||||
}
|
||||
|
||||
/// Mirrors the Python vLLM `ChatCompletionResponseChoice` class.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub(super) struct ChatCompletionChoice {
|
||||
pub index: u32,
|
||||
@@ -371,12 +370,12 @@ impl fmt::Display for AssistantRole {
|
||||
}
|
||||
|
||||
/// Mirrors the Python vLLM response `ChatMessage` class.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub(super) struct ChatCompletionMessage {
|
||||
pub role: AssistantRole,
|
||||
pub content: Option<String>,
|
||||
#[serde(skip_serializing_if = "Vec::is_empty")]
|
||||
pub tool_calls: Vec<ToolCall>,
|
||||
pub tool_calls: Option<Vec<ToolCall>>,
|
||||
pub reasoning: Option<String>,
|
||||
}
|
||||
|
||||
|
||||
@@ -196,9 +196,7 @@ impl Normalizable for CompletionRequest {
|
||||
}
|
||||
|
||||
/// Mirrors the Python vLLM `CompletionResponse` class.
|
||||
///
|
||||
/// Do not skip serializing `None` fields here: non-streaming response types
|
||||
/// should serialize `None` as explicit `null`.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub(super) struct CompletionResponse {
|
||||
pub id: String,
|
||||
@@ -212,6 +210,7 @@ pub(super) struct CompletionResponse {
|
||||
}
|
||||
|
||||
/// Mirrors the Python vLLM `CompletionResponseChoice` class.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub(super) struct CompletionChoice {
|
||||
pub index: u32,
|
||||
|
||||
@@ -311,9 +311,7 @@ pub enum MessageContent {
|
||||
// ============================================================================
|
||||
|
||||
/// Mirrors the Python vLLM `UsageInfo` class.
|
||||
///
|
||||
/// Do not skip serializing `None` fields here: non-streaming response types
|
||||
/// should serialize `None` as explicit `null`.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct Usage {
|
||||
pub prompt_tokens: usize,
|
||||
@@ -404,12 +402,14 @@ pub struct LogProbs {
|
||||
}
|
||||
|
||||
/// Mirrors the Python vLLM `ChatCompletionLogProbs` class.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct ChatLogProbs {
|
||||
pub content: Option<Vec<ChatLogProbsContent>>,
|
||||
}
|
||||
|
||||
/// Mirrors the Python vLLM `ChatCompletionLogProbsContent` class.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct ChatLogProbsContent {
|
||||
pub token: String,
|
||||
@@ -419,6 +419,7 @@ pub struct ChatLogProbsContent {
|
||||
}
|
||||
|
||||
/// Mirrors the Python vLLM `ChatCompletionLogProb` class.
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct TopLogProb {
|
||||
pub token: String,
|
||||
@@ -435,6 +436,7 @@ pub struct ErrorResponse {
|
||||
pub error: ErrorDetail,
|
||||
}
|
||||
|
||||
#[serde_with::skip_serializing_none]
|
||||
#[derive(Debug, Clone, Deserialize, Serialize)]
|
||||
pub struct ErrorDetail {
|
||||
pub message: String,
|
||||
|
||||
@@ -2,11 +2,10 @@ use std::sync::Arc;
|
||||
|
||||
use axum::Json;
|
||||
use axum::extract::rejection::QueryRejection;
|
||||
use axum::extract::{Query, RawQuery, State};
|
||||
use axum::extract::{Query, State};
|
||||
use axum::http::StatusCode;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use vllm_engine_core_client::protocol::utility::PauseMode;
|
||||
use vllm_metrics::{METRICS, SleepStateLabels};
|
||||
|
||||
use crate::error::ApiError;
|
||||
use crate::state::AppState;
|
||||
@@ -25,18 +24,10 @@ pub(crate) struct SleepParams {
|
||||
mode: PauseMode,
|
||||
}
|
||||
|
||||
/// Parse repeated `tags` query parameters (`?tags=weights&tags=kv_cache`),
|
||||
/// mirroring FastAPI's `tags: list[str] | None = Query(None)`. Plain
|
||||
/// `axum::extract::Query` cannot deserialize repeated keys into a `Vec`, so
|
||||
/// the raw query string is split by hand; unknown parameters are ignored like
|
||||
/// in the Python frontend.
|
||||
fn parse_wake_up_tags(query: Option<&str>) -> Option<Vec<String>> {
|
||||
let query = query?;
|
||||
let tags: Vec<String> = form_urlencoded::parse(query.as_bytes())
|
||||
.filter(|(key, _)| key == "tags")
|
||||
.map(|(_, value)| value.into_owned())
|
||||
.collect();
|
||||
(!tags.is_empty()).then_some(tags)
|
||||
#[derive(Debug, Default, Deserialize)]
|
||||
pub(crate) struct WakeUpParams {
|
||||
#[serde(default)]
|
||||
tags: Option<Vec<String>>,
|
||||
}
|
||||
|
||||
const fn default_sleep_level() -> u32 {
|
||||
@@ -47,36 +38,6 @@ fn invalid_query(error: QueryRejection) -> ApiError {
|
||||
ApiError::invalid_request(error.body_text(), Some("mode"))
|
||||
}
|
||||
|
||||
/// Update the `vllm:engine_sleep_state` gauges, mirroring the Python
|
||||
/// `PrometheusStatLogger.record_sleep_state` semantics: any successful sleep
|
||||
/// records the level flags, and any successful wake-up (even a partial one
|
||||
/// with tags) records the engine as awake.
|
||||
pub(crate) fn record_sleep_state(state: &AppState, sleep_level: Option<u32>) {
|
||||
let (awake, weights_offloaded, discard_all) = match sleep_level {
|
||||
Some(level) => (0, u64::from(level == 1), u64::from(level == 2)),
|
||||
None => (1, 0, 0),
|
||||
};
|
||||
|
||||
let model_name = state.primary_model_name();
|
||||
for engine in state.engine_core_client().known_engine_indices() {
|
||||
for (sleep_state, value) in [
|
||||
("awake", awake),
|
||||
("weights_offloaded", weights_offloaded),
|
||||
("discard_all", discard_all),
|
||||
] {
|
||||
METRICS
|
||||
.scheduler
|
||||
.engine_sleep_state
|
||||
.get_or_create(&SleepStateLabels {
|
||||
model_name: model_name.to_string(),
|
||||
engine,
|
||||
sleep_state,
|
||||
})
|
||||
.set(value);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Put the engine to sleep.
|
||||
pub async fn sleep(
|
||||
State(state): State<Arc<AppState>>,
|
||||
@@ -90,26 +51,20 @@ pub async fn sleep(
|
||||
.await
|
||||
.map_err(|error| utility_call_error("sleep", error))?;
|
||||
|
||||
record_sleep_state(&state, Some(params.level));
|
||||
|
||||
Ok(StatusCode::OK)
|
||||
}
|
||||
|
||||
/// Wake the engine from sleep mode.
|
||||
pub async fn wake_up(
|
||||
State(state): State<Arc<AppState>>,
|
||||
RawQuery(query): RawQuery,
|
||||
Query(params): Query<WakeUpParams>,
|
||||
) -> Result<StatusCode, ApiError> {
|
||||
let tags = parse_wake_up_tags(query.as_deref());
|
||||
|
||||
state
|
||||
.engine_core_client()
|
||||
.wake_up(tags)
|
||||
.wake_up(params.tags)
|
||||
.await
|
||||
.map_err(|error| utility_call_error("wake_up", error))?;
|
||||
|
||||
record_sleep_state(&state, None);
|
||||
|
||||
Ok(StatusCode::OK)
|
||||
}
|
||||
|
||||
@@ -125,24 +80,3 @@ pub async fn is_sleeping(
|
||||
|
||||
Ok(Json(IsSleepingResponse { is_sleeping }))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::parse_wake_up_tags;
|
||||
|
||||
#[test]
|
||||
fn parse_wake_up_tags_handles_absent_single_and_repeated() {
|
||||
assert_eq!(parse_wake_up_tags(None), None);
|
||||
assert_eq!(parse_wake_up_tags(Some("")), None);
|
||||
assert_eq!(
|
||||
parse_wake_up_tags(Some("tags=weights")),
|
||||
Some(vec!["weights".to_string()])
|
||||
);
|
||||
assert_eq!(
|
||||
parse_wake_up_tags(Some("tags=weights&tags=kv_cache")),
|
||||
Some(vec!["weights".to_string(), "kv_cache".to_string()])
|
||||
);
|
||||
// Unknown parameters are ignored, like in the Python frontend.
|
||||
assert_eq!(parse_wake_up_tags(Some("other=1")), None);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -559,7 +559,7 @@ fn qwen_multimodal_model_info() -> vllm_chat::multimodal::MultimodalModelInfo {
|
||||
));
|
||||
fs::write(
|
||||
&config_path,
|
||||
r#"{"model_type":"qwen2_vl","vision_token_id":151655}"#,
|
||||
r#"{"model_type":"qwen2_vl","image_token_id":151655}"#,
|
||||
)
|
||||
.expect("write qwen test config");
|
||||
let info = vllm_chat::multimodal::MultimodalModelInfo::from_paths(
|
||||
@@ -2191,28 +2191,6 @@ async fn non_stream_chat_returns_json_response() {
|
||||
assert_eq!(json["usage"]["prompt_tokens"], 22);
|
||||
assert_eq!(json["usage"]["completion_tokens"], 3);
|
||||
assert_eq!(json["usage"]["total_tokens"], 25);
|
||||
|
||||
// Unset optional fields are serialized as explicit `null` on
|
||||
// non-streaming responses...
|
||||
let response_object = json.as_object().expect("response object");
|
||||
let choice = json["choices"][0].as_object().expect("choice object");
|
||||
let message = choice["message"].as_object().expect("message object");
|
||||
for (object, key) in [
|
||||
(response_object, "system_fingerprint"),
|
||||
(response_object, "prompt_token_ids"),
|
||||
(response_object, "kv_transfer_params"),
|
||||
(choice, "logprobs"),
|
||||
(choice, "stop_reason"),
|
||||
(choice, "token_ids"),
|
||||
(message, "reasoning"),
|
||||
] {
|
||||
assert!(
|
||||
object.contains_key(key) && object[key].is_null(),
|
||||
"expected explicit null `{key}`: {json}"
|
||||
);
|
||||
}
|
||||
// ...except `tool_calls`, which Python pops from the payload when empty.
|
||||
assert!(!message.contains_key("tool_calls"), "{json}");
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
@@ -2978,25 +2956,6 @@ async fn non_stream_completions_return_json_response() {
|
||||
assert_eq!(json["choices"][0]["text"], "hi");
|
||||
assert_eq!(json["choices"][0]["finish_reason"], "stop");
|
||||
assert_eq!(json["usage"]["completion_tokens"], 3);
|
||||
|
||||
// Unset optional fields are serialized as explicit `null` on
|
||||
// non-streaming responses.
|
||||
let response_object = json.as_object().expect("response object");
|
||||
let choice = json["choices"][0].as_object().expect("choice object");
|
||||
for (object, key) in [
|
||||
(response_object, "system_fingerprint"),
|
||||
(response_object, "kv_transfer_params"),
|
||||
(choice, "logprobs"),
|
||||
(choice, "stop_reason"),
|
||||
(choice, "prompt_logprobs"),
|
||||
(choice, "token_ids"),
|
||||
(choice, "prompt_token_ids"),
|
||||
] {
|
||||
assert!(
|
||||
object.contains_key(key) && object[key].is_null(),
|
||||
"expected explicit null `{key}`: {json}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
@@ -4412,10 +4371,10 @@ async fn include_reasoning_false_suppresses_reasoning_in_non_stream_chat() {
|
||||
let json: serde_json::Value = serde_json::from_str(&text).expect("decode json");
|
||||
|
||||
assert_eq!(json["choices"][0]["message"]["content"], "answer");
|
||||
// Suppressed fields are serialized as explicit `null` on non-streaming
|
||||
// responses.
|
||||
assert!(
|
||||
json["choices"][0]["message"]["reasoning"].is_null(),
|
||||
json["choices"][0]["message"]
|
||||
.as_object()
|
||||
.is_some_and(|message| !message.contains_key("reasoning")),
|
||||
"{text}"
|
||||
);
|
||||
}
|
||||
@@ -4517,14 +4476,14 @@ async fn include_reasoning_false_suppresses_non_stream_output_metadata() {
|
||||
let choice = json["choices"][0].as_object().expect("choice object");
|
||||
|
||||
assert_eq!(json["choices"][0]["message"]["content"], "answer");
|
||||
// Suppressed fields are serialized as explicit `null` on non-streaming
|
||||
// responses.
|
||||
assert!(
|
||||
json["choices"][0]["message"]["reasoning"].is_null(),
|
||||
json["choices"][0]["message"]
|
||||
.as_object()
|
||||
.is_some_and(|message| !message.contains_key("reasoning")),
|
||||
"{text}"
|
||||
);
|
||||
assert!(choice["logprobs"].is_null(), "{text}");
|
||||
assert!(choice["token_ids"].is_null(), "{text}");
|
||||
assert!(!choice.contains_key("logprobs"), "{text}");
|
||||
assert!(!choice.contains_key("token_ids"), "{text}");
|
||||
assert!(json["prompt_token_ids"].is_array(), "{text}");
|
||||
}
|
||||
|
||||
|
||||
@@ -102,13 +102,12 @@ pub struct DetokenizeRequest {
|
||||
pub tokens: Vec<u32>,
|
||||
}
|
||||
|
||||
/// Do not skip serializing `None` fields here: non-streaming response types
|
||||
/// should serialize `None` as explicit `null`.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct TokenizeResponse {
|
||||
pub count: usize,
|
||||
pub max_model_len: u32,
|
||||
pub tokens: Vec<u32>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub token_strs: Option<Vec<String>>,
|
||||
}
|
||||
|
||||
|
||||
@@ -10,7 +10,7 @@ pub enum TokenIdsError {
|
||||
#[error("allowed_token_ids should not be empty")]
|
||||
EmptyAllowedTokenIds,
|
||||
#[error(
|
||||
"token_id(s) {token_ids:?} in {parameter} are out of vocabulary. \
|
||||
"token_id(s) {token_ids:?} in {parameter} contain out-of-vocab token ids. \
|
||||
Vocabulary size: {vocab_size}"
|
||||
)]
|
||||
OutOfVocab {
|
||||
|
||||
@@ -21,7 +21,7 @@ def _patch_hf_api(side_effect):
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def hf_tokenizer() -> PreTrainedTokenizerBase:
|
||||
return AutoTokenizer.from_pretrained("gpt2")
|
||||
return AutoTokenizer.from_pretrained("openai-community/gpt2")
|
||||
|
||||
|
||||
_FAKE_ROWS = {
|
||||
|
||||
@@ -12,7 +12,7 @@ from vllm.benchmarks.datasets import get_samples
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def hf_tokenizer() -> PreTrainedTokenizerBase:
|
||||
return AutoTokenizer.from_pretrained("gpt2")
|
||||
return AutoTokenizer.from_pretrained("openai-community/gpt2")
|
||||
|
||||
|
||||
def _write_jsonl(path: Path, n_rows: int) -> None:
|
||||
|
||||
@@ -17,7 +17,7 @@ from vllm.benchmarks.datasets import (
|
||||
@pytest.fixture(scope="session")
|
||||
def hf_tokenizer() -> PreTrainedTokenizerBase:
|
||||
# Use a small, commonly available tokenizer
|
||||
return AutoTokenizer.from_pretrained("gpt2")
|
||||
return AutoTokenizer.from_pretrained("openai-community/gpt2")
|
||||
|
||||
|
||||
class Params(NamedTuple):
|
||||
|
||||
@@ -16,7 +16,7 @@ from vllm.benchmarks.datasets import RandomMultiModalDataset, SampleRequest
|
||||
@pytest.fixture(scope="session")
|
||||
def hf_tokenizer() -> PreTrainedTokenizerBase:
|
||||
"""Use a small, commonly available tokenizer."""
|
||||
return AutoTokenizer.from_pretrained("gpt2")
|
||||
return AutoTokenizer.from_pretrained("openai-community/gpt2")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
|
||||
@@ -13,7 +13,7 @@ from vllm.benchmarks.datasets.create_txt_slices_dataset import create_txt_slices
|
||||
@pytest.fixture(scope="session")
|
||||
def hf_tokenizer() -> PreTrainedTokenizerBase:
|
||||
# Use a small, commonly available tokenizer
|
||||
return AutoTokenizer.from_pretrained("gpt2")
|
||||
return AutoTokenizer.from_pretrained("openai-community/gpt2")
|
||||
|
||||
|
||||
text_content = """
|
||||
@@ -39,7 +39,7 @@ def test_create_txt_slices_jsonl(
|
||||
create_txt_slices_jsonl(
|
||||
input_path=str(txt_path),
|
||||
output_path=str(jsonl_path),
|
||||
tokenizer_name="gpt2",
|
||||
tokenizer_name="openai-community/gpt2",
|
||||
num_prompts=10,
|
||||
input_len=10,
|
||||
output_len=10,
|
||||
|
||||
@@ -79,6 +79,7 @@ def run_e2e_fusion_test(monkeypatch, caplog_mp_spawn):
|
||||
):
|
||||
monkeypatch.setenv("VLLM_USE_DEEP_GEMM", "1" if use_deepgemm else "0")
|
||||
monkeypatch.setenv("VLLM_ROCM_USE_AITER", "1" if use_aiter else "0")
|
||||
monkeypatch.setenv("VLLM_ROCM_USE_AITER_CUSTOM_AR", "1" if use_aiter else "0")
|
||||
from vllm._aiter_ops import rocm_aiter_ops
|
||||
|
||||
rocm_aiter_ops.refresh_env_variables()
|
||||
|
||||
@@ -13,6 +13,7 @@ from vllm._custom_ops import cutlass_scaled_fp4_mm, scaled_fp4_quant
|
||||
from vllm.compilation.passes.fusion.allreduce_rms_fusion import (
|
||||
AllReduceFusionPass,
|
||||
RocmAiterAllReduceFusionPass,
|
||||
_select_flashinfer_allreduce_use_oneshot,
|
||||
)
|
||||
from vllm.compilation.passes.fx_utils import find_op_nodes
|
||||
from vllm.compilation.passes.utility.fix_functionalization import (
|
||||
@@ -30,6 +31,9 @@ from vllm.config import (
|
||||
set_current_vllm_config,
|
||||
)
|
||||
from vllm.distributed import tensor_model_parallel_all_reduce
|
||||
from vllm.distributed.device_communicators.aiter_custom_all_reduce import (
|
||||
AiterCustomAllreduce,
|
||||
)
|
||||
from vllm.distributed.parallel_state import (
|
||||
init_distributed_environment,
|
||||
initialize_model_parallel,
|
||||
@@ -45,6 +49,35 @@ from vllm.utils.torch_utils import set_random_seed
|
||||
DEVICE_TYPE = current_platform.device_type
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("workspace_backend", "device_capability", "world_size", "tensor_size", "expected"),
|
||||
[
|
||||
("mnnvl", 103, 8, 2 * 1024 * 1024, None),
|
||||
("trtllm", 103, 8, 2 * 1024 * 1024, True),
|
||||
("trtllm", 103, 8, 2 * 1024 * 1024 + 1, False),
|
||||
("trtllm", 100, 4, 4 * 1024 * 1024, True),
|
||||
("trtllm", 100, 4, 4 * 1024 * 1024 + 1, False),
|
||||
("trtllm", None, 8, 128 * 1024 * 1024, True),
|
||||
],
|
||||
)
|
||||
def test_select_flashinfer_allreduce_use_oneshot(
|
||||
workspace_backend: str,
|
||||
device_capability: int | None,
|
||||
world_size: int,
|
||||
tensor_size: int,
|
||||
expected: bool | None,
|
||||
):
|
||||
assert (
|
||||
_select_flashinfer_allreduce_use_oneshot(
|
||||
workspace_backend,
|
||||
device_capability,
|
||||
world_size,
|
||||
tensor_size,
|
||||
)
|
||||
is expected
|
||||
)
|
||||
|
||||
|
||||
class TestAllReduceRMSNormModel(torch.nn.Module):
|
||||
def __init__(
|
||||
self,
|
||||
@@ -504,8 +537,12 @@ def all_reduce_fusion_pass_on_test_model(
|
||||
"MASTER_ADDR": "localhost",
|
||||
"MASTER_PORT": "12345",
|
||||
"VLLM_FLASHINFER_ALLREDUCE_BACKEND": flashinfer_allreduce_backend,
|
||||
"VLLM_ROCM_USE_AITER": str(int(use_aiter)),
|
||||
"VLLM_ROCM_USE_AITER_CUSTOM_AR": str(int(use_aiter)),
|
||||
}
|
||||
)
|
||||
if use_aiter:
|
||||
rocm_aiter_ops.refresh_env_variables()
|
||||
|
||||
init_distributed_environment()
|
||||
|
||||
@@ -616,7 +653,7 @@ def test_rocm_aiter_all_reduce_rmsnorm_group_quant_fp8_fusion_pass_replace(
|
||||
m.setenv("VLLM_ROCM_USE_AITER", "1")
|
||||
rocm_aiter_ops.refresh_env_variables()
|
||||
|
||||
if not rocm_aiter_ops.has_fused_allreduce_rmsnorm_quant_per_group():
|
||||
if not AiterCustomAllreduce.build_supports_per_group_quant():
|
||||
pytest.skip(
|
||||
"aiter build is missing 'fused_ar_rms_per_group_quant' (needs "
|
||||
"ROCm/aiter PR #2823); the new patterns aren't registered."
|
||||
@@ -671,6 +708,7 @@ def rocm_aiter_group_quant_fusion_pass_on_test_model(
|
||||
"MASTER_ADDR": "localhost",
|
||||
"MASTER_PORT": "12345",
|
||||
"VLLM_ROCM_USE_AITER": "1",
|
||||
"VLLM_ROCM_USE_AITER_CUSTOM_AR": "1",
|
||||
}
|
||||
)
|
||||
rocm_aiter_ops.refresh_env_variables()
|
||||
|
||||
@@ -502,7 +502,7 @@ def test_gpt2_cache_hit(monkeypatch: pytest.MonkeyPatch):
|
||||
m.setenv("VLLM_USE_AOT_COMPILE", "1")
|
||||
# First compilation - initialize model and generate
|
||||
llm_model = LLM(
|
||||
model="gpt2",
|
||||
model="openai-community/gpt2",
|
||||
compilation_config=CompilationConfig(
|
||||
mode=CompilationMode.VLLM_COMPILE,
|
||||
),
|
||||
@@ -519,7 +519,7 @@ def test_gpt2_cache_hit(monkeypatch: pytest.MonkeyPatch):
|
||||
# Second compilation - should hit cache
|
||||
m.setenv("VLLM_FORCE_AOT_LOAD", "1")
|
||||
llm_model = LLM(
|
||||
model="gpt2",
|
||||
model="openai-community/gpt2",
|
||||
compilation_config=CompilationConfig(
|
||||
mode=CompilationMode.VLLM_COMPILE,
|
||||
),
|
||||
|
||||
@@ -24,7 +24,7 @@ from vllm.utils.torch_utils import is_torch_equal_or_newer
|
||||
def get_test_models():
|
||||
"""Get list of models to test based on PyTorch version"""
|
||||
models = [
|
||||
"gpt2",
|
||||
"openai-community/gpt2",
|
||||
"Qwen/Qwen2-7B-Instruct",
|
||||
"meta-llama/Llama-3.1-8B",
|
||||
]
|
||||
|
||||
@@ -0,0 +1,52 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
|
||||
from transformers import PretrainedConfig
|
||||
|
||||
from vllm.config.speculative import MTPModelTypes, SpeculativeConfig
|
||||
from vllm.transformers_utils.model_arch_config_convertor import (
|
||||
BailingHybridMTPModelArchConfigConvertor,
|
||||
)
|
||||
|
||||
|
||||
def _bailing_config() -> PretrainedConfig:
|
||||
config = PretrainedConfig(
|
||||
architectures=["BailingMoeV2_5ForCausalLM"],
|
||||
hidden_size=4096,
|
||||
kv_lora_rank=512,
|
||||
num_attention_heads=32,
|
||||
num_experts=256,
|
||||
num_hidden_layers=32,
|
||||
num_key_value_heads=32,
|
||||
num_nextn_predict_layers=1,
|
||||
qk_rope_head_dim=64,
|
||||
vocab_size=157184,
|
||||
)
|
||||
config.model_type = "bailing_hybrid"
|
||||
return config
|
||||
|
||||
|
||||
def test_bailing_hybrid_mtp_hf_config_override():
|
||||
config = _bailing_config()
|
||||
|
||||
overridden = SpeculativeConfig.hf_config_override(config)
|
||||
|
||||
assert overridden.model_type == "bailing_hybrid_mtp"
|
||||
assert overridden.architectures == ["BailingMoeV25MTPModel"]
|
||||
assert overridden.n_predict == 1
|
||||
assert "bailing_hybrid_mtp" in MTPModelTypes.__args__
|
||||
|
||||
|
||||
def test_bailing_hybrid_mtp_model_arch_config():
|
||||
config = _bailing_config()
|
||||
config.model_type = "bailing_hybrid_mtp"
|
||||
config.architectures = ["BailingMoeV25MTPModel"]
|
||||
|
||||
model_arch_config = BailingHybridMTPModelArchConfigConvertor(
|
||||
config, config
|
||||
).convert()
|
||||
|
||||
assert model_arch_config.model_type == "bailing_hybrid_mtp"
|
||||
assert model_arch_config.architectures == ["BailingMoeV25MTPModel"]
|
||||
assert model_arch_config.total_num_hidden_layers == 1
|
||||
assert model_arch_config.is_deepseek_mla
|
||||
@@ -114,7 +114,7 @@ TEXT_GENERATION_MODELS = {
|
||||
"tiiuae/falcon-7b": PPTestSettings.fast(),
|
||||
"google/gemma-1.1-2b-it": PPTestSettings.fast(),
|
||||
"google/gemma-2-9b": PPTestSettings.fast(),
|
||||
"gpt2": PPTestSettings.fast(),
|
||||
"openai-community/gpt2": PPTestSettings.fast(),
|
||||
"EleutherAI/gpt-j-6b": PPTestSettings.fast(),
|
||||
"EleutherAI/pythia-1.4b": PPTestSettings.fast(),
|
||||
"ibm/PowerLM-3b": PPTestSettings.fast(),
|
||||
|
||||
@@ -0,0 +1,134 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
|
||||
import pytest
|
||||
import ray
|
||||
import torch
|
||||
import torch.distributed as dist
|
||||
|
||||
from vllm._aiter_ops import is_aiter_found, rocm_aiter_ops
|
||||
from vllm.distributed.communication_op import tensor_model_parallel_all_reduce # noqa
|
||||
from vllm.distributed.parallel_state import get_tp_group, graph_capture
|
||||
from vllm.envs import disable_envs_cache
|
||||
from vllm.platforms import current_platform
|
||||
|
||||
from ..utils import (
|
||||
assert_rocm_custom_allreduce_backend_state,
|
||||
ensure_model_parallel_initialized,
|
||||
init_test_distributed_environment,
|
||||
multi_gpu_test,
|
||||
multi_process_parallel,
|
||||
)
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
not current_platform.is_rocm(),
|
||||
reason="ROCm-only AITER custom allreduce tests",
|
||||
)
|
||||
|
||||
test_cases = [
|
||||
((2, 7168), torch.float16),
|
||||
((2, 7168), torch.bfloat16),
|
||||
((128, 8192), torch.float16),
|
||||
((128, 8192), torch.bfloat16),
|
||||
]
|
||||
|
||||
|
||||
def _configure_aiter_custom_ar_env(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False)
|
||||
monkeypatch.delenv("HIP_VISIBLE_DEVICES", raising=False)
|
||||
monkeypatch.setenv("VLLM_ROCM_USE_AITER", "1")
|
||||
monkeypatch.setenv("VLLM_ROCM_USE_AITER_CUSTOM_AR", "1")
|
||||
monkeypatch.setenv("VLLM_ROCM_QUICK_REDUCE_QUANTIZATION", "NONE")
|
||||
disable_envs_cache()
|
||||
rocm_aiter_ops.refresh_env_variables()
|
||||
|
||||
|
||||
def _assert_aiter_handles_input(inp: torch.Tensor) -> None:
|
||||
aiter_ar_comm = get_tp_group().device_communicator.aiter_ar_comm
|
||||
assert aiter_ar_comm is not None
|
||||
assert aiter_ar_comm.should_custom_ar(inp), (
|
||||
f"AITER CustomAllreduce does not support input shape {inp.shape}."
|
||||
)
|
||||
|
||||
|
||||
@ray.remote(num_gpus=1, max_calls=1)
|
||||
def graph_allreduce(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
tp_size,
|
||||
pp_size,
|
||||
rank,
|
||||
distributed_init_port,
|
||||
) -> None:
|
||||
with monkeypatch.context() as m:
|
||||
_configure_aiter_custom_ar_env(m)
|
||||
|
||||
device = torch.device(f"cuda:{rank}")
|
||||
torch.accelerator.set_device_index(device)
|
||||
init_test_distributed_environment(tp_size, pp_size, rank, distributed_init_port)
|
||||
ensure_model_parallel_initialized(tp_size, pp_size)
|
||||
assert_rocm_custom_allreduce_backend_state(True, "NONE")
|
||||
group = get_tp_group().device_group
|
||||
|
||||
# A small all_reduce for warmup.
|
||||
# this is needed because device communicators might be created lazily
|
||||
# (e.g. NCCL). This will ensure that the communicator is initialized
|
||||
# before any communication happens, so that this group can be used for
|
||||
# graph capture immediately.
|
||||
data = torch.zeros(1)
|
||||
data = data.to(device=device)
|
||||
dist.all_reduce(data, group=group)
|
||||
torch.accelerator.synchronize()
|
||||
del data
|
||||
|
||||
for shape, dtype in test_cases:
|
||||
with graph_capture(device=device) as graph_capture_context:
|
||||
inp = torch.ones(shape, dtype=dtype, device=device)
|
||||
_assert_aiter_handles_input(inp)
|
||||
expected = inp * tp_size
|
||||
|
||||
torch.accelerator.synchronize()
|
||||
graph = torch.cuda.CUDAGraph()
|
||||
with torch.cuda.graph(graph, stream=graph_capture_context.stream):
|
||||
out = tensor_model_parallel_all_reduce(inp)
|
||||
|
||||
graph.replay()
|
||||
torch.testing.assert_close(out, expected)
|
||||
|
||||
|
||||
@ray.remote(num_gpus=1, max_calls=1)
|
||||
def eager_allreduce(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
tp_size,
|
||||
pp_size,
|
||||
rank,
|
||||
distributed_init_port,
|
||||
) -> None:
|
||||
with monkeypatch.context() as m:
|
||||
_configure_aiter_custom_ar_env(m)
|
||||
|
||||
device = torch.device(f"cuda:{rank}")
|
||||
torch.accelerator.set_device_index(device)
|
||||
init_test_distributed_environment(tp_size, pp_size, rank, distributed_init_port)
|
||||
ensure_model_parallel_initialized(tp_size, pp_size)
|
||||
assert_rocm_custom_allreduce_backend_state(True, "NONE")
|
||||
|
||||
for shape, dtype in test_cases:
|
||||
inp = torch.ones(shape, dtype=dtype, device=device)
|
||||
_assert_aiter_handles_input(inp)
|
||||
expected = inp * tp_size
|
||||
out = tensor_model_parallel_all_reduce(inp)
|
||||
torch.testing.assert_close(out, expected)
|
||||
|
||||
|
||||
@pytest.mark.skipif(not is_aiter_found(), reason="AITER is not installed")
|
||||
@multi_gpu_test(num_gpus=2)
|
||||
@pytest.mark.parametrize("tp_size", [2])
|
||||
@pytest.mark.parametrize("pipeline_parallel_size", [1])
|
||||
@pytest.mark.parametrize("test_target", [eager_allreduce, graph_allreduce])
|
||||
def test_rocm_aiter_custom_allreduce(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
tp_size,
|
||||
pipeline_parallel_size,
|
||||
test_target,
|
||||
):
|
||||
multi_process_parallel(monkeypatch, tp_size, pipeline_parallel_size, test_target)
|
||||
@@ -78,7 +78,6 @@ def _create_mock_engine():
|
||||
mock_engine.model_config = MockModelConfig()
|
||||
mock_engine.input_processor = MagicMock()
|
||||
|
||||
# renderer is accessed by OpenAIServing.__init__ and serving.py
|
||||
mock_renderer = MagicMock()
|
||||
mock_renderer.tokenizer = get_tokenizer(MODEL_NAME)
|
||||
mock_engine.renderer = mock_renderer
|
||||
|
||||
@@ -111,3 +111,79 @@ async def test_batched_chat_completions_with_json_schema(
|
||||
parsed = json.loads(choice["message"]["content"])
|
||||
assert "answer" in parsed
|
||||
assert parsed["answer"] in ("yes", "no")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
"model_name",
|
||||
[MODEL_NAME],
|
||||
)
|
||||
async def test_batched_chat_completions_logprobs_not_token_id_placeholders(
|
||||
server: RemoteOpenAIServer, model_name: str
|
||||
) -> None:
|
||||
# Regression test: requesting `return_token_ids` alongside logprobs must not
|
||||
# corrupt the logprob `token` fields into "token_id:{id}" placeholders. That
|
||||
# placeholder rendering is controlled by `return_tokens_as_token_ids`, which
|
||||
# this request leaves unset.
|
||||
conversations = [
|
||||
[{"role": "user", "content": "Reply with exactly the word: alpha"}],
|
||||
]
|
||||
|
||||
async with httpx.AsyncClient() as http_client:
|
||||
response = await http_client.post(
|
||||
f"{server.url_for('v1/chat/completions/batch')}",
|
||||
json={
|
||||
"model": model_name,
|
||||
"messages": conversations,
|
||||
"logprobs": True,
|
||||
"top_logprobs": 1,
|
||||
"return_token_ids": True,
|
||||
},
|
||||
timeout=60,
|
||||
)
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
data = response.json()
|
||||
|
||||
content = data["choices"][0]["logprobs"]["content"]
|
||||
assert content
|
||||
for entry in content:
|
||||
assert not entry["token"].startswith("token_id:")
|
||||
for top in entry["top_logprobs"]:
|
||||
assert not top["token"].startswith("token_id:")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
"model_name",
|
||||
[MODEL_NAME],
|
||||
)
|
||||
async def test_batched_chat_completions_return_tokens_as_token_ids(
|
||||
server: RemoteOpenAIServer, model_name: str
|
||||
) -> None:
|
||||
# Complementary check: when `return_tokens_as_token_ids` is explicitly set,
|
||||
# the logprob tokens *should* be rendered as "token_id:{id}" placeholders,
|
||||
# proving the new field is actually wired through.
|
||||
conversations = [
|
||||
[{"role": "user", "content": "Reply with exactly the word: alpha"}],
|
||||
]
|
||||
|
||||
async with httpx.AsyncClient() as http_client:
|
||||
response = await http_client.post(
|
||||
f"{server.url_for('v1/chat/completions/batch')}",
|
||||
json={
|
||||
"model": model_name,
|
||||
"messages": conversations,
|
||||
"logprobs": True,
|
||||
"top_logprobs": 1,
|
||||
"return_tokens_as_token_ids": True,
|
||||
},
|
||||
timeout=60,
|
||||
)
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
data = response.json()
|
||||
|
||||
content = data["choices"][0]["logprobs"]["content"]
|
||||
assert content
|
||||
assert all(entry["token"].startswith("token_id:") for entry in content)
|
||||
|
||||
@@ -18,7 +18,7 @@ from vllm.renderers.embed_utils import safe_load_prompt_embeds
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_empty_prompt():
|
||||
model_name = "gpt2"
|
||||
model_name = "openai-community/gpt2"
|
||||
server_args = ["--enforce-eager"]
|
||||
with RemoteOpenAIServer(model_name, server_args) as remote_server:
|
||||
client = remote_server.get_async_client()
|
||||
@@ -38,7 +38,7 @@ async def test_empty_prompt():
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_out_of_vocab_token_ids():
|
||||
model_name = "gpt2"
|
||||
model_name = "openai-community/gpt2"
|
||||
server_args = ["--enforce-eager"]
|
||||
with RemoteOpenAIServer(model_name, server_args) as remote_server:
|
||||
client = remote_server.get_async_client()
|
||||
|
||||
@@ -71,8 +71,9 @@ def test_lm_eval_accuracy_v1_engine():
|
||||
|
||||
more_args = []
|
||||
|
||||
# Limit compilation time for V1
|
||||
if current_platform.is_tpu():
|
||||
# Limit compilation time for V1 on TPU
|
||||
# Avoid OOM on XPU
|
||||
if current_platform.is_tpu() or current_platform.is_xpu():
|
||||
more_args = ["--max-num-seqs", "64"]
|
||||
|
||||
run_test(more_args)
|
||||
|
||||
@@ -7,20 +7,20 @@ from unittest.mock import MagicMock
|
||||
import pytest
|
||||
|
||||
import vllm.envs as envs
|
||||
from vllm.entrypoints.openai.engine.serving import GenerationError, OpenAIServing
|
||||
from vllm.entrypoints.generate.base.serving import GenerateBaseServing, GenerationError
|
||||
from vllm.envs import disable_envs_cache
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_raise_if_error_raises_generation_error():
|
||||
"""test _raise_if_error raises GenerationError"""
|
||||
# create a minimal OpenAIServing instance
|
||||
# create a minimal GenerateBaseServing instance
|
||||
mock_engine = MagicMock()
|
||||
mock_engine.model_config = MagicMock()
|
||||
mock_engine.model_config.max_model_len = 100
|
||||
mock_models = MagicMock()
|
||||
|
||||
serving = OpenAIServing(
|
||||
serving = GenerateBaseServing(
|
||||
engine_client=mock_engine,
|
||||
models=mock_models,
|
||||
request_logger=None,
|
||||
@@ -47,7 +47,7 @@ async def test_convert_generation_error_to_streaming_response():
|
||||
mock_engine.model_config.max_model_len = 100
|
||||
mock_models = MagicMock()
|
||||
|
||||
serving = OpenAIServing(
|
||||
serving = GenerateBaseServing(
|
||||
engine_client=mock_engine,
|
||||
models=mock_models,
|
||||
request_logger=None,
|
||||
@@ -77,7 +77,7 @@ def test_is_model_supported_skip_name_validation_env(
|
||||
mock_models = MagicMock()
|
||||
mock_models.is_base_model.return_value = False
|
||||
|
||||
serving = OpenAIServing(
|
||||
serving = GenerateBaseServing(
|
||||
engine_client=mock_engine,
|
||||
models=mock_models,
|
||||
request_logger=None,
|
||||
|
||||
@@ -0,0 +1,114 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
|
||||
import json
|
||||
|
||||
import openai # use the official client for correctness check
|
||||
import pytest
|
||||
|
||||
MODEL_NAME = "Qwen/Qwen3-1.7B"
|
||||
NAMESPACE = "mcp__computer_use"
|
||||
TOOL_NAME = "get_app_state"
|
||||
FLAT_TOOL_NAME = f"{NAMESPACE}__{TOOL_NAME}"
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "namespace",
|
||||
"name": NAMESPACE,
|
||||
"description": "Computer control tools.",
|
||||
"tools": [
|
||||
{
|
||||
"type": "function",
|
||||
"name": TOOL_NAME,
|
||||
"description": "Get the current state of a desktop application.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"app": {
|
||||
"type": "string",
|
||||
"description": "Application name, for example Chrome.",
|
||||
}
|
||||
},
|
||||
"required": ["app"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
}
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
prompt = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Use the computer app state tool to inspect Google Chrome.",
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def _assert_namespace_tool_call(tool_call) -> None:
|
||||
assert tool_call.type == "function_call"
|
||||
assert tool_call.name == TOOL_NAME
|
||||
assert tool_call.namespace == NAMESPACE
|
||||
assert tool_call.name != FLAT_TOOL_NAME
|
||||
|
||||
args = json.loads(tool_call.arguments)
|
||||
assert args["app"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("model_name", [MODEL_NAME])
|
||||
async def test_namespace_tool_separator(client: openai.AsyncOpenAI, model_name: str):
|
||||
response = await client.responses.create(
|
||||
model=model_name,
|
||||
input=prompt,
|
||||
tools=tools,
|
||||
temperature=0.0,
|
||||
)
|
||||
|
||||
assert len(response.output) >= 1
|
||||
tool_call = next(
|
||||
(out for out in response.output if out.type == "function_call"), None
|
||||
)
|
||||
assert tool_call is not None
|
||||
_assert_namespace_tool_call(tool_call)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("model_name", [MODEL_NAME])
|
||||
async def test_namespace_tool_separator_streaming(
|
||||
client: openai.AsyncOpenAI, model_name: str
|
||||
):
|
||||
stream = await client.responses.create(
|
||||
model=model_name,
|
||||
input=prompt,
|
||||
tools=tools,
|
||||
temperature=0.0,
|
||||
stream=True,
|
||||
)
|
||||
events = [event async for event in stream]
|
||||
|
||||
added_call = next(
|
||||
(
|
||||
event.item
|
||||
for event in events
|
||||
if event.type == "response.output_item.added"
|
||||
and getattr(event.item, "type", None) == "function_call"
|
||||
),
|
||||
None,
|
||||
)
|
||||
done_call = next(
|
||||
(
|
||||
event.item
|
||||
for event in events
|
||||
if event.type == "response.output_item.done"
|
||||
and getattr(event.item, "type", None) == "function_call"
|
||||
),
|
||||
None,
|
||||
)
|
||||
|
||||
assert added_call is not None
|
||||
assert added_call.name == TOOL_NAME
|
||||
assert added_call.namespace == NAMESPACE
|
||||
|
||||
assert done_call is not None
|
||||
_assert_namespace_tool_call(done_call)
|
||||
@@ -127,13 +127,13 @@ def test_responses_api_logprobs_with_return_tokens_as_token_ids():
|
||||
"""Test that return_tokens_as_token_ids works in Responses API logprobs."""
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from vllm.entrypoints.openai.engine.serving import OpenAIServing
|
||||
from vllm.entrypoints.generate.base.serving import GenerateBaseServing
|
||||
from vllm.entrypoints.openai.responses.serving import OpenAIServingResponses
|
||||
from vllm.logprobs import Logprob as SampleLogprob
|
||||
|
||||
serving = MagicMock(spec=OpenAIServingResponses)
|
||||
serving.return_tokens_as_token_ids = True
|
||||
serving._get_decoded_token = OpenAIServing._get_decoded_token
|
||||
serving._get_decoded_token = GenerateBaseServing._get_decoded_token
|
||||
|
||||
tokenizer = MagicMock()
|
||||
tokenizer.decode = lambda token_id: "decoded"
|
||||
|
||||
@@ -40,5 +40,4 @@ async def test_show_version(server: RemoteOpenAIServer):
|
||||
response = client.get(server.url_for("version"))
|
||||
response.raise_for_status()
|
||||
|
||||
# Tolerate additive fields (e.g. the Rust frontend reports its own version).
|
||||
assert response.json()["version"] == VLLM_VERSION
|
||||
assert response.json() == {"version": VLLM_VERSION}
|
||||
|
||||
@@ -83,8 +83,7 @@ async def test_show_version(server: RemoteOpenAIServer):
|
||||
response = requests.get(server.url_for("version"))
|
||||
response.raise_for_status()
|
||||
|
||||
# Tolerate additive fields (e.g. the Rust frontend reports its own version).
|
||||
assert response.json()["version"] == VLLM_VERSION
|
||||
assert response.json() == {"version": VLLM_VERSION}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
||||
@@ -14,7 +14,7 @@ from vllm.entrypoints.openai.engine.protocol import (
|
||||
)
|
||||
from vllm.entrypoints.openai.models.protocol import BaseModelPath
|
||||
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
|
||||
from vllm.entrypoints.pooling.base.serving import PoolingServingBase
|
||||
from vllm.entrypoints.pooling.base.serving import PoolingBaseServing
|
||||
from vllm.entrypoints.pooling.typing import PoolingServeContext
|
||||
from vllm.entrypoints.serve.lora.protocol import (
|
||||
LoadLoRAAdapterRequest,
|
||||
@@ -136,7 +136,7 @@ async def test_unload_lora_adapter_not_found():
|
||||
assert response.error.code == HTTPStatus.NOT_FOUND
|
||||
|
||||
|
||||
class _ConcretePoolingServing(PoolingServingBase):
|
||||
class _ConcretePoolingServing(PoolingBaseServing):
|
||||
"""Minimal concrete subclass used only in these unit tests."""
|
||||
|
||||
request_id_prefix = "test"
|
||||
@@ -178,7 +178,7 @@ def test_pooling_maybe_get_adapters_lora_name_sets_lora_request():
|
||||
serving = _make_pooling_serving(lora_name)
|
||||
ctx = _make_pooling_ctx(lora_name)
|
||||
|
||||
serving._maybe_get_adapters(ctx)
|
||||
ctx.lora_request = serving._maybe_get_adapters(ctx.request)
|
||||
|
||||
assert ctx.lora_request is not None
|
||||
assert ctx.lora_request.lora_name == lora_name
|
||||
@@ -190,4 +190,4 @@ def test_pooling_maybe_get_adapters_unknown_model_raises():
|
||||
ctx = _make_pooling_ctx("unknown-model")
|
||||
|
||||
with pytest.raises(VLLMNotFoundError):
|
||||
serving._maybe_get_adapters(ctx)
|
||||
serving._maybe_get_adapters(ctx.request)
|
||||
|
||||
@@ -7,7 +7,7 @@ from unittest.mock import AsyncMock, Mock
|
||||
|
||||
import pytest
|
||||
|
||||
from vllm.entrypoints.speech_to_text.base.serving import OpenAISpeechToText
|
||||
from vllm.entrypoints.speech_to_text.base.serving import SpeechToTextBaseServing
|
||||
from vllm.entrypoints.speech_to_text.transcription.protocol import TranscriptionResponse
|
||||
|
||||
|
||||
@@ -43,7 +43,7 @@ async def test_non_streaming_cancel_aborts_engine_requests(
|
||||
is_tracing_enabled=AsyncMock(return_value=False),
|
||||
)
|
||||
|
||||
server = OpenAISpeechToText.__new__(OpenAISpeechToText)
|
||||
server = SpeechToTextBaseServing.__new__(SpeechToTextBaseServing)
|
||||
server.engine_client = engine_client
|
||||
server.task_type = "transcribe"
|
||||
server.models = SimpleNamespace(model_name=lambda: "audio")
|
||||
@@ -112,7 +112,7 @@ async def test_non_streaming_cancel_advances_all_chunk_generators():
|
||||
{"prompt": "chunk-1"},
|
||||
{"prompt": "chunk-2"},
|
||||
]
|
||||
server = OpenAISpeechToText.__new__(OpenAISpeechToText)
|
||||
server = SpeechToTextBaseServing.__new__(SpeechToTextBaseServing)
|
||||
server.engine_client = engine_client
|
||||
server.task_type = "transcribe"
|
||||
server.models = SimpleNamespace(model_name=lambda: "audio")
|
||||
@@ -170,7 +170,7 @@ async def test_language_detection_cancel_aborts_engine_request():
|
||||
abort=AsyncMock(),
|
||||
)
|
||||
|
||||
server = OpenAISpeechToText.__new__(OpenAISpeechToText)
|
||||
server = SpeechToTextBaseServing.__new__(SpeechToTextBaseServing)
|
||||
server.engine_client = engine_client
|
||||
server.asr_config = SimpleNamespace()
|
||||
server.tokenizer = Mock()
|
||||
|
||||
+4
-2
@@ -25,7 +25,7 @@ from vllm.entrypoints.openai.engine.protocol import (
|
||||
)
|
||||
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
|
||||
from vllm.entrypoints.speech_to_text.base.serving import (
|
||||
OpenAISpeechToText,
|
||||
SpeechToTextBaseServing,
|
||||
asr_inter_chunk_separator,
|
||||
)
|
||||
from vllm.entrypoints.speech_to_text.transcription.protocol import TranscriptionRequest
|
||||
@@ -234,7 +234,9 @@ async def test_create_transcription_non_streaming_joins_chunks_by_language():
|
||||
"vllm.model_executor.model_loader.get_model_cls",
|
||||
return_value=_StubTranscriptionModel,
|
||||
),
|
||||
patch.object(OpenAISpeechToText, "_preprocess_speech_to_text", preprocess_mock),
|
||||
patch.object(
|
||||
SpeechToTextBaseServing, "_preprocess_speech_to_text", preprocess_mock
|
||||
),
|
||||
):
|
||||
serving = OpenAIServingTranscription(engine_client, models, request_logger=None)
|
||||
|
||||
|
||||
+2
-1
@@ -68,7 +68,8 @@ def server(request):
|
||||
if request.param is not None:
|
||||
args += ["--attention-backend", request.param]
|
||||
if "AITER" in request.param:
|
||||
env_dict = _AITER_ENV
|
||||
# TODO: re-enable once AITER reenables fp16 unified attention.
|
||||
pytest.skip("ROCM_AITER_UNIFIED_ATTN does not support fp16")
|
||||
with RemoteOpenAIServer(MODEL_NAME, args, env_dict=env_dict) as remote_server:
|
||||
yield remote_server
|
||||
|
||||
|
||||
@@ -74,7 +74,7 @@ class FakeHarmonyParser(HarmonyParser):
|
||||
self.reasoning_parser = None
|
||||
self.tool_parser = None
|
||||
self._chunk_results: list[ChunkResult] = []
|
||||
self._flush_results: list[Segment | None] = []
|
||||
self._flush_results: list[list[Segment]] = []
|
||||
self.processed_chunks: list[list[int]] = []
|
||||
|
||||
def enqueue_chunk_result(
|
||||
@@ -89,7 +89,7 @@ class FakeHarmonyParser(HarmonyParser):
|
||||
)
|
||||
)
|
||||
|
||||
def enqueue_flush_result(self, segment: Segment | None) -> None:
|
||||
def enqueue_flush_result(self, segment: list[Segment]) -> None:
|
||||
self._flush_results.append(segment)
|
||||
|
||||
def process_chunk(self, token_ids) -> ChunkResult:
|
||||
@@ -98,10 +98,10 @@ class FakeHarmonyParser(HarmonyParser):
|
||||
return self._chunk_results.pop(0)
|
||||
return ChunkResult(segments=[], reasoning_token_count=0)
|
||||
|
||||
def flush(self) -> Segment | None:
|
||||
def flush(self) -> list[Segment]:
|
||||
if self._flush_results:
|
||||
return self._flush_results.pop(0)
|
||||
return None
|
||||
return []
|
||||
|
||||
|
||||
def make_harmony_context(
|
||||
@@ -598,13 +598,21 @@ async def test_streaming_message_synchronization():
|
||||
content=[TextContent(text=response_text)],
|
||||
recipient=Role.USER,
|
||||
)
|
||||
flush_segment = Segment(
|
||||
channel="commentary",
|
||||
recipient=None,
|
||||
delta="",
|
||||
completed_message=message,
|
||||
)
|
||||
parser.enqueue_flush_result(flush_segment)
|
||||
flush_segments = [
|
||||
Segment(
|
||||
channel="final",
|
||||
recipient=None,
|
||||
delta=response_text,
|
||||
completed_message=None,
|
||||
),
|
||||
Segment(
|
||||
channel="final",
|
||||
recipient=None,
|
||||
delta="",
|
||||
completed_message=message,
|
||||
),
|
||||
]
|
||||
parser.enqueue_flush_result(flush_segments)
|
||||
|
||||
# Create another output to trigger synchronization via flush()
|
||||
context.append_output(
|
||||
@@ -618,8 +626,9 @@ async def test_streaming_message_synchronization():
|
||||
assert context.num_init_messages == 1
|
||||
assert context._messages[2].content[0].text == response_text
|
||||
assert context.last_append_flush_status is True
|
||||
assert len(context.last_append_segments) == 1
|
||||
assert context.last_append_segments[0].completed_message is message
|
||||
assert len(context.last_append_segments) == 2
|
||||
assert context.last_append_segments[-2].delta == response_text
|
||||
assert context.last_append_segments[-1].completed_message is message
|
||||
|
||||
|
||||
def test_turn_metrics_copy_and_reset():
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user