forked from Karylab-cklius/vllm
@@ -20,6 +20,6 @@ nvidia-cudnn-frontend>=1.13.0,<1.19.0
|
||||
# Required for faster safetensors model loading
|
||||
fastsafetensors >= 0.2.2
|
||||
|
||||
# QuACK and Cutlass DSL for FA4 (cute-DSL implementation)
|
||||
nvidia-cutlass-dsl>=4.4.2
|
||||
# For dequantize_and_gather_k_cache_cutedsl
|
||||
nvidia-cutlass-dsl>=4.5.0
|
||||
quack-kernels>=0.3.3
|
||||
|
||||
Reference in New Issue
Block a user