forked from Karylab-cklius/vllm
Compare commits
627
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
381b691620 | ||
|
|
d6247d7173 | ||
|
|
a0c092ee72 | ||
|
|
43eaefba5a | ||
|
|
e0cfa52d22 | ||
|
|
242c591d5a | ||
|
|
625871b52c | ||
|
|
72297d859b | ||
|
|
f51193b9ae | ||
|
|
aeaa50a71c | ||
|
|
542a8fad6d | ||
|
|
9a4e5f9539 | ||
|
|
c44e191b01 | ||
|
|
5b14019576 | ||
|
|
dad7a6383b | ||
|
|
5b29c958c7 | ||
|
|
df2735ea2e | ||
|
|
32e657e689 | ||
|
|
ad5d29db70 | ||
|
|
f5a7cce9b6 | ||
|
|
6370e53f24 | ||
|
|
65a1a16594 | ||
|
|
100d655a23 | ||
|
|
7de49bab7e | ||
|
+13 |
7c6729b769 | ||
|
|
6f00a1ae3b | ||
|
|
6f91edf96d | ||
|
|
db7a79cbf7 | ||
|
|
dc1be79031 | ||
|
|
0bb548b60e | ||
|
|
58f9659397 | ||
|
|
f37f03db4a | ||
|
|
32a423ac0a | ||
|
|
30c2718eaa | ||
|
|
7398a30d79 | ||
|
|
17a74b745b | ||
|
|
54ab69b14e | ||
|
|
7f4c52f2ba | ||
|
|
6fbbcf2151 | ||
|
|
56f31af62a | ||
|
|
fe65aa6a97 | ||
|
|
176256b962 | ||
|
|
5369f7b7b8 | ||
|
|
e7f6a39db8 | ||
|
|
a07fac758f | ||
|
|
bb3b61f2fd | ||
|
|
2899dca843 | ||
|
|
0b6aa3c47c | ||
|
|
d552a68645 | ||
|
|
118bcde449 | ||
|
|
6c7e679f04 | ||
|
|
1db989bbf1 | ||
|
|
8a7b3c2990 | ||
|
|
05a0814863 | ||
|
|
4f56321d7e | ||
|
|
01661cc57f | ||
|
|
ba702e978e | ||
|
|
6453fc0b8c | ||
|
|
4fb483ca86 | ||
|
|
30217b0e80 | ||
|
|
0d0504b54c | ||
|
|
1e81853afc | ||
|
|
b6cbba8bc8 | ||
|
|
94100b5915 | ||
|
|
62d8db7c05 | ||
|
|
601fa9a74e | ||
|
|
9b9fc4039c | ||
|
|
98e91a9600 | ||
|
|
948107acf7 | ||
|
|
35efdf6b34 | ||
|
|
d2bfc6fe20 | ||
|
|
912d6b619d | ||
|
|
bf9f23003c | ||
|
|
25ace8fe5d | ||
|
|
247470f23a | ||
|
|
88402a41c4 | ||
|
|
5ed3faa43d | ||
|
|
61ac368021 | ||
|
|
99b57a4823 | ||
|
|
b09688a6e7 | ||
|
|
03a2d03367 | ||
|
|
9069a57139 | ||
|
|
90245f4190 | ||
|
|
f472ab0a4c | ||
|
|
74587939b1 | ||
|
|
d223c900d8 | ||
|
|
52c3c4a42f | ||
|
|
a8f296083f | ||
|
|
fbb1ef6803 | ||
|
|
33fe71a4d3 | ||
|
|
d18ed2304a | ||
|
|
60915c972c | ||
|
|
7aea73d83d | ||
|
|
73af7a362a | ||
|
|
e68bfc2828 | ||
|
|
02b6ecf07c | ||
|
|
1206891822 | ||
|
|
60b3d39cd3 | ||
|
|
60417b4b74 | ||
|
|
272abd5f48 | ||
|
|
1e34a13539 | ||
|
|
ebcef33766 | ||
|
|
28158b2fc3 | ||
|
|
53f6dd5c6f | ||
|
|
99115fcdcd | ||
|
|
1053e248f0 | ||
|
|
b5bcb3ce88 | ||
|
|
b2f9e4caa4 | ||
|
|
831d3848f1 | ||
|
|
fd10e8946d | ||
|
|
ed13deb376 | ||
|
|
99de48e98f | ||
|
|
bf2b45b5d6 | ||
|
|
8112b6c997 | ||
|
|
15d65f8669 | ||
|
|
e3c2fc3b3c | ||
|
|
2b465b2c42 | ||
|
|
3f47a8384d | ||
|
|
04502deca2 | ||
|
|
d2ca3002d9 | ||
|
|
27d7061ef6 | ||
|
|
ef9975d021 | ||
|
|
56c96b0d91 | ||
|
|
59a6b0411d | ||
|
|
dbccc5ae32 | ||
|
|
a89015c6df | ||
|
|
96fa3f42c9 | ||
|
|
81962bb699 | ||
|
|
92e8518d37 | ||
|
|
77cba0259f | ||
|
|
30fbd05537 | ||
|
|
0906123953 | ||
|
|
bc3629b1c4 | ||
|
|
312ea82e75 | ||
|
|
394beb633b | ||
|
|
7f599d7854 | ||
|
|
eb290ab673 | ||
|
|
cbc3a87200 | ||
|
|
afc94523c9 | ||
|
|
8061dc26bd | ||
|
|
fd9d2ede6f | ||
|
|
e09900436c | ||
|
|
5d07e268b1 | ||
|
|
d742856610 | ||
|
|
544cb724c8 | ||
|
|
c314af1abf | ||
|
|
f19ee27e39 | ||
|
|
5f89a03dcb | ||
|
|
53397fbfac | ||
|
|
49f31d7cee | ||
|
|
74d3b799e1 | ||
|
|
29fdeab254 | ||
|
|
ff6173997d | ||
|
|
8de50e46d4 | ||
|
|
da99ffcc13 | ||
|
|
8040ef2426 | ||
|
|
bf4f633b4c | ||
|
|
854c33f380 | ||
|
|
ac87549cbd | ||
|
|
439f336212 | ||
|
|
ffc4f08c8e | ||
|
|
50aa830482 | ||
|
|
f0553889c0 | ||
|
|
fdaa0d9e59 | ||
|
|
0934b26790 | ||
|
|
9e50e1037e | ||
|
|
b5b61c622c | ||
|
|
b68d7ef262 | ||
|
|
7154856f3d | ||
|
|
3f1d40960f | ||
|
|
0da6e7f3d6 | ||
|
|
5559679229 | ||
|
|
da3a252fd1 | ||
|
|
21fd9e85a0 | ||
|
|
8d28b48d01 | ||
|
|
0164022c90 | ||
|
|
30b0714031 | ||
|
|
2e860de498 | ||
|
|
7eca0e1a64 | ||
|
|
7a29a3c54c | ||
|
|
1240c74c0a | ||
|
|
48ebd6f2f1 | ||
|
|
b153ae6089 | ||
|
|
7a6a5b3667 | ||
|
|
0111002323 | ||
|
|
d30b1ecd1b | ||
|
|
dbd80cc031 | ||
|
|
70009fb934 | ||
|
|
ee1d996367 | ||
|
|
6b0103d1c9 | ||
|
|
9321aff536 | ||
|
|
26d725c334 | ||
|
|
7fe6d3c76b | ||
|
|
2e0da24150 | ||
|
+1 |
0b0bd2b5f6 | ||
|
|
33ef67e9fb | ||
|
|
3e74c60b9c | ||
|
|
d1a8ba63d9 | ||
|
|
1423569ff5 | ||
|
|
9a50464698 | ||
|
|
b9b6306ebe | ||
|
|
ca0defa343 | ||
|
|
0b1a8bb1f6 | ||
|
|
fe5145765f | ||
|
|
dbcc1cdd0a | ||
|
|
a82f1b388f | ||
|
|
190be7dad2 | ||
|
|
94682b79f4 | ||
|
|
0ba2aa35a8 | ||
|
|
aaaeda98dc | ||
|
|
d9cd774198 | ||
|
|
70052fb924 | ||
|
|
318b527cc2 | ||
|
|
6a1acac3fe | ||
|
|
213f681f81 | ||
|
|
33c4f3551c | ||
|
|
caa9cad31e | ||
|
|
7513d071bd | ||
|
|
89f6aa3a9e | ||
|
|
84d26b9ee3 | ||
|
|
9e6746b3c7 | ||
|
|
972848f276 | ||
|
|
2279575cd9 | ||
|
|
5d8e90a966 | ||
|
|
e222c33f2f | ||
|
|
c064fa52b6 | ||
|
|
7e51939e25 | ||
|
|
9863102ed9 | ||
|
|
8c13ee5735 | ||
|
|
41798069f3 | ||
|
|
866fea2b99 | ||
|
|
d02df748bf | ||
|
|
453f01783d | ||
|
|
7b40fb9645 | ||
|
|
8eac21a602 | ||
|
|
a454a1dd25 | ||
|
|
833483f357 | ||
|
|
163ecba377 | ||
|
|
589a5b884b | ||
|
|
5c5434e2d8 | ||
|
|
dd72658e7d | ||
|
|
0d77325b10 | ||
|
|
2ac125123a | ||
|
|
7bdf8cc37c | ||
|
|
bf27e34ebb | ||
|
|
275556c35c | ||
|
|
d65acd83d8 | ||
|
|
80c9d5d5e0 | ||
|
|
da54a5bf05 | ||
|
|
1479bd9e9d | ||
|
|
0231dd5467 | ||
|
|
2659467497 | ||
|
|
a49d37c6b9 | ||
|
|
4501a6d56b | ||
|
|
e18f0037a5 | ||
|
|
b354734d17 | ||
|
|
b91a40e729 | ||
|
|
75ccdf3145 | ||
|
|
c6fe94b4d5 | ||
|
|
46f01a50ac | ||
|
|
f00efc5265 | ||
|
|
b0cb1da1bd | ||
|
|
0e36e3bbd1 | ||
|
|
494845e79f | ||
|
|
0416dab275 | ||
|
|
c8db00b16c | ||
|
|
80c7683923 | ||
|
|
638d6e9757 | ||
|
|
1ad84fea86 | ||
|
|
12213c6795 | ||
|
|
10c75477b0 | ||
|
|
ac36a7a1e7 | ||
|
|
521aa80f71 | ||
|
|
a76df87db8 | ||
|
|
a4904ba903 | ||
|
|
f83de6d44c | ||
|
|
239fc73553 | ||
|
|
76bf55240c | ||
|
|
9a698f3255 | ||
|
|
4080263bb2 | ||
|
|
fc5fda105f | ||
|
|
b07ec92faa | ||
|
|
27ffbfde8d | ||
|
|
229e01e9e1 | ||
|
|
191146dba5 | ||
|
|
f3a920a076 | ||
|
|
149daf0d72 | ||
|
|
917fdb5bf7 | ||
|
|
4b594b4aa1 | ||
|
|
7d10a4cfce | ||
|
|
910cc8543a | ||
|
|
431934522b | ||
|
|
61a09532f2 | ||
|
|
3de4b2bf3c | ||
|
|
b44311b6ef | ||
|
|
b0d7875180 | ||
|
|
53c2f20dd9 | ||
|
|
37e370fe93 | ||
|
|
2dc5a72e7e | ||
|
|
c79ff5f918 | ||
|
|
1a659a0c37 | ||
|
|
0f6cf7f628 | ||
|
|
c79ad3ae21 | ||
|
|
61c9ef986a | ||
|
|
d6dbdb9b0d | ||
|
|
06da482fb4 | ||
|
|
2f75e7f712 | ||
|
|
7c21548ce3 | ||
|
|
9df2f91232 | ||
|
|
387189c429 | ||
|
|
75576c63be | ||
|
|
16aca639b7 | ||
|
|
6049424b7e | ||
|
|
060b5f61dc | ||
|
|
ec59c1579f | ||
|
|
1750e443f2 | ||
|
|
ba18929079 | ||
|
|
0500ca6a58 | ||
|
|
a1c15bcb0f | ||
|
|
4809de7317 | ||
|
|
05781e21dd | ||
|
|
85f638a2b8 | ||
|
|
08e5067561 | ||
|
|
a7d00ec051 | ||
|
|
b8fb56d970 | ||
|
|
96a739289e | ||
|
|
60d443f738 | ||
|
|
1dca300653 | ||
|
|
fca252d59e | ||
|
|
33178f9006 | ||
|
|
b2b8f679d0 | ||
|
|
de6ec294ef | ||
|
|
61e10f0116 | ||
|
|
6e96891ba0 | ||
|
|
47f1b47a73 | ||
|
|
5aab491bc9 | ||
|
|
5812e1a66b | ||
|
|
8950394e0a | ||
|
|
7bb49be4d1 | ||
|
|
c67650f04b | ||
|
|
f890e1dbe2 | ||
|
|
040cbf95cc | ||
|
|
5b3762a7f0 | ||
|
|
4d30c510ce | ||
|
|
6700813f86 | ||
|
|
eb44b3aaa4 | ||
|
|
7a98c7a392 | ||
|
|
0d9e60619b | ||
|
|
1134545b6f | ||
|
|
3e0c887511 | ||
|
|
adfbbc1005 | ||
|
|
adc98f04d0 | ||
|
|
8def3cdde2 | ||
|
|
616c9bd0f4 | ||
|
|
8688a06d67 | ||
|
|
f25953cc59 | ||
|
|
d9aa35161d | ||
|
|
6bcda970fd | ||
|
|
ea0e9c8f2e | ||
|
|
94ed0bf4e0 | ||
|
|
1940c8441e | ||
|
|
72d16aee15 | ||
|
|
e78a0c8e59 | ||
|
|
97a98006b0 | ||
|
|
0a684ab0c0 | ||
|
|
0d9210a502 | ||
|
|
1d874867ea | ||
|
|
2e2e626b40 | ||
|
|
af91f4b3e4 | ||
|
|
2396a61108 | ||
|
|
97a668152b | ||
|
|
58b2012aa2 | ||
|
|
b7c20d0cfa | ||
|
|
a2b1f9fc3b | ||
|
|
642076d26c | ||
|
|
5feb3950e5 | ||
|
|
4ec199b66a | ||
|
|
7ca017778f | ||
|
|
fbfe58133d | ||
|
|
9dd62d80ab | ||
|
|
f878367898 | ||
|
|
bd091079cb | ||
|
|
b23bd73f54 | ||
|
|
e2d7adeb64 | ||
|
|
15cb8e140d | ||
|
|
f007cceb42 | ||
|
|
0a5069e4e3 | ||
|
|
8ce53a616e | ||
|
|
ae10e855ab | ||
|
|
530ee36a0d | ||
|
|
d835ad572c | ||
|
|
47d0597ca2 | ||
|
|
818cf61e91 | ||
|
|
c01618fdc8 | ||
|
|
823eaf667d | ||
|
|
f1f1259692 | ||
|
|
df13b5aef5 | ||
|
|
4938d44a3b | ||
|
|
37bf988c2f | ||
|
|
9459fc6471 | ||
|
|
5245c80564 | ||
|
|
9bc266d923 | ||
|
|
5c9f6557d7 | ||
|
|
dcfebf93f4 | ||
|
|
752bd10647 | ||
|
|
2730b657c4 | ||
|
|
1dcbbd9cac | ||
|
|
ace9fda495 | ||
|
|
ef0aa7ca2f | ||
|
|
e6d1310b2a | ||
|
|
ac5f38a0f7 | ||
|
|
b6ff8a2f50 | ||
|
|
9243e0124e | ||
|
|
df362b2d6d | ||
|
|
7c2acd38b7 | ||
|
|
a287eb163f | ||
|
|
e94243893d | ||
|
|
29c0ec4d63 | ||
|
|
c7ce03bcbd | ||
|
|
c233d90aa8 | ||
|
|
d96aee0951 | ||
|
|
c71a583aa9 | ||
|
|
f12b80c6ef | ||
|
|
da64db78b9 | ||
|
|
425c4eafb0 | ||
|
|
02c01f442b | ||
|
|
fae543015c | ||
|
|
c9be3a8aa1 | ||
|
|
41ea2dd44a | ||
|
|
088c0be268 | ||
|
|
fcd2255d16 | ||
|
|
b5433b6f50 | ||
|
|
cc25f028b7 | ||
|
|
c4cd2bd544 | ||
|
|
5784507da4 | ||
|
|
bf578e1abd | ||
|
|
efed8a1e83 | ||
|
|
11d291511a | ||
|
|
877dae9c68 | ||
|
|
c4dd6d78fd | ||
|
|
ce2aecc4dc | ||
|
|
f38f3d11fb | ||
|
|
d4b4562917 | ||
|
|
7b3192523e | ||
|
|
4c6e2e4b30 | ||
|
|
8502958810 | ||
|
|
ce4bdcbda4 | ||
|
|
d5b1ec2684 | ||
|
|
867ff69733 | ||
|
|
109b736b86 | ||
|
|
69d4f5ef63 | ||
|
|
426d48bfa1 | ||
|
|
26c909ed74 | ||
|
|
fb1d8ccaf5 | ||
|
|
9354f22204 | ||
|
|
17fdd42100 | ||
|
|
472d330c21 | ||
|
|
3b6c96a101 | ||
|
|
4d4e04f452 | ||
|
|
67fe73b2b4 | ||
|
+1 |
ee8f36d0b3 | ||
|
+1 |
f3e9497e92 | ||
|
|
fe784ff22e | ||
|
|
b88abb5036 | ||
|
|
67f9046e4a | ||
|
|
f17be06fbe | ||
|
|
2cab53ddee | ||
|
|
ab0a20d151 | ||
|
|
4a394bfcda | ||
|
|
c95c663049 | ||
|
|
ab3c1aedf3 | ||
|
+1 |
fb5ec0dc9e | ||
|
|
971dac2caa | ||
|
|
efa2e424f6 | ||
|
|
02bf9c7907 | ||
|
|
f61163e6c7 | ||
|
|
626c90b2d5 | ||
|
+1 |
251f7e478e | ||
|
|
ce65385618 | ||
|
|
7d56fe2adc | ||
|
|
75bdad40b5 | ||
|
|
d08eebad16 | ||
|
|
3e90d015ba | ||
|
|
7cd1d57b74 | ||
|
|
b8168e33e0 | ||
|
|
d803b44dbe | ||
|
|
530852f959 | ||
|
|
a317bc5739 | ||
|
|
a9531edfa6 | ||
|
|
8c3393f373 | ||
|
|
9f8cbfd8eb | ||
|
|
ea1d65fe6d | ||
|
|
f44f3d6f79 | ||
|
|
cc706b05a5 | ||
|
|
85e296950c | ||
|
|
dc9f845ddc | ||
|
|
12f2c515a7 | ||
|
+1 |
6570c9800c | ||
|
|
8bfd683901 | ||
|
|
7dc2698632 | ||
|
|
59b964f37d | ||
|
|
6a9f24aa8c | ||
|
|
ba47bb5be1 | ||
|
|
df8a0900df | ||
|
|
2db39c7049 | ||
|
|
3935829f89 | ||
|
|
7746961277 | ||
|
|
5de1add806 | ||
|
|
915dffaa5f | ||
|
|
81e13a0591 | ||
|
|
f95e3f0edb | ||
|
|
5a65ba5f17 | ||
|
|
9d1c695be5 | ||
|
|
3c1bc1fc0d | ||
|
|
3a5e88e629 | ||
|
|
0becb7486b | ||
|
|
2dab187f75 | ||
|
|
015b0320de | ||
|
|
4238b011a7 | ||
|
|
eb33ff34dd | ||
|
|
2bd8957627 | ||
|
|
3034c8d389 | ||
|
|
ecf4aa5ce2 | ||
|
|
49e777cf08 | ||
|
|
b7950e798f | ||
|
|
de100ffb62 | ||
|
|
43cd340247 | ||
|
|
1d99f0f421 | ||
|
|
0885b51981 | ||
|
|
6036bf110a | ||
|
|
2fa63e0fff | ||
|
|
61141ed265 | ||
|
|
05eed72aec | ||
|
|
5810e884f1 | ||
|
|
615834ee58 | ||
|
|
5811ed6a05 | ||
|
|
1b30ae4ca4 | ||
|
|
4e04bcbce6 | ||
|
|
66b6c684ab | ||
|
|
c0302d9497 | ||
|
|
313fae3e89 | ||
|
|
7aab6e2684 | ||
|
|
9dd2e72828 | ||
|
|
d119beb1b9 | ||
|
|
12a8057bfe | ||
|
|
e281ac663a | ||
|
|
adce068118 | ||
|
|
b6770d7b54 | ||
|
|
3b39fd284a | ||
|
|
6472131298 | ||
|
|
37aa52821d | ||
|
|
96d2ceda4b | ||
|
|
fdf2cf66d3 | ||
|
|
9b2be4e9a5 | ||
|
|
3ad85e0de4 | ||
|
|
4f7fffb92f | ||
|
|
6e073440b1 | ||
|
|
f7aadae5e5 | ||
|
|
442c421e79 | ||
|
|
0bd6b85a1f | ||
|
|
3ca242d1b6 | ||
|
|
7e950521b3 | ||
|
|
0f0f28b537 | ||
|
|
520a20ba4e | ||
|
|
9182e86971 | ||
|
|
313d01f507 | ||
|
|
05d4f8bba3 | ||
|
|
0b54201a04 | ||
|
|
32e632dfeb | ||
|
|
7ffb98e248 | ||
|
|
cdaa40d2a8 | ||
|
|
ca3618bc69 | ||
|
|
b2f7d2560a | ||
|
|
1ff9429655 | ||
|
|
af453e5647 | ||
|
|
32aef44388 | ||
|
|
7a74a9662b | ||
|
|
b6754f536e | ||
|
|
793cf79c89 | ||
|
|
50ac1c7bab | ||
|
|
f04d3f640e | ||
|
|
0a9396a25e | ||
|
|
038ec293b1 | ||
|
|
894ebb27f5 | ||
|
|
c9a788eedc | ||
|
|
0762f2afeb | ||
|
|
31be872f55 | ||
|
|
94c0ef3001 | ||
|
|
af1f036a70 | ||
|
|
95aab66e95 | ||
|
|
dcf4072da9 | ||
|
|
382bbd5144 | ||
|
|
b50ef9c6ed | ||
|
|
9e289c553c | ||
|
|
c4f5cd60da | ||
|
|
0b0ef8d7eb | ||
|
|
21472f32ea | ||
|
|
fec64fea75 | ||
|
|
8b8af2caf7 | ||
|
|
7738ef35b8 | ||
|
|
9a21f0d1a3 | ||
|
|
8ac8375270 | ||
|
|
7dc447dda7 | ||
|
|
7fc97042c3 | ||
|
|
18c4067a54 | ||
|
|
550218b136 | ||
|
|
5c342876a6 | ||
|
|
9427c45386 | ||
|
|
43c8cbf79b | ||
|
|
62286308c9 | ||
|
|
26587f9519 | ||
|
|
93e3bc8f30 | ||
|
|
c2c9f7c5e2 | ||
|
|
1be6e937b2 | ||
|
|
b3cfca996c | ||
|
|
487dfb3418 | ||
|
|
107a03ba63 | ||
|
|
56a357ed33 | ||
|
|
bea70c7cfc | ||
|
|
75fe92a316 | ||
|
|
b7b58d1eba | ||
|
|
36484e464a | ||
|
|
9e57de7197 | ||
|
|
8c5dafcd09 | ||
|
|
05fa8183a6 | ||
|
|
d973cce3ca | ||
|
|
775c1589ea |
@@ -0,0 +1,103 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
|
||||||
|
"""Audit vLLM compiled libraries for PyTorch stable ABI compliance."""
|
||||||
|
|
||||||
|
import fnmatch
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from torch_abi_audit import inspect_package
|
||||||
|
from torch_abi_audit.report import ExtensionReport, PackageReport
|
||||||
|
|
||||||
|
# Temporary allowlist of extensions not yet on the stable ABI.
|
||||||
|
# Shrink and remove over time.
|
||||||
|
ALLOWED_UNSTABLE_LIBRARIES: tuple[str, ...] = (
|
||||||
|
"_flashkda_C.abi3.so",
|
||||||
|
"vllm_flash_attn/_vllm_fa2_C.abi3.so",
|
||||||
|
"vllm_flash_attn/_vllm_fa3_C.abi3.so",
|
||||||
|
"third_party/deep_gemm/_C*.so",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _relative_path(lib: ExtensionReport, package_root: Path) -> str:
|
||||||
|
try:
|
||||||
|
return lib.path.relative_to(package_root).as_posix()
|
||||||
|
except ValueError:
|
||||||
|
return lib.path.name
|
||||||
|
|
||||||
|
|
||||||
|
def _is_torch_unstable(lib: ExtensionReport) -> bool:
|
||||||
|
return lib.error is None and lib.torch.uses_torch and not lib.torch.stable
|
||||||
|
|
||||||
|
|
||||||
|
def _matches_allowlist(rel_path: str, patterns: tuple[str, ...]) -> bool:
|
||||||
|
return any(fnmatch.fnmatch(rel_path, pattern) for pattern in patterns)
|
||||||
|
|
||||||
|
|
||||||
|
def _iter_libs(report: PackageReport) -> tuple[ExtensionReport, ...]:
|
||||||
|
return (*report.extensions, *report.bundled_libs)
|
||||||
|
|
||||||
|
|
||||||
|
def _collect_unstable(report: PackageReport) -> list[str]:
|
||||||
|
return sorted(
|
||||||
|
_relative_path(lib, report.root)
|
||||||
|
for lib in _iter_libs(report)
|
||||||
|
if _is_torch_unstable(lib)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _find_stale_allowlist_entries(
|
||||||
|
report: PackageReport, patterns: tuple[str, ...]
|
||||||
|
) -> list[str]:
|
||||||
|
"""Allowlist patterns that match a built library which is no longer unstable."""
|
||||||
|
stale: list[str] = []
|
||||||
|
for pattern in patterns:
|
||||||
|
for lib in _iter_libs(report):
|
||||||
|
if lib.error is not None:
|
||||||
|
continue
|
||||||
|
if not fnmatch.fnmatch(_relative_path(lib, report.root), pattern):
|
||||||
|
continue
|
||||||
|
if not _is_torch_unstable(lib):
|
||||||
|
stale.append(pattern)
|
||||||
|
break
|
||||||
|
return stale
|
||||||
|
|
||||||
|
|
||||||
|
def check_torch_abi(
|
||||||
|
package: str = "vllm",
|
||||||
|
patterns: tuple[str, ...] = ALLOWED_UNSTABLE_LIBRARIES,
|
||||||
|
) -> int:
|
||||||
|
report = inspect_package(package)
|
||||||
|
if report.error:
|
||||||
|
print(f"error: failed to inspect {package!r}: {report.error}", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
unstable = _collect_unstable(report)
|
||||||
|
unexpected = [
|
||||||
|
rel_path for rel_path in unstable if not _matches_allowlist(rel_path, patterns)
|
||||||
|
]
|
||||||
|
stale = _find_stale_allowlist_entries(report, patterns)
|
||||||
|
|
||||||
|
if unexpected or stale:
|
||||||
|
if unexpected:
|
||||||
|
print(
|
||||||
|
"Not allowed: torch-unstable libraries outside "
|
||||||
|
f"ALLOWED_UNSTABLE_LIBRARIES: {', '.join(unexpected)}",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
if stale:
|
||||||
|
print(
|
||||||
|
"Not allowed: stale ALLOWED_UNSTABLE_LIBRARIES entries: "
|
||||||
|
f"{', '.join(stale)}",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
return 1
|
||||||
|
|
||||||
|
print("Torch stable ABI check passed.")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
print(">>> Auditing vLLM extension modules for PyTorch stable ABI compliance")
|
||||||
|
sys.exit(check_torch_abi())
|
||||||
@@ -14,6 +14,7 @@ run_all_patterns:
|
|||||||
- "setup.py"
|
- "setup.py"
|
||||||
- "csrc/"
|
- "csrc/"
|
||||||
- "cmake/"
|
- "cmake/"
|
||||||
|
- ".buildkite/check-torch-abi.py"
|
||||||
run_all_exclude_patterns:
|
run_all_exclude_patterns:
|
||||||
- "docker/Dockerfile."
|
- "docker/Dockerfile."
|
||||||
- "csrc/cpu/"
|
- "csrc/cpu/"
|
||||||
|
|||||||
@@ -18,6 +18,8 @@ steps:
|
|||||||
TERM: "xterm-256color"
|
TERM: "xterm-256color"
|
||||||
retry:
|
retry:
|
||||||
automatic:
|
automatic:
|
||||||
|
- exit_status: 1 # Transient Docker/BuildKit failure
|
||||||
|
limit: 1
|
||||||
- exit_status: -1 # Agent was lost
|
- exit_status: -1 # Agent was lost
|
||||||
limit: 1
|
limit: 1
|
||||||
- exit_status: -10 # Agent was lost
|
- exit_status: -10 # Agent was lost
|
||||||
@@ -46,6 +48,8 @@ steps:
|
|||||||
VLLM_BRANCH: "$BUILDKITE_COMMIT"
|
VLLM_BRANCH: "$BUILDKITE_COMMIT"
|
||||||
retry:
|
retry:
|
||||||
automatic:
|
automatic:
|
||||||
|
- exit_status: 1 # Transient Docker/BuildKit failure
|
||||||
|
limit: 1
|
||||||
- exit_status: -1 # Agent was lost
|
- exit_status: -1 # Agent was lost
|
||||||
limit: 1
|
limit: 1
|
||||||
- exit_status: -10 # Agent was lost
|
- exit_status: -10 # Agent was lost
|
||||||
@@ -72,6 +76,8 @@ steps:
|
|||||||
VLLM_BRANCH: "$BUILDKITE_COMMIT"
|
VLLM_BRANCH: "$BUILDKITE_COMMIT"
|
||||||
retry:
|
retry:
|
||||||
automatic:
|
automatic:
|
||||||
|
- exit_status: 1 # Transient Docker/BuildKit failure
|
||||||
|
limit: 1
|
||||||
- exit_status: -1 # Agent was lost
|
- exit_status: -1 # Agent was lost
|
||||||
limit: 1
|
limit: 1
|
||||||
- exit_status: -10 # Agent was lost
|
- exit_status: -10 # Agent was lost
|
||||||
|
|||||||
@@ -18,6 +18,8 @@ steps:
|
|||||||
- tests/kernels/quantization/test_cpu_fp8_scaled_mm.py
|
- tests/kernels/quantization/test_cpu_fp8_scaled_mm.py
|
||||||
- tests/kernels/mamba/cpu/test_cpu_gdn_ops.py
|
- tests/kernels/mamba/cpu/test_cpu_gdn_ops.py
|
||||||
- tests/kernels/mamba/test_cpu_short_conv.py
|
- tests/kernels/mamba/test_cpu_short_conv.py
|
||||||
|
- tests/kernels/mamba/test_causal_conv1d.py
|
||||||
|
- tests/kernels/mamba/test_mamba_ssm.py
|
||||||
commands:
|
commands:
|
||||||
- |
|
- |
|
||||||
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 30m "
|
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 30m "
|
||||||
@@ -28,7 +30,9 @@ steps:
|
|||||||
pytest -x -v -s tests/kernels/test_onednn.py
|
pytest -x -v -s tests/kernels/test_onednn.py
|
||||||
pytest -x -v -s tests/kernels/test_awq_int4_to_int8.py
|
pytest -x -v -s tests/kernels/test_awq_int4_to_int8.py
|
||||||
pytest -x -v -s tests/kernels/quantization/test_cpu_fp8_scaled_mm.py
|
pytest -x -v -s tests/kernels/quantization/test_cpu_fp8_scaled_mm.py
|
||||||
pytest -x -v -s tests/kernels/mamba/cpu/test_cpu_gdn_ops.py"
|
pytest -x -v -s tests/kernels/mamba/cpu/test_cpu_gdn_ops.py
|
||||||
|
pytest -x -v -s tests/kernels/mamba/test_causal_conv1d.py
|
||||||
|
pytest -x -v -s tests/kernels/mamba/test_mamba_ssm.py"
|
||||||
|
|
||||||
# Note: SDE can't be downloaded from CI host because of AWS WAF
|
# Note: SDE can't be downloaded from CI host because of AWS WAF
|
||||||
# - label: CPU-Compatibility Tests
|
# - label: CPU-Compatibility Tests
|
||||||
|
|||||||
@@ -18,7 +18,7 @@ steps:
|
|||||||
- label: "XPU example Test"
|
- label: "XPU example Test"
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 50
|
||||||
optional: true
|
optional: true
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
@@ -39,7 +39,7 @@ steps:
|
|||||||
- label: "XPU V1 test"
|
- label: "XPU V1 test"
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 70
|
||||||
optional: true
|
optional: true
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
@@ -60,7 +60,7 @@ steps:
|
|||||||
- label: "XPU server test"
|
- label: "XPU server test"
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
optional: true
|
optional: true
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ depends_on:
|
|||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
steps:
|
steps:
|
||||||
- label: XPU Sleep Mode
|
- label: XPU Sleep Mode
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
|
|||||||
@@ -0,0 +1,26 @@
|
|||||||
|
group: Benchmarks
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
steps:
|
||||||
|
- label: Benchmarks CLI Test
|
||||||
|
key: benchmarks-cli-test
|
||||||
|
timeout_in_minutes: 40
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/
|
||||||
|
- tests/benchmarks/
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'cd tests &&
|
||||||
|
pytest -v -s benchmarks/'
|
||||||
@@ -2,6 +2,44 @@ group: Engine Intel
|
|||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
steps:
|
steps:
|
||||||
|
- label: Engine
|
||||||
|
key: engine
|
||||||
|
timeout_in_minutes: 40
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/compilation/
|
||||||
|
- vllm/config/
|
||||||
|
- vllm/engine/
|
||||||
|
- vllm/entrypoints/logger.py
|
||||||
|
- vllm/envs.py
|
||||||
|
- vllm/logger.py
|
||||||
|
- vllm/logging_utils/
|
||||||
|
- vllm/platforms/
|
||||||
|
- vllm/sequence.py
|
||||||
|
- vllm/triton_utils/
|
||||||
|
- vllm/utils/
|
||||||
|
- tests/engine
|
||||||
|
- tests/test_sequence
|
||||||
|
- tests/test_config
|
||||||
|
- tests/test_logger
|
||||||
|
- tests/test_vllm_port
|
||||||
|
- tests/test_jit_monitor.py
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'cd tests &&
|
||||||
|
pytest -v -s engine/test_arg_utils.py test_sequence.py test_logger.py test_vllm_port.py test_jit_monitor.py'
|
||||||
|
|
||||||
- label: Engine (1 GPU)
|
- label: Engine (1 GPU)
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
@@ -23,3 +61,41 @@ steps:
|
|||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'cd tests &&
|
'cd tests &&
|
||||||
pytest -v -s v1/engine --ignore v1/engine/test_preprocess_error_handling.py'
|
pytest -v -s v1/engine --ignore v1/engine/test_preprocess_error_handling.py'
|
||||||
|
|
||||||
|
- label: V1 e2e (2 GPUs)
|
||||||
|
timeout_in_minutes: 30
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 2+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/compilation/
|
||||||
|
- vllm/config/
|
||||||
|
- vllm/distributed/
|
||||||
|
- vllm/engine/
|
||||||
|
- vllm/envs.py
|
||||||
|
- vllm/forward_context.py
|
||||||
|
- vllm/inputs/
|
||||||
|
- vllm/logger.py
|
||||||
|
- vllm/logging_utils/
|
||||||
|
- vllm/model_executor/
|
||||||
|
- vllm/multimodal/
|
||||||
|
- vllm/platforms/
|
||||||
|
- vllm/sampling_params.py
|
||||||
|
- vllm/transformers_utils/
|
||||||
|
- vllm/triton_utils/
|
||||||
|
- vllm/utils/
|
||||||
|
- vllm/v1/
|
||||||
|
- tests/v1/e2e/spec_decode
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'cd tests &&
|
||||||
|
pytest -v -s v1/e2e/spec_decode/test_spec_decode.py -k "tensor_parallelism"'
|
||||||
|
|||||||
@@ -86,7 +86,7 @@ steps:
|
|||||||
pytest -v -s lora/test_punica_ops.py::test_add_lora_fused_moe_early_exit'
|
pytest -v -s lora/test_punica_ops.py::test_add_lora_fused_moe_early_exit'
|
||||||
|
|
||||||
- label: LoRA Punica FP8/XPU Ops
|
- label: LoRA Punica FP8/XPU Ops
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 60
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ depends_on:
|
|||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
steps:
|
steps:
|
||||||
- label: V1 Core + KV + Metrics
|
- label: V1 Core + KV + Metrics
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
@@ -33,7 +33,7 @@ steps:
|
|||||||
pytest -v -s v1/executor'
|
pytest -v -s v1/executor'
|
||||||
|
|
||||||
- label: V1 Sample + Logits
|
- label: V1 Sample + Logits
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 90
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
@@ -125,13 +125,13 @@ steps:
|
|||||||
pytest -v -s v1/kv_offload &&
|
pytest -v -s v1/kv_offload &&
|
||||||
pytest -v -s v1/kv_connector/unit/test_offloading_connector.py'
|
pytest -v -s v1/kv_connector/unit/test_offloading_connector.py'
|
||||||
|
|
||||||
- label: NixlConnector PD accuracy (2 GPUs)
|
- label: NixlConnector PD accuracy (4 GPUs)
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 60
|
||||||
num_devices: 2
|
num_devices: 4
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
gpu: 2+
|
gpu: 4+
|
||||||
mem: 16+
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
@@ -148,11 +148,14 @@ steps:
|
|||||||
- >-
|
- >-
|
||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'cd tests &&
|
'cd tests &&
|
||||||
bash v1/kv_connector/nixl_integration/run_xpu_disagg_accuracy_test.sh'
|
bash v1/kv_connector/nixl_integration/run_xpu_disagg_accuracy_test.sh &&
|
||||||
|
PREFILLER_TP_SIZE=2 DECODER_TP_SIZE=1 bash v1/kv_connector/nixl_integration/run_xpu_disagg_accuracy_test.sh &&
|
||||||
|
PREFILLER_TP_SIZE=1 DECODER_TP_SIZE=2 bash v1/kv_connector/nixl_integration/run_xpu_disagg_accuracy_test.sh &&
|
||||||
|
PREFILLER_TP_SIZE=2 DECODER_TP_SIZE=2 bash v1/kv_connector/nixl_integration/run_xpu_disagg_accuracy_test.sh'
|
||||||
|
|
||||||
- label: Regression
|
- label: Regression
|
||||||
key: regression
|
key: regression
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 50
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
@@ -186,7 +189,7 @@ steps:
|
|||||||
|
|
||||||
- label: Metrics, Tracing (2 GPUs)
|
- label: Metrics, Tracing (2 GPUs)
|
||||||
key: metrics-tracing-2-gpus
|
key: metrics-tracing-2-gpus
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
@@ -222,7 +225,7 @@ steps:
|
|||||||
|
|
||||||
- label: Async Engine, Inputs, Utils, Worker
|
- label: Async Engine, Inputs, Utils, Worker
|
||||||
key: async-engine-inputs-utils-worker
|
key: async-engine-inputs-utils-worker
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 55
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
@@ -259,3 +262,25 @@ steps:
|
|||||||
pytest -v -s detokenizer &&
|
pytest -v -s detokenizer &&
|
||||||
pytest -v -s -m "not cpu_test" ./multimodal &&
|
pytest -v -s -m "not cpu_test" ./multimodal &&
|
||||||
pytest -v -s utils_ --ignore=utils_/test_mem_utils.py'
|
pytest -v -s utils_ --ignore=utils_/test_mem_utils.py'
|
||||||
|
|
||||||
|
- label: Fusion Unit Tests
|
||||||
|
timeout_in_minutes: 30
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/compilation/
|
||||||
|
- tests/compile/passes/test_qk_norm_rope_fusion.py
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'cd tests &&
|
||||||
|
pytest -v -s compile/passes/test_qk_norm_rope_fusion.py'
|
||||||
@@ -0,0 +1,33 @@
|
|||||||
|
group: Model Executor Intel
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
steps:
|
||||||
|
- label: Model Executor (Intel)
|
||||||
|
key: model-executor-intel
|
||||||
|
timeout_in_minutes: 45
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/engine/arg_utils.py
|
||||||
|
- vllm/config/model.py
|
||||||
|
- vllm/model_executor
|
||||||
|
- tests/model_executor
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'apt-get update && apt-get install -y curl libsodium23 &&
|
||||||
|
pip3 install tensorizer==2.10.1 &&
|
||||||
|
pip3 install runai-model-streamer[s3,gcs,azure]\>=0.15.7 &&
|
||||||
|
export VLLM_WORKER_MULTIPROC_METHOD=spawn &&
|
||||||
|
export PYTHONFAULTHANDLER=1 &&
|
||||||
|
cd tests &&
|
||||||
|
pytest -v -s model_executor -m "not slow_test" --ignore="model_executor/layers/test_rocm_unquantized_gemm.py" --deselect="tests/model_executor/model_loader/test_reload.py::test_kv_scale_reload"'
|
||||||
@@ -8,7 +8,7 @@ steps:
|
|||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
gpu: 2+
|
gpu: 2+
|
||||||
mem: 16+
|
mem: 24+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -28,7 +28,9 @@ steps:
|
|||||||
'export VLLM_USE_V2_MODEL_RUNNER=1 &&
|
'export VLLM_USE_V2_MODEL_RUNNER=1 &&
|
||||||
cd tests &&
|
cd tests &&
|
||||||
pytest -v -s v1/engine/test_llm_engine.py -k "not test_engine_metrics" &&
|
pytest -v -s v1/engine/test_llm_engine.py -k "not test_engine_metrics" &&
|
||||||
|
pytest -v -s v1/e2e/general/test_context_length.py &&
|
||||||
ENFORCE_EAGER=1 pytest -v -s v1/e2e/general/test_async_scheduling.py -k "not ngram" &&
|
ENFORCE_EAGER=1 pytest -v -s v1/e2e/general/test_async_scheduling.py -k "not ngram" &&
|
||||||
|
pytest -v -s entrypoints/llm/test_struct_output_generate.py -k "xgrammar and not speculative_config6 and not speculative_config7 and not speculative_config8 and not speculative_config0" &&
|
||||||
pytest -v -s v1/e2e/general/test_min_tokens.py'
|
pytest -v -s v1/e2e/general/test_min_tokens.py'
|
||||||
|
|
||||||
- label: Model Runner V2 Examples (Intel)
|
- label: Model Runner V2 Examples (Intel)
|
||||||
@@ -60,3 +62,55 @@ steps:
|
|||||||
python3 basic/offline_inference/generate.py --model facebook/opt-125m &&
|
python3 basic/offline_inference/generate.py --model facebook/opt-125m &&
|
||||||
python3 generate/multimodal/vision_language_offline.py --seed 0 &&
|
python3 generate/multimodal/vision_language_offline.py --seed 0 &&
|
||||||
python3 features/automatic_prefix_caching/prefix_caching_offline.py'
|
python3 features/automatic_prefix_caching/prefix_caching_offline.py'
|
||||||
|
|
||||||
|
- label: Model Runner V2 Distributed (2 GPUs)
|
||||||
|
timeout_in_minutes: 50
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 2+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/v1/worker/gpu/
|
||||||
|
- vllm/v1/worker/gpu_worker.py
|
||||||
|
- tests/basic_correctness/test_basic_correctness.py
|
||||||
|
- tests/v1/distributed/test_async_llm_dp.py
|
||||||
|
- tests/v1/distributed/test_eagle_dp.py
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'export VLLM_USE_V2_MODEL_RUNNER=1 &&
|
||||||
|
cd tests &&
|
||||||
|
TARGET_TEST_SUITE=L4 pytest -v -s basic_correctness/test_basic_correctness.py -m "distributed\(num_gpus=2\)" -k "not ray and not True"'
|
||||||
|
|
||||||
|
- label: Model Runner V2 Spec Decode
|
||||||
|
timeout_in_minutes: 50
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/v1/worker/gpu/
|
||||||
|
- vllm/v1/worker/gpu_worker.py
|
||||||
|
- tests/v1/spec_decode/test_max_len.py
|
||||||
|
- tests/v1/spec_decode/test_rejection_sampler_utils.py
|
||||||
|
- tests/v1/e2e/spec_decode/test_spec_decode.py
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'export VLLM_USE_V2_MODEL_RUNNER=1 &&
|
||||||
|
cd tests &&
|
||||||
|
pytest -v -s v1/spec_decode/test_synthetic_rejection_sampler_utils.py'
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Distributed Model Tests (2 GPUs)
|
- label: Distributed Model Tests (2 GPUs)
|
||||||
key: distributed-model-tests-2-gpus
|
key: distributed-model-tests-2-gpus
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 65
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: "Multi-Modal Models (Standard) 1: qwen2"
|
- label: "Multi-Modal Models (Standard) 1: qwen2"
|
||||||
key: multi-modal-models-standard-1-qwen2
|
key: multi-modal-models-standard-1-qwen2
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 70
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
@@ -29,7 +29,7 @@ steps:
|
|||||||
|
|
||||||
- label: "Multi-Modal Models (Standard) 2: qwen3 + gemma"
|
- label: "Multi-Modal Models (Standard) 2: qwen3 + gemma"
|
||||||
key: multi-modal-models-standard-2-qwen3-gemma
|
key: multi-modal-models-standard-2-qwen3-gemma
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 70
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
@@ -52,7 +52,7 @@ steps:
|
|||||||
|
|
||||||
- label: "Multi-Modal Models (Standard) 3: llava + qwen2_vl"
|
- label: "Multi-Modal Models (Standard) 3: llava + qwen2_vl"
|
||||||
key: multi-modal-models-standard-3-llava-qwen2-vl
|
key: multi-modal-models-standard-3-llava-qwen2-vl
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 65
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
@@ -100,7 +100,7 @@ steps:
|
|||||||
|
|
||||||
- label: Multi-Modal Processor # 44min
|
- label: Multi-Modal Processor # 44min
|
||||||
key: multi-modal-processor
|
key: multi-modal-processor
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 60
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
|
|||||||
@@ -0,0 +1,29 @@
|
|||||||
|
group: Samplers Intel
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
steps:
|
||||||
|
- label: Samplers Test (FlashInfer)
|
||||||
|
key: samplers-test-flashinfer-intel
|
||||||
|
timeout_in_minutes: 40
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/model_executor/layers
|
||||||
|
- vllm/sampling_metadata.py
|
||||||
|
- tests/samplers
|
||||||
|
- tests/conftest.py
|
||||||
|
- vllm/entrypoints/generate/beam_search
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'cd tests &&
|
||||||
|
VLLM_USE_FLASHINFER_SAMPLER=1 pytest -v -s samplers'
|
||||||
@@ -17,7 +17,7 @@ steps:
|
|||||||
- label: "XPU example Test"
|
- label: "XPU example Test"
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 50
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
@@ -76,7 +76,7 @@ steps:
|
|||||||
- label: "XPU V1 test"
|
- label: "XPU V1 test"
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 70
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
@@ -99,12 +99,13 @@ steps:
|
|||||||
pytest -v -s v1/worker --ignore=v1/worker/test_gpu_model_runner.py --ignore=v1/worker/test_worker_memory_snapshot.py &&
|
pytest -v -s v1/worker --ignore=v1/worker/test_gpu_model_runner.py --ignore=v1/worker/test_worker_memory_snapshot.py &&
|
||||||
pytest -v -s v1/structured_output &&
|
pytest -v -s v1/structured_output &&
|
||||||
pytest -v -s v1/test_serial_utils.py &&
|
pytest -v -s v1/test_serial_utils.py &&
|
||||||
|
pytest -v -s v1/e2e/general/test_correctness_sliding_window.py --deselect="tests/v1/e2e/general/test_correctness_sliding_window.py::test_sliding_window_retrieval[True-1-5-google/gemma-3-1b-it]" &&
|
||||||
pytest -v -s v1/spec_decode --ignore=v1/spec_decode/test_max_len.py --ignore=v1/spec_decode/test_speculators_eagle3.py --ignore=v1/spec_decode/test_acceptance_length.py --ignore=v1/spec_decode/test_speculators_correctness.py &&
|
pytest -v -s v1/spec_decode --ignore=v1/spec_decode/test_max_len.py --ignore=v1/spec_decode/test_speculators_eagle3.py --ignore=v1/spec_decode/test_acceptance_length.py --ignore=v1/spec_decode/test_speculators_correctness.py &&
|
||||||
pytest -v -s v1/kv_connector/unit --ignore=v1/kv_connector/unit/test_multi_connector.py --ignore=v1/kv_connector/unit/test_example_connector.py --ignore=v1/kv_connector/unit/test_lmcache_integration.py --ignore=v1/kv_connector/unit/test_hf3fs_client.py --ignore=v1/kv_connector/unit/test_hf3fs_connector.py --ignore=v1/kv_connector/unit/test_hf3fs_metadata_server.py --ignore=v1/kv_connector/unit/test_offloading_connector.py'
|
pytest -v -s v1/kv_connector/unit --ignore=v1/kv_connector/unit/test_multi_connector.py --ignore=v1/kv_connector/unit/test_example_connector.py --ignore=v1/kv_connector/unit/test_lmcache_integration.py --ignore=v1/kv_connector/unit/test_hf3fs_client.py --ignore=v1/kv_connector/unit/test_hf3fs_connector.py --ignore=v1/kv_connector/unit/test_hf3fs_metadata_server.py --ignore=v1/kv_connector/unit/test_offloading_connector.py'
|
||||||
- label: "XPU server test"
|
- label: "XPU server test"
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
agent_tags:
|
agent_tags:
|
||||||
label: production
|
label: production
|
||||||
@@ -144,7 +145,32 @@ steps:
|
|||||||
- >-
|
- >-
|
||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'cd tests &&
|
'cd tests &&
|
||||||
pytest -v -s quantization/test_auto_round.py'
|
pytest -v -s quantization/test_auto_round.py &&
|
||||||
|
pytest -v -s quantization/test_online.py'
|
||||||
|
- label: "XPU GPQA Eval (GPT-OSS)"
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
|
no_plugin: true
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/
|
||||||
|
- tests/evals/gpt_oss/
|
||||||
|
- .buildkite/intel_jobs/test-intel.yaml
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'pip install "gpt-oss[eval]==0.0.5" &&
|
||||||
|
cd tests &&
|
||||||
|
pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-xpu.txt'
|
||||||
- label: "XPU compressed tensors FP8 test"
|
- label: "XPU compressed tensors FP8 test"
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
# For hf script, without -t option (tensor parallel size).
|
# For hf script, without -t option (tensor parallel size).
|
||||||
# bash .buildkite/lm-eval-harness/run-lm-eval-mmlupro-vllm-baseline.sh -m meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8 -l 250 -t 8 -f 5
|
# bash .buildkite/lm-eval-harness/run-lm-eval-mmlupro-vllm-baseline.sh -m meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8 -l 250 -t 8 -f 5
|
||||||
model_name: "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8"
|
model_name: "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8"
|
||||||
|
rocm_safetensors_load_strategy: lazy
|
||||||
required_gpu_arch:
|
required_gpu_arch:
|
||||||
- gfx942
|
- gfx942
|
||||||
- gfx950
|
- gfx950
|
||||||
|
|||||||
@@ -72,6 +72,11 @@ def launch_lm_eval(eval_config, tp_size):
|
|||||||
if moe_backend is not None:
|
if moe_backend is not None:
|
||||||
model_args += f"moe_backend={moe_backend},"
|
model_args += f"moe_backend={moe_backend},"
|
||||||
|
|
||||||
|
if current_platform.is_rocm():
|
||||||
|
rocm_load_strategy = eval_config.get("rocm_safetensors_load_strategy")
|
||||||
|
if rocm_load_strategy is not None:
|
||||||
|
model_args += f"safetensors_load_strategy={rocm_load_strategy},"
|
||||||
|
|
||||||
env_vars = eval_config.get("env_vars", None)
|
env_vars = eval_config.get("env_vars", None)
|
||||||
with scoped_env_vars(env_vars):
|
with scoped_env_vars(env_vars):
|
||||||
results = lm_eval.simple_evaluate(
|
results = lm_eval.simple_evaluate(
|
||||||
|
|||||||
@@ -28,11 +28,6 @@
|
|||||||
"dataset_path": "./ShareGPT_V3_unfiltered_cleaned_split.json"
|
"dataset_path": "./ShareGPT_V3_unfiltered_cleaned_split.json"
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
{
|
|
||||||
"dataset_name": "sharegpt",
|
|
||||||
"dataset_path": "./ShareGPT_V3_unfiltered_cleaned_split.json"
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"test_name": "serving_llama8B_tp1_random_128_128",
|
"test_name": "serving_llama8B_tp1_random_128_128",
|
||||||
"server_parameters": {
|
"server_parameters": {
|
||||||
|
|||||||
+444
-371
File diff suppressed because it is too large
Load Diff
@@ -3,7 +3,8 @@
|
|||||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
#
|
#
|
||||||
# Append a build artifact line to the Buildkite annotation.
|
# Append a build artifact line to the Buildkite annotation.
|
||||||
# Usage: annotate-build-artifact.sh <label> <value>
|
# Usage: annotate-build-artifact.sh <label> <value> <context>
|
||||||
set -e
|
set -e
|
||||||
echo "- **${1}**: \`${2}\`" | \
|
echo "- **${1}**: \`${2}\`" | \
|
||||||
buildkite-agent annotate --append --style 'info' --context 'release-artifacts'
|
buildkite-agent annotate --append --style 'info' \
|
||||||
|
--context "${3:?context is required}"
|
||||||
|
|||||||
Executable
+35
@@ -0,0 +1,35 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
#
|
||||||
|
# Build the macOS arm64 CPU wheel natively on a macOS agent (the `macmini`
|
||||||
|
# queue) into artifacts/dist/ for upload-nightly-wheels.sh.
|
||||||
|
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
# The macmini queue uses persistent checkouts, so refresh tags for setuptools-scm.
|
||||||
|
git fetch --tags --force origin
|
||||||
|
|
||||||
|
# The Rust frontend build needs protoc.
|
||||||
|
if ! command -v protoc >/dev/null 2>&1; then
|
||||||
|
brew install protobuf
|
||||||
|
fi
|
||||||
|
|
||||||
|
# upload-nightly-wheels.sh expects exactly one wheel.
|
||||||
|
rm -rf artifacts/dist
|
||||||
|
mkdir -p artifacts/dist
|
||||||
|
|
||||||
|
export VLLM_TARGET_DEVICE=cpu
|
||||||
|
export VLLM_REQUIRE_RUST_FRONTEND=1
|
||||||
|
export MACOSX_DEPLOYMENT_TARGET=11.0
|
||||||
|
# uv's CPython is universal2; force an arm64-only build and tag so the wheel
|
||||||
|
# isn't mislabelled universal2 and installed on Intel Macs where import fails.
|
||||||
|
export ARCHFLAGS="-arch arm64"
|
||||||
|
export _PYTHON_HOST_PLATFORM="macosx-11.0-arm64"
|
||||||
|
export CMAKE_BUILD_PARALLEL_LEVEL="${CMAKE_BUILD_PARALLEL_LEVEL:-4}"
|
||||||
|
|
||||||
|
uv venv --python 3.12
|
||||||
|
uv pip install -r requirements/build/cpu.txt --index-strategy unsafe-best-match
|
||||||
|
uv build --wheel --no-build-isolation -o artifacts/dist
|
||||||
|
|
||||||
|
ls -l artifacts/dist/*.whl
|
||||||
@@ -15,9 +15,9 @@ set -euo pipefail
|
|||||||
|
|
||||||
DEFAULT_REPO_SLUG="vllm-project/vllm"
|
DEFAULT_REPO_SLUG="vllm-project/vllm"
|
||||||
DEFAULT_CI_HCL_SOURCE="docker/ci-rocm.hcl"
|
DEFAULT_CI_HCL_SOURCE="docker/ci-rocm.hcl"
|
||||||
DEFAULT_CI_BASE_CONTENT_FILES="requirements/common.txt requirements/rocm.txt requirements/test/rocm.txt docker/Dockerfile.rocm_base docker/ci-rocm.hcl docker/docker-bake-rocm.hcl tools/install_torchcodec_rocm.sh tests/vllm_test_utils .buildkite/scripts/ci-bake-rocm.sh .buildkite/scripts/rocm/build-ci-base.sh"
|
DEFAULT_CI_BASE_CONTENT_FILES="requirements/common.txt requirements/rocm.txt requirements/test/rocm.txt docker/Dockerfile.rocm_base docker/ci-rocm.hcl docker/docker-bake-rocm.hcl tools/install_torchcodec_rocm.sh tools/install_protoc.sh rust-toolchain.toml tests/vllm_test_utils .buildkite/scripts/ci-bake-rocm.sh .buildkite/scripts/rocm/build-ci-base.sh"
|
||||||
DEFAULT_CI_BASE_DOCKERFILE="docker/Dockerfile.rocm"
|
DEFAULT_CI_BASE_DOCKERFILE="docker/Dockerfile.rocm"
|
||||||
DEFAULT_CI_BASE_DOCKERFILE_STAGES="base build_rixl build_rocshmem build_deepep mori_base ci_base"
|
DEFAULT_CI_BASE_DOCKERFILE_STAGES="base rust_toolchain_input_0 rust_toolchain_input_1 rust-toolchain-input rust-toolchain build_nixl build_rocshmem build_deepep mori_base ci_base"
|
||||||
DEFAULT_CI_BASE_METADATA_VERSION="1"
|
DEFAULT_CI_BASE_METADATA_VERSION="1"
|
||||||
IMAGE_EXISTED_BEFORE_BUILD=0
|
IMAGE_EXISTED_BEFORE_BUILD=0
|
||||||
|
|
||||||
@@ -285,7 +285,7 @@ get_content_arg_names() {
|
|||||||
fi | awk 'NF && !seen[$0]++'
|
fi | awk 'NF && !seen[$0]++'
|
||||||
}
|
}
|
||||||
|
|
||||||
compute_ci_base_content_hash() {
|
compute_ci_base_content_hash_once() {
|
||||||
local -a content_paths=()
|
local -a content_paths=()
|
||||||
local -a content_args=()
|
local -a content_args=()
|
||||||
local dockerfile="${CI_BASE_DOCKERFILE:-}"
|
local dockerfile="${CI_BASE_DOCKERFILE:-}"
|
||||||
@@ -301,7 +301,8 @@ compute_ci_base_content_hash() {
|
|||||||
if [[ -n "${dockerfile}" ]]; then
|
if [[ -n "${dockerfile}" ]]; then
|
||||||
printf 'dockerfile:%s\n' "${dockerfile}"
|
printf 'dockerfile:%s\n' "${dockerfile}"
|
||||||
printf 'resolved-build-args:\n'
|
printf 'resolved-build-args:\n'
|
||||||
hash_dockerfile_arg_values "${dockerfile}" "${content_args[@]}"
|
hash_dockerfile_arg_values "${dockerfile}" "${content_args[@]}" \
|
||||||
|
|| return 1
|
||||||
if [[ -n "${stages}" ]]; then
|
if [[ -n "${stages}" ]]; then
|
||||||
printf 'dockerfile-stages:%s\n' "${stages}"
|
printf 'dockerfile-stages:%s\n' "${stages}"
|
||||||
if [[ -f "${dockerfile}" ]]; then
|
if [[ -f "${dockerfile}" ]]; then
|
||||||
@@ -314,6 +315,53 @@ compute_ci_base_content_hash() {
|
|||||||
} | sha256sum | cut -d' ' -f1
|
} | sha256sum | cut -d' ' -f1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
compute_ci_base_content_hash() {
|
||||||
|
local attempts="${CI_BASE_HASH_ATTEMPTS:-3}"
|
||||||
|
local delay_secs="${CI_BASE_HASH_RETRY_DELAY:-5}"
|
||||||
|
local attempt=0
|
||||||
|
local hash=""
|
||||||
|
local failed=0
|
||||||
|
local -a hashes=()
|
||||||
|
|
||||||
|
if [[ ! "${attempts}" =~ ^[1-9][0-9]*$ ]]; then
|
||||||
|
echo "Invalid CI_BASE_HASH_ATTEMPTS: ${attempts}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
if [[ ! "${delay_secs}" =~ ^[0-9]+$ ]]; then
|
||||||
|
echo "Invalid CI_BASE_HASH_RETRY_DELAY: ${delay_secs}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
for ((attempt = 1; attempt <= attempts; attempt++)); do
|
||||||
|
if ! hash=$(compute_ci_base_content_hash_once); then
|
||||||
|
echo "ci_base content hash calculation ${attempt}/${attempts} failed" >&2
|
||||||
|
failed=1
|
||||||
|
else
|
||||||
|
hashes+=("${hash}")
|
||||||
|
echo "ci_base content hash calculation ${attempt}/${attempts}: ${hash}" >&2
|
||||||
|
fi
|
||||||
|
|
||||||
|
if ((attempt < attempts)); then
|
||||||
|
sleep "${delay_secs}"
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if ((failed)) || ((${#hashes[@]} != attempts)); then
|
||||||
|
echo "Could not calculate a reliable ci_base content hash" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
for hash in "${hashes[@]:1}"; do
|
||||||
|
if [[ "${hash}" != "${hashes[0]}" ]]; then
|
||||||
|
echo "ci_base content hash changed between calculations" >&2
|
||||||
|
printf ' observed: %s\n' "${hashes[@]}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
printf '%s\n' "${hashes[0]}"
|
||||||
|
}
|
||||||
|
|
||||||
extract_dockerfile_arg_default() {
|
extract_dockerfile_arg_default() {
|
||||||
local dockerfile="$1"
|
local dockerfile="$1"
|
||||||
local arg_name="$2"
|
local arg_name="$2"
|
||||||
@@ -366,7 +414,11 @@ hash_dockerfile_arg_values() {
|
|||||||
printf 'arg:%s=%s\n' "${arg_name}" "${arg_value:-<empty>}"
|
printf 'arg:%s=%s\n' "${arg_name}" "${arg_value:-<empty>}"
|
||||||
if [[ "${arg_name}" == "BASE_IMAGE" && -n "${arg_value}" ]]; then
|
if [[ "${arg_name}" == "BASE_IMAGE" && -n "${arg_value}" ]]; then
|
||||||
digest=$(resolve_image_digest "${arg_value}")
|
digest=$(resolve_image_digest "${arg_value}")
|
||||||
printf 'arg:%s.digest=%s\n' "${arg_name}" "${digest:-unknown}"
|
if [[ -z "${digest}" ]]; then
|
||||||
|
echo "Failed to resolve digest for BASE_IMAGE=${arg_value}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
printf 'arg:%s.digest=%s\n' "${arg_name}" "${digest}"
|
||||||
fi
|
fi
|
||||||
done
|
done
|
||||||
}
|
}
|
||||||
@@ -764,7 +816,7 @@ configure_ci_base_image_refs() {
|
|||||||
fi
|
fi
|
||||||
set_buildkite_metadata "rocm-ci-base-image" "${CI_BASE_IMAGE_TAG}"
|
set_buildkite_metadata "rocm-ci-base-image" "${CI_BASE_IMAGE_TAG}"
|
||||||
set_buildkite_metadata "rocm-ci-base-image-content" "${content_tag}"
|
set_buildkite_metadata "rocm-ci-base-image-content" "${content_tag}"
|
||||||
set_buildkite_metadata "rocm-ci-base-image-commit" "${CI_BASE_IMAGE_TAG_COMMIT:-}"
|
set_buildkite_metadata "rocm-ci-base-image-commit" "${CI_BASE_IMAGE_TAG_COMMIT_REF:-}"
|
||||||
set_buildkite_metadata "rocm-ci-base-image-stable" "${CI_BASE_IMAGE_TAG_STABLE:-}"
|
set_buildkite_metadata "rocm-ci-base-image-stable" "${CI_BASE_IMAGE_TAG_STABLE:-}"
|
||||||
return 0
|
return 0
|
||||||
fi
|
fi
|
||||||
@@ -1107,8 +1159,8 @@ ci_base_metadata_pairs() {
|
|||||||
metadata_pair "vllm.rocm.nic_backend" "$(resolve_dockerfile_arg_value "${dockerfile}" "NIC_BACKEND")"
|
metadata_pair "vllm.rocm.nic_backend" "$(resolve_dockerfile_arg_value "${dockerfile}" "NIC_BACKEND")"
|
||||||
metadata_pair "vllm.rocm.ainic_version" "$(resolve_dockerfile_arg_value "${dockerfile}" "AINIC_VERSION")"
|
metadata_pair "vllm.rocm.ainic_version" "$(resolve_dockerfile_arg_value "${dockerfile}" "AINIC_VERSION")"
|
||||||
metadata_pair "vllm.rocm.ubuntu_codename" "$(resolve_dockerfile_arg_value "${dockerfile}" "UBUNTU_CODENAME")"
|
metadata_pair "vllm.rocm.ubuntu_codename" "$(resolve_dockerfile_arg_value "${dockerfile}" "UBUNTU_CODENAME")"
|
||||||
metadata_pair "vllm.rocm.rixl_repo" "$(resolve_dockerfile_arg_value "${dockerfile}" "RIXL_REPO")"
|
metadata_pair "vllm.rocm.nixl_repo" "$(resolve_dockerfile_arg_value "${dockerfile}" "NIXL_REPO")"
|
||||||
metadata_pair "vllm.rocm.rixl_commit" "${RIXL_BRANCH:-$(resolve_dockerfile_arg_value "${dockerfile}" "RIXL_BRANCH")}"
|
metadata_pair "vllm.rocm.nixl_commit" "${NIXL_BRANCH:-$(resolve_dockerfile_arg_value "${dockerfile}" "NIXL_BRANCH")}"
|
||||||
metadata_pair "vllm.rocm.ucx_repo" "$(resolve_dockerfile_arg_value "${dockerfile}" "UCX_REPO")"
|
metadata_pair "vllm.rocm.ucx_repo" "$(resolve_dockerfile_arg_value "${dockerfile}" "UCX_REPO")"
|
||||||
metadata_pair "vllm.rocm.ucx_commit" "${UCX_BRANCH:-$(resolve_dockerfile_arg_value "${dockerfile}" "UCX_BRANCH")}"
|
metadata_pair "vllm.rocm.ucx_commit" "${UCX_BRANCH:-$(resolve_dockerfile_arg_value "${dockerfile}" "UCX_BRANCH")}"
|
||||||
metadata_pair "vllm.rocm.rocshmem_repo" "$(resolve_dockerfile_arg_value "${dockerfile}" "ROCSHMEM_REPO")"
|
metadata_pair "vllm.rocm.rocshmem_repo" "$(resolve_dockerfile_arg_value "${dockerfile}" "ROCSHMEM_REPO")"
|
||||||
@@ -1117,7 +1169,7 @@ ci_base_metadata_pairs() {
|
|||||||
metadata_pair "vllm.rocm.deepep_commit" "${DEEPEP_BRANCH:-$(resolve_dockerfile_arg_value "${dockerfile}" "DEEPEP_BRANCH")}"
|
metadata_pair "vllm.rocm.deepep_commit" "${DEEPEP_BRANCH:-$(resolve_dockerfile_arg_value "${dockerfile}" "DEEPEP_BRANCH")}"
|
||||||
metadata_pair "vllm.rocm.deepep_nic" "$(resolve_dockerfile_arg_value "${dockerfile}" "DEEPEP_NIC")"
|
metadata_pair "vllm.rocm.deepep_nic" "$(resolve_dockerfile_arg_value "${dockerfile}" "DEEPEP_NIC")"
|
||||||
metadata_pair "vllm.rocm.deepep_rocm_arch" "$(resolve_dockerfile_arg_value "${dockerfile}" "DEEPEP_ROCM_ARCH")"
|
metadata_pair "vllm.rocm.deepep_rocm_arch" "$(resolve_dockerfile_arg_value "${dockerfile}" "DEEPEP_ROCM_ARCH")"
|
||||||
metadata_pair "vllm.rocm.rixl_cache_key" "${RIXL_CACHE_KEY:-}"
|
metadata_pair "vllm.rocm.nixl_cache_key" "${NIXL_CACHE_KEY:-}"
|
||||||
metadata_pair "vllm.rocm.rocshmem_cache_key" "${ROCSHMEM_CACHE_KEY:-}"
|
metadata_pair "vllm.rocm.rocshmem_cache_key" "${ROCSHMEM_CACHE_KEY:-}"
|
||||||
metadata_pair "vllm.rocm.deepep_cache_key" "${DEEPEP_CACHE_KEY:-}"
|
metadata_pair "vllm.rocm.deepep_cache_key" "${DEEPEP_CACHE_KEY:-}"
|
||||||
|
|
||||||
@@ -1211,12 +1263,24 @@ uses_rocm_csrc_cache() {
|
|||||||
esac
|
esac
|
||||||
}
|
}
|
||||||
|
|
||||||
|
uses_rocm_rust_cache() {
|
||||||
|
case "${TARGET}" in
|
||||||
|
rust-rocm-ci|test-rocm-ci|test-rocm-ci-with-wheel|test-rocm-ci-with-artifacts|export-wheel-rocm)
|
||||||
|
return 0
|
||||||
|
;;
|
||||||
|
*)
|
||||||
|
return 1
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
}
|
||||||
|
|
||||||
compute_rocm_csrc_content_hash() {
|
compute_rocm_csrc_content_hash() {
|
||||||
local bake_dir=""
|
local bake_dir=""
|
||||||
local dockerfile_rocm=""
|
local dockerfile_rocm=""
|
||||||
local -a content_paths=(
|
local -a content_paths=(
|
||||||
"requirements/common.txt"
|
"requirements/common.txt"
|
||||||
"requirements/rocm.txt"
|
"requirements/rocm.txt"
|
||||||
|
"pyproject.toml"
|
||||||
"setup.py"
|
"setup.py"
|
||||||
"CMakeLists.txt"
|
"CMakeLists.txt"
|
||||||
"cmake"
|
"cmake"
|
||||||
@@ -1260,6 +1324,56 @@ compute_rocm_csrc_content_hash_if_needed() {
|
|||||||
echo "ROCm csrc content cache ref: ${ROCM_CSRC_CONTENT_CACHE_REF}"
|
echo "ROCm csrc content cache ref: ${ROCM_CSRC_CONTENT_CACHE_REF}"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
compute_rocm_rust_content_hash() {
|
||||||
|
local bake_dir=""
|
||||||
|
local dockerfile_rocm=""
|
||||||
|
local -a content_paths=(
|
||||||
|
"requirements/build/rust.txt"
|
||||||
|
"rust/Cargo.lock"
|
||||||
|
"rust/Cargo.toml"
|
||||||
|
"rust/proto"
|
||||||
|
"rust/src"
|
||||||
|
"rust-toolchain.toml"
|
||||||
|
"tools/build_rust.py"
|
||||||
|
"tools/install_protoc.sh"
|
||||||
|
"build_rust.sh"
|
||||||
|
)
|
||||||
|
local -a content_args=()
|
||||||
|
|
||||||
|
bake_dir=$(dirname "${VLLM_BAKE_FILE}")
|
||||||
|
dockerfile_rocm="${bake_dir}/Dockerfile.rocm"
|
||||||
|
mapfile -t content_args < <(
|
||||||
|
get_content_arg_names "${dockerfile_rocm}" "base rust_toolchain_input_0 rust_toolchain_input_1 rust-toolchain-input rust_input_0 rust_input_1 rust-input rust-toolchain rust-build" "${ROCM_RUST_CONTENT_ARGS:-}"
|
||||||
|
)
|
||||||
|
|
||||||
|
{
|
||||||
|
printf 'rust-input-files-hash:%s\n' "$(compute_content_hash "${content_paths[@]}")"
|
||||||
|
printf 'dockerfile:%s\n' "${dockerfile_rocm}"
|
||||||
|
printf 'resolved-build-args:\n'
|
||||||
|
hash_dockerfile_arg_values "${dockerfile_rocm}" "${content_args[@]}"
|
||||||
|
printf 'dockerfile-stages:base rust_toolchain_input_0 rust_toolchain_input_1 rust-toolchain-input rust_input_0 rust_input_1 rust-input rust-toolchain rust-build\n'
|
||||||
|
if [[ -f "${dockerfile_rocm}" ]]; then
|
||||||
|
hash_dockerfile_stages "${dockerfile_rocm}" "base rust_toolchain_input_0 rust_toolchain_input_1 rust-toolchain-input rust_input_0 rust_input_1 rust-input rust-toolchain rust-build"
|
||||||
|
else
|
||||||
|
printf 'missing:%s\n' "${dockerfile_rocm}"
|
||||||
|
fi
|
||||||
|
} | sha256sum | cut -d' ' -f1
|
||||||
|
}
|
||||||
|
|
||||||
|
compute_rocm_rust_content_hash_if_needed() {
|
||||||
|
local cache_repo="${DOCKERHUB_CACHE_REPO:-rocm/vllm-ci-cache}"
|
||||||
|
|
||||||
|
if [[ "${ROCM_RUST_CONTENT_CACHE:-1}" == "0" ]] || ! uses_rocm_rust_cache; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
ROCM_RUST_CONTENT_HASH=$(compute_rocm_rust_content_hash)
|
||||||
|
ROCM_RUST_CONTENT_CACHE_REF="${cache_repo}:rust-rocm-input-${ROCM_RUST_CONTENT_HASH}"
|
||||||
|
export ROCM_RUST_CONTENT_HASH
|
||||||
|
export ROCM_RUST_CONTENT_CACHE_REF
|
||||||
|
echo "ROCm Rust content cache ref: ${ROCM_RUST_CONTENT_CACHE_REF}"
|
||||||
|
}
|
||||||
|
|
||||||
write_hcl_string_list_entries() {
|
write_hcl_string_list_entries() {
|
||||||
local indent="$1"
|
local indent="$1"
|
||||||
local value=""
|
local value=""
|
||||||
@@ -1317,6 +1431,7 @@ write_rocm_build_arg_override() {
|
|||||||
"${CI_BASE_DOCKERFILE_STAGES:-${DEFAULT_CI_BASE_DOCKERFILE_STAGES}}" \
|
"${CI_BASE_DOCKERFILE_STAGES:-${DEFAULT_CI_BASE_DOCKERFILE_STAGES}}" \
|
||||||
"${CI_BASE_CONTENT_ARGS:-}"
|
"${CI_BASE_CONTENT_ARGS:-}"
|
||||||
get_content_arg_names "${dockerfile_rocm}" "base csrc-build" "${ROCM_CSRC_CONTENT_ARGS:-}"
|
get_content_arg_names "${dockerfile_rocm}" "base csrc-build" "${ROCM_CSRC_CONTENT_ARGS:-}"
|
||||||
|
get_content_arg_names "${dockerfile_rocm}" "base rust_toolchain_input_0 rust_toolchain_input_1 rust-toolchain-input rust_input_0 rust_input_1 rust-input rust-toolchain rust-build" "${ROCM_RUST_CONTENT_ARGS:-}"
|
||||||
} | awk 'NF && !seen[$0]++'
|
} | awk 'NF && !seen[$0]++'
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -1365,46 +1480,133 @@ validate_cache_export_mode() {
|
|||||||
esac
|
esac
|
||||||
}
|
}
|
||||||
|
|
||||||
|
validate_content_cache_export_mode() {
|
||||||
|
local mode="$1"
|
||||||
|
local env_name="$2"
|
||||||
|
|
||||||
|
case "${mode}" in
|
||||||
|
missing|always|never)
|
||||||
|
;;
|
||||||
|
*)
|
||||||
|
echo "Error: ${env_name} must be one of: missing, always, never"
|
||||||
|
exit 1
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
}
|
||||||
|
|
||||||
|
should_export_content_cache_ref() {
|
||||||
|
local cache_ref="$1"
|
||||||
|
local cache_name="$2"
|
||||||
|
local mode="${ROCM_CONTENT_CACHE_EXPORT_MODE:-missing}"
|
||||||
|
|
||||||
|
case "${mode}" in
|
||||||
|
always)
|
||||||
|
echo "${cache_name} content cache export mode is always; exporting ${cache_ref}"
|
||||||
|
return 0
|
||||||
|
;;
|
||||||
|
never)
|
||||||
|
echo "${cache_name} content cache export mode is never; not exporting ${cache_ref}"
|
||||||
|
return 1
|
||||||
|
;;
|
||||||
|
missing|"")
|
||||||
|
if docker buildx imagetools inspect "${cache_ref}" >/dev/null 2>&1; then
|
||||||
|
echo "${cache_name} content cache exists; not re-exporting ${cache_ref}"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
echo "${cache_name} content cache missing; will export ${cache_ref}"
|
||||||
|
return 0
|
||||||
|
;;
|
||||||
|
*)
|
||||||
|
echo "Error: ROCM_CONTENT_CACHE_EXPORT_MODE must be one of: missing, always, never"
|
||||||
|
exit 1
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
}
|
||||||
|
|
||||||
write_rocm_cache_override() {
|
write_rocm_cache_override() {
|
||||||
local cache_repo="${DOCKERHUB_CACHE_REPO:-rocm/vllm-ci-cache}"
|
local cache_repo="${DOCKERHUB_CACHE_REPO:-rocm/vllm-ci-cache}"
|
||||||
|
local content_cache_export_mode="${ROCM_CONTENT_CACHE_EXPORT_MODE:-missing}"
|
||||||
local csrc_cache_to_mode="${ROCM_CSRC_CACHE_TO_MODE:-max}"
|
local csrc_cache_to_mode="${ROCM_CSRC_CACHE_TO_MODE:-max}"
|
||||||
|
local rust_cache_to_mode="${ROCM_RUST_CACHE_TO_MODE:-max}"
|
||||||
local rocm_cache_to_mode="${ROCM_FINAL_CACHE_TO_MODE:-min}"
|
local rocm_cache_to_mode="${ROCM_FINAL_CACHE_TO_MODE:-min}"
|
||||||
local -a content_cache_from=()
|
local -a csrc_content_cache_from=()
|
||||||
|
local -a rust_content_cache_from=()
|
||||||
|
local -a combined_content_cache_from=()
|
||||||
local -a csrc_cache_to=()
|
local -a csrc_cache_to=()
|
||||||
|
local -a rust_cache_to=()
|
||||||
local -a rocm_cache_to=()
|
local -a rocm_cache_to=()
|
||||||
local -a export_wheel_cache_to=()
|
local -a export_wheel_cache_to=()
|
||||||
|
local export_csrc_cache=1
|
||||||
|
local export_rust_cache=1
|
||||||
|
|
||||||
if ! uses_rocm_csrc_cache; then
|
if ! uses_rocm_csrc_cache && ! uses_rocm_rust_cache; then
|
||||||
return 0
|
return 0
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
validate_content_cache_export_mode \
|
||||||
|
"${content_cache_export_mode}" \
|
||||||
|
"ROCM_CONTENT_CACHE_EXPORT_MODE"
|
||||||
validate_cache_export_mode "${csrc_cache_to_mode}" "ROCM_CSRC_CACHE_TO_MODE"
|
validate_cache_export_mode "${csrc_cache_to_mode}" "ROCM_CSRC_CACHE_TO_MODE"
|
||||||
|
validate_cache_export_mode "${rust_cache_to_mode}" "ROCM_RUST_CACHE_TO_MODE"
|
||||||
validate_cache_export_mode "${rocm_cache_to_mode}" "ROCM_FINAL_CACHE_TO_MODE"
|
validate_cache_export_mode "${rocm_cache_to_mode}" "ROCM_FINAL_CACHE_TO_MODE"
|
||||||
|
echo "ROCm content cache export mode: ${content_cache_export_mode}"
|
||||||
echo "ROCm csrc cache export mode: ${csrc_cache_to_mode}"
|
echo "ROCm csrc cache export mode: ${csrc_cache_to_mode}"
|
||||||
|
echo "ROCm Rust cache export mode: ${rust_cache_to_mode}"
|
||||||
echo "ROCm final image cache export mode: ${rocm_cache_to_mode}"
|
echo "ROCm final image cache export mode: ${rocm_cache_to_mode}"
|
||||||
|
|
||||||
if [[ -n "${ROCM_CSRC_CONTENT_CACHE_REF:-}" ]]; then
|
if [[ -n "${ROCM_CSRC_CONTENT_CACHE_REF:-}" ]]; then
|
||||||
content_cache_from+=("type=registry,ref=${ROCM_CSRC_CONTENT_CACHE_REF}")
|
csrc_content_cache_from+=("type=registry,ref=${ROCM_CSRC_CONTENT_CACHE_REF}")
|
||||||
csrc_cache_to+=(
|
if should_export_content_cache_ref "${ROCM_CSRC_CONTENT_CACHE_REF}" "ROCm csrc"; then
|
||||||
"type=registry,ref=${ROCM_CSRC_CONTENT_CACHE_REF},mode=${csrc_cache_to_mode},ignore-error=true"
|
csrc_cache_to+=(
|
||||||
)
|
"type=registry,ref=${ROCM_CSRC_CONTENT_CACHE_REF},mode=${csrc_cache_to_mode},ignore-error=true"
|
||||||
|
)
|
||||||
|
else
|
||||||
|
export_csrc_cache=0
|
||||||
|
fi
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
if [[ -n "${ROCM_RUST_CONTENT_CACHE_REF:-}" ]]; then
|
||||||
|
rust_content_cache_from+=("type=registry,ref=${ROCM_RUST_CONTENT_CACHE_REF}")
|
||||||
|
if should_export_content_cache_ref "${ROCM_RUST_CONTENT_CACHE_REF}" "ROCm Rust"; then
|
||||||
|
rust_cache_to+=(
|
||||||
|
"type=registry,ref=${ROCM_RUST_CONTENT_CACHE_REF},mode=${rust_cache_to_mode},ignore-error=true"
|
||||||
|
)
|
||||||
|
else
|
||||||
|
export_rust_cache=0
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
combined_content_cache_from=("${csrc_content_cache_from[@]}" "${rust_content_cache_from[@]}")
|
||||||
|
|
||||||
# Docker Hub cache exports are best-effort. A cache-only target failure can
|
# Docker Hub cache exports are best-effort. A cache-only target failure can
|
||||||
# otherwise cancel the sibling image target before its manifest is pushed.
|
# otherwise cancel the sibling image target before its manifest is pushed.
|
||||||
if [[ -n "${BUILDKITE_COMMIT:-}" ]]; then
|
if [[ -n "${BUILDKITE_COMMIT:-}" ]]; then
|
||||||
csrc_cache_to+=(
|
if [[ ${export_csrc_cache} -eq 1 ]]; then
|
||||||
"type=registry,ref=${cache_repo}:csrc-rocm-${BUILDKITE_COMMIT},mode=${csrc_cache_to_mode},ignore-error=true"
|
csrc_cache_to+=(
|
||||||
)
|
"type=registry,ref=${cache_repo}:csrc-rocm-${BUILDKITE_COMMIT},mode=${csrc_cache_to_mode},ignore-error=true"
|
||||||
|
)
|
||||||
|
fi
|
||||||
|
if [[ ${export_rust_cache} -eq 1 ]]; then
|
||||||
|
rust_cache_to+=(
|
||||||
|
"type=registry,ref=${cache_repo}:rust-rocm-${BUILDKITE_COMMIT},mode=${rust_cache_to_mode},ignore-error=true"
|
||||||
|
)
|
||||||
|
fi
|
||||||
rocm_cache_to+=(
|
rocm_cache_to+=(
|
||||||
"type=registry,ref=${cache_repo}:rocm-${BUILDKITE_COMMIT},mode=${rocm_cache_to_mode},ignore-error=true"
|
"type=registry,ref=${cache_repo}:rocm-${BUILDKITE_COMMIT},mode=${rocm_cache_to_mode},ignore-error=true"
|
||||||
)
|
)
|
||||||
fi
|
fi
|
||||||
|
|
||||||
if [[ -n "${ROCM_CACHE_BRANCH_TAG:-}" ]]; then
|
if [[ -n "${ROCM_CACHE_BRANCH_TAG:-}" ]]; then
|
||||||
csrc_cache_to+=(
|
if [[ ${export_csrc_cache} -eq 1 ]]; then
|
||||||
"type=registry,ref=${cache_repo}:csrc-rocm-branch-${ROCM_CACHE_BRANCH_TAG},mode=${csrc_cache_to_mode},ignore-error=true"
|
csrc_cache_to+=(
|
||||||
)
|
"type=registry,ref=${cache_repo}:csrc-rocm-branch-${ROCM_CACHE_BRANCH_TAG},mode=${csrc_cache_to_mode},ignore-error=true"
|
||||||
|
)
|
||||||
|
fi
|
||||||
|
if [[ ${export_rust_cache} -eq 1 ]]; then
|
||||||
|
rust_cache_to+=(
|
||||||
|
"type=registry,ref=${cache_repo}:rust-rocm-branch-${ROCM_CACHE_BRANCH_TAG},mode=${rust_cache_to_mode},ignore-error=true"
|
||||||
|
)
|
||||||
|
fi
|
||||||
rocm_cache_to+=(
|
rocm_cache_to+=(
|
||||||
"type=registry,ref=${cache_repo}:rocm-branch-${ROCM_CACHE_BRANCH_TAG},mode=${rocm_cache_to_mode},ignore-error=true"
|
"type=registry,ref=${cache_repo}:rocm-branch-${ROCM_CACHE_BRANCH_TAG},mode=${rocm_cache_to_mode},ignore-error=true"
|
||||||
)
|
)
|
||||||
@@ -1422,7 +1624,7 @@ target "csrc-rocm-ci" {
|
|||||||
cache-from = concat(
|
cache-from = concat(
|
||||||
get_cache_from_rocm_csrc(),
|
get_cache_from_rocm_csrc(),
|
||||||
EOF
|
EOF
|
||||||
write_hcl_string_list " " "${content_cache_from[@]}"
|
write_hcl_string_list " " "${csrc_content_cache_from[@]}"
|
||||||
cat <<EOF
|
cat <<EOF
|
||||||
)
|
)
|
||||||
EOF
|
EOF
|
||||||
@@ -1430,11 +1632,23 @@ EOF
|
|||||||
cat <<EOF
|
cat <<EOF
|
||||||
}
|
}
|
||||||
|
|
||||||
|
target "rust-rocm-ci" {
|
||||||
|
cache-from = concat(
|
||||||
|
get_cache_from_rocm_rust(),
|
||||||
|
EOF
|
||||||
|
write_hcl_string_list " " "${rust_content_cache_from[@]}"
|
||||||
|
cat <<EOF
|
||||||
|
)
|
||||||
|
EOF
|
||||||
|
write_hcl_string_list_attr " " "cache-to" "${rust_cache_to[@]}"
|
||||||
|
cat <<EOF
|
||||||
|
}
|
||||||
|
|
||||||
target "test-rocm-ci" {
|
target "test-rocm-ci" {
|
||||||
cache-from = concat(
|
cache-from = concat(
|
||||||
get_cache_from_rocm(),
|
get_cache_from_rocm(),
|
||||||
EOF
|
EOF
|
||||||
write_hcl_string_list " " "${content_cache_from[@]}"
|
write_hcl_string_list " " "${combined_content_cache_from[@]}"
|
||||||
cat <<EOF
|
cat <<EOF
|
||||||
)
|
)
|
||||||
EOF
|
EOF
|
||||||
@@ -1446,7 +1660,7 @@ target "export-wheel-rocm" {
|
|||||||
cache-from = concat(
|
cache-from = concat(
|
||||||
get_cache_from_rocm(),
|
get_cache_from_rocm(),
|
||||||
EOF
|
EOF
|
||||||
write_hcl_string_list " " "${content_cache_from[@]}"
|
write_hcl_string_list " " "${combined_content_cache_from[@]}"
|
||||||
cat <<EOF
|
cat <<EOF
|
||||||
)
|
)
|
||||||
EOF
|
EOF
|
||||||
@@ -1472,7 +1686,7 @@ extract_dependency_pins() {
|
|||||||
return 0
|
return 0
|
||||||
fi
|
fi
|
||||||
|
|
||||||
for var in RIXL_BRANCH UCX_BRANCH ROCSHMEM_BRANCH DEEPEP_BRANCH; do
|
for var in NIXL_BRANCH UCX_BRANCH ROCSHMEM_BRANCH DEEPEP_BRANCH; do
|
||||||
if [[ -n "${!var:-}" ]]; then
|
if [[ -n "${!var:-}" ]]; then
|
||||||
echo "Using provided ${var}: ${!var}"
|
echo "Using provided ${var}: ${!var}"
|
||||||
continue
|
continue
|
||||||
@@ -1492,30 +1706,30 @@ extract_dependency_pins() {
|
|||||||
compute_dependency_cache_keys() {
|
compute_dependency_cache_keys() {
|
||||||
local bake_dir=""
|
local bake_dir=""
|
||||||
local dockerfile_rocm=""
|
local dockerfile_rocm=""
|
||||||
local rixl_branch=""
|
local nixl_branch=""
|
||||||
local ucx_branch=""
|
local ucx_branch=""
|
||||||
local rocshmem_branch=""
|
local rocshmem_branch=""
|
||||||
local deepep_branch=""
|
local deepep_branch=""
|
||||||
local rixl_material=""
|
local nixl_material=""
|
||||||
local rocshmem_material=""
|
local rocshmem_material=""
|
||||||
local deepep_material=""
|
local deepep_material=""
|
||||||
|
|
||||||
bake_dir=$(dirname "${VLLM_BAKE_FILE}")
|
bake_dir=$(dirname "${VLLM_BAKE_FILE}")
|
||||||
dockerfile_rocm="${bake_dir}/Dockerfile.rocm"
|
dockerfile_rocm="${bake_dir}/Dockerfile.rocm"
|
||||||
rixl_branch=$(resolve_dockerfile_arg_value "${dockerfile_rocm}" "RIXL_BRANCH")
|
nixl_branch=$(resolve_dockerfile_arg_value "${dockerfile_rocm}" "NIXL_BRANCH")
|
||||||
ucx_branch=$(resolve_dockerfile_arg_value "${dockerfile_rocm}" "UCX_BRANCH")
|
ucx_branch=$(resolve_dockerfile_arg_value "${dockerfile_rocm}" "UCX_BRANCH")
|
||||||
rocshmem_branch=$(resolve_dockerfile_arg_value "${dockerfile_rocm}" "ROCSHMEM_BRANCH")
|
rocshmem_branch=$(resolve_dockerfile_arg_value "${dockerfile_rocm}" "ROCSHMEM_BRANCH")
|
||||||
deepep_branch=$(resolve_dockerfile_arg_value "${dockerfile_rocm}" "DEEPEP_BRANCH")
|
deepep_branch=$(resolve_dockerfile_arg_value "${dockerfile_rocm}" "DEEPEP_BRANCH")
|
||||||
|
|
||||||
if [[ -n "${rixl_branch}" && -n "${ucx_branch}" ]]; then
|
if [[ -n "${nixl_branch}" && -n "${ucx_branch}" ]]; then
|
||||||
rixl_material=$(compose_stage_cache_material "${dockerfile_rocm}" "base build_rixl")
|
nixl_material=$(compose_stage_cache_material "${dockerfile_rocm}" "base build_nixl")
|
||||||
RIXL_CACHE_KEY=$(
|
NIXL_CACHE_KEY=$(
|
||||||
compose_dependency_cache_key \
|
compose_dependency_cache_key \
|
||||||
"${rixl_branch}-ucx-${ucx_branch}" \
|
"${nixl_branch}-ucx-${ucx_branch}" \
|
||||||
"${rixl_material}"
|
"${nixl_material}"
|
||||||
)
|
)
|
||||||
export RIXL_CACHE_KEY
|
export NIXL_CACHE_KEY
|
||||||
echo "RIXL dependency cache key: ${RIXL_CACHE_KEY}"
|
echo "NIXL dependency cache key: ${NIXL_CACHE_KEY}"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
if [[ -n "${rocshmem_branch}" ]]; then
|
if [[ -n "${rocshmem_branch}" ]]; then
|
||||||
@@ -1566,11 +1780,11 @@ dependency_cache_ref_for_target() {
|
|||||||
local cache_repo="${DOCKERHUB_CACHE_REPO:-rocm/vllm-ci-cache}"
|
local cache_repo="${DOCKERHUB_CACHE_REPO:-rocm/vllm-ci-cache}"
|
||||||
|
|
||||||
case "${target}" in
|
case "${target}" in
|
||||||
rixl-rocm-ci)
|
nixl-rocm-ci)
|
||||||
if [[ -n "${RIXL_CACHE_KEY:-}" ]]; then
|
if [[ -n "${NIXL_CACHE_KEY:-}" ]]; then
|
||||||
printf '%s\n' "${cache_repo}:rixl-rocm-${RIXL_CACHE_KEY}"
|
printf '%s\n' "${cache_repo}:nixl-rocm-${NIXL_CACHE_KEY}"
|
||||||
elif [[ -n "${RIXL_BRANCH:-}" ]]; then
|
elif [[ -n "${NIXL_BRANCH:-}" ]]; then
|
||||||
printf '%s\n' "${cache_repo}:rixl-rocm-${RIXL_BRANCH}-ucx-${UCX_BRANCH:-}"
|
printf '%s\n' "${cache_repo}:nixl-rocm-${NIXL_BRANCH}-ucx-${UCX_BRANCH:-}"
|
||||||
fi
|
fi
|
||||||
;;
|
;;
|
||||||
rocshmem-rocm-ci)
|
rocshmem-rocm-ci)
|
||||||
@@ -1601,7 +1815,7 @@ add_dependency_cache_target() {
|
|||||||
|
|
||||||
resolve_ci_base_dependency_targets() {
|
resolve_ci_base_dependency_targets() {
|
||||||
local mode="${ROCM_DEP_CACHE_EXPORT_MODE:-missing}"
|
local mode="${ROCM_DEP_CACHE_EXPORT_MODE:-missing}"
|
||||||
local rixl_ref=""
|
local nixl_ref=""
|
||||||
local rocshmem_ref=""
|
local rocshmem_ref=""
|
||||||
local deepep_ref=""
|
local deepep_ref=""
|
||||||
|
|
||||||
@@ -1610,7 +1824,7 @@ resolve_ci_base_dependency_targets() {
|
|||||||
case "${mode}" in
|
case "${mode}" in
|
||||||
always)
|
always)
|
||||||
echo "ROCM_DEP_CACHE_EXPORT_MODE=always; exporting all dependency caches serially"
|
echo "ROCM_DEP_CACHE_EXPORT_MODE=always; exporting all dependency caches serially"
|
||||||
for target in rixl-rocm-ci rocshmem-rocm-ci deepep-rocm-ci; do
|
for target in nixl-rocm-ci rocshmem-rocm-ci deepep-rocm-ci; do
|
||||||
if [[ -n "$(dependency_cache_ref_for_target "${target}")" ]]; then
|
if [[ -n "$(dependency_cache_ref_for_target "${target}")" ]]; then
|
||||||
add_dependency_cache_target "${target}"
|
add_dependency_cache_target "${target}"
|
||||||
fi
|
fi
|
||||||
@@ -1630,13 +1844,13 @@ resolve_ci_base_dependency_targets() {
|
|||||||
;;
|
;;
|
||||||
esac
|
esac
|
||||||
|
|
||||||
if [[ "${mode}" != "always" && -n "${RIXL_CACHE_KEY:-}" ]]; then
|
if [[ "${mode}" != "always" && -n "${NIXL_CACHE_KEY:-}" ]]; then
|
||||||
rixl_ref=$(dependency_cache_ref_for_target "rixl-rocm-ci")
|
nixl_ref=$(dependency_cache_ref_for_target "nixl-rocm-ci")
|
||||||
if dependency_cache_ref_exists "${rixl_ref}"; then
|
if dependency_cache_ref_exists "${nixl_ref}"; then
|
||||||
echo "RIXL dependency cache exists: ${rixl_ref}"
|
echo "NIXL dependency cache exists: ${nixl_ref}"
|
||||||
else
|
else
|
||||||
echo "RIXL dependency cache missing; will seed: ${rixl_ref}"
|
echo "NIXL dependency cache missing; will seed: ${nixl_ref}"
|
||||||
add_dependency_cache_target "rixl-rocm-ci"
|
add_dependency_cache_target "nixl-rocm-ci"
|
||||||
fi
|
fi
|
||||||
fi
|
fi
|
||||||
|
|
||||||
@@ -1736,8 +1950,8 @@ confirm_remote_image_push() {
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
if [[ -z "${remote_revision}" \
|
if [[ -z "${remote_revision}" \
|
||||||
&& ${IMAGE_EXISTED_BEFORE_BUILD} -eq 0 \
|
&& ${IMAGE_EXISTED_BEFORE_BUILD} -eq 0 ]] \
|
||||||
&& image_tag_is_commit_scoped ]]; then
|
&& image_tag_is_commit_scoped; then
|
||||||
echo "Remote image exists under a commit-scoped tag; accepting push despite missing revision label."
|
echo "Remote image exists under a commit-scoped tag; accepting push despite missing revision label."
|
||||||
return 0
|
return 0
|
||||||
fi
|
fi
|
||||||
@@ -1867,36 +2081,57 @@ upload_wheel_artifacts_if_present() {
|
|||||||
local wheel_dir="./wheel-export"
|
local wheel_dir="./wheel-export"
|
||||||
local artifact_dir="artifacts/vllm-rocm-install"
|
local artifact_dir="artifacts/vllm-rocm-install"
|
||||||
local archive_name="vllm-rocm-install.tar.gz"
|
local archive_name="vllm-rocm-install.tar.gz"
|
||||||
|
local metadata_dir="${wheel_dir}/.vllm-ci-artifact"
|
||||||
|
local native_base_image=""
|
||||||
local whl=""
|
local whl=""
|
||||||
local whl_name=""
|
local whl_name=""
|
||||||
|
local -a wheels=()
|
||||||
|
|
||||||
if ! should_upload_wheel_artifacts; then
|
if ! should_upload_wheel_artifacts; then
|
||||||
return 0
|
return 0
|
||||||
fi
|
fi
|
||||||
|
|
||||||
if [[ ! -d "${wheel_dir}" ]] || ! ls "${wheel_dir}"/*.whl >/dev/null 2>&1; then
|
if [[ -d "${wheel_dir}" ]]; then
|
||||||
echo "No ROCm wheel artifacts found in ${wheel_dir}"
|
mapfile -t wheels < <(find "${wheel_dir}" -maxdepth 1 -type f -name '*.whl' -print)
|
||||||
return 0
|
fi
|
||||||
|
if [[ ${#wheels[@]} -ne 1 ]]; then
|
||||||
|
echo "Expected exactly one ROCm wheel in ${wheel_dir}; found ${#wheels[@]}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
whl="${wheels[0]}"
|
||||||
|
whl_name=$(basename "${whl}")
|
||||||
|
native_base_image="${CI_BASE_IMAGE_TAG_COMMIT_REF:-${CI_BASE_IMAGE:-}}"
|
||||||
|
if [[ -z "${native_base_image}" ]]; then
|
||||||
|
echo "Native ROCm artifact requires a ci_base image reference" >&2
|
||||||
|
return 1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
echo "--- :package: Uploading ROCm vLLM install artifact"
|
echo "--- :package: Uploading ROCm vLLM install artifact"
|
||||||
mkdir -p "${artifact_dir}"
|
rm -rf "${artifact_dir}" "${metadata_dir}"
|
||||||
|
mkdir -p "${artifact_dir}" "${metadata_dir}"
|
||||||
|
|
||||||
|
printf '%s\n' "${BUILDKITE_COMMIT:-local}" > "${metadata_dir}/commit.txt"
|
||||||
|
printf '%s\n' "${native_base_image}" > "${metadata_dir}/native-base-image.txt"
|
||||||
|
printf '%s\n' "${CI_BASE_IMAGE:-}" > "${metadata_dir}/ci-base-image.txt"
|
||||||
|
printf '%s\n' "${IMAGE_TAG:-}" > "${metadata_dir}/fallback-image.txt"
|
||||||
|
printf '%s\n' "${whl_name}" > "${metadata_dir}/wheel-filename.txt"
|
||||||
|
|
||||||
tar -C "${wheel_dir}" -czf "${artifact_dir}/${archive_name}" .
|
tar -C "${wheel_dir}" -czf "${artifact_dir}/${archive_name}" .
|
||||||
|
(
|
||||||
|
cd "${artifact_dir}"
|
||||||
|
sha256sum "${archive_name}" > "${archive_name}.sha256"
|
||||||
|
)
|
||||||
echo "Created ${archive_name}: $(du -sh "${artifact_dir}/${archive_name}" | cut -f1)"
|
echo "Created ${archive_name}: $(du -sh "${artifact_dir}/${archive_name}" | cut -f1)"
|
||||||
printf '%s\n' "${CI_BASE_IMAGE:-}" > "${artifact_dir}/ci-base-image.txt"
|
cp "${metadata_dir}"/*.txt "${artifact_dir}/"
|
||||||
printf '%s\n' "${IMAGE_TAG:-}" > "${artifact_dir}/fallback-image.txt"
|
cp "${whl}" "${artifact_dir}/${whl_name}"
|
||||||
|
echo "Copied ${whl_name}: $(du -sh "${artifact_dir}/${whl_name}" | cut -f1)"
|
||||||
for whl in "${wheel_dir}"/*.whl; do
|
|
||||||
[[ -f "${whl}" ]] || continue
|
|
||||||
whl_name=$(basename "${whl}")
|
|
||||||
cp "${whl}" "${artifact_dir}/${whl_name}"
|
|
||||||
echo "Copied ${whl_name}: $(du -sh "${artifact_dir}/${whl_name}" | cut -f1)"
|
|
||||||
done
|
|
||||||
|
|
||||||
if command -v buildkite-agent >/dev/null 2>&1; then
|
if command -v buildkite-agent >/dev/null 2>&1; then
|
||||||
buildkite-agent artifact upload "${artifact_dir}/*"
|
buildkite-agent artifact upload "${artifact_dir}/*" || return 1
|
||||||
echo "ROCm vLLM install artifacts uploaded to ${artifact_dir}/"
|
echo "ROCm vLLM install artifacts uploaded to ${artifact_dir}/"
|
||||||
|
elif [[ "${BUILDKITE:-false}" == "true" ]]; then
|
||||||
|
echo "buildkite-agent not found; cannot upload required ROCm artifacts" >&2
|
||||||
|
return 1
|
||||||
else
|
else
|
||||||
echo "Not in Buildkite, skipping artifact upload"
|
echo "Not in Buildkite, skipping artifact upload"
|
||||||
fi
|
fi
|
||||||
@@ -1920,6 +2155,7 @@ main() {
|
|||||||
compute_dependency_cache_keys
|
compute_dependency_cache_keys
|
||||||
write_ci_base_label_override
|
write_ci_base_label_override
|
||||||
compute_rocm_csrc_content_hash_if_needed
|
compute_rocm_csrc_content_hash_if_needed
|
||||||
|
compute_rocm_rust_content_hash_if_needed
|
||||||
write_rocm_cache_override
|
write_rocm_cache_override
|
||||||
resolve_ci_base_dependency_targets
|
resolve_ci_base_dependency_targets
|
||||||
print_bake_config
|
print_bake_config
|
||||||
@@ -1927,6 +2163,11 @@ main() {
|
|||||||
echo "BAKE_PRINT_ONLY=1 set; skipping build"
|
echo "BAKE_PRINT_ONLY=1 set; skipping build"
|
||||||
return 0
|
return 0
|
||||||
fi
|
fi
|
||||||
|
if should_upload_wheel_artifacts; then
|
||||||
|
# wheel-export is an output directory, not a BuildKit cache. Starting
|
||||||
|
# clean prevents a failed/retried export from packaging a stale wheel.
|
||||||
|
rm -rf ./wheel-export
|
||||||
|
fi
|
||||||
seed_dependency_caches_if_needed
|
seed_dependency_caches_if_needed
|
||||||
run_bake
|
run_bake
|
||||||
upload_wheel_artifacts_if_present
|
upload_wheel_artifacts_if_present
|
||||||
|
|||||||
@@ -45,8 +45,10 @@ $PYTHON .buildkite/scripts/generate-nightly-index.py --version "$SUBPATH" --curr
|
|||||||
echo "Uploading indices to $S3_COMMIT_PREFIX"
|
echo "Uploading indices to $S3_COMMIT_PREFIX"
|
||||||
aws s3 cp --recursive "$INDICES_OUTPUT_DIR/" "$S3_COMMIT_PREFIX"
|
aws s3 cp --recursive "$INDICES_OUTPUT_DIR/" "$S3_COMMIT_PREFIX"
|
||||||
|
|
||||||
# copy to /nightly/ only if it is on the main branch and not a PR
|
# copy to /nightly/ only when enabled for a main branch build that is not a PR
|
||||||
if [[ "$BUILDKITE_BRANCH" == "main" && "$BUILDKITE_PULL_REQUEST" == "false" ]]; then
|
if [[ "${UPDATE_NIGHTLY_INDEX:-1}" == "1" && \
|
||||||
|
"$BUILDKITE_BRANCH" == "main" && \
|
||||||
|
"$BUILDKITE_PULL_REQUEST" == "false" ]]; then
|
||||||
echo "Uploading indices to overwrite /nightly/"
|
echo "Uploading indices to overwrite /nightly/"
|
||||||
aws s3 cp --recursive "$INDICES_OUTPUT_DIR/" "s3://$BUCKET/nightly/"
|
aws s3 cp --recursive "$INDICES_OUTPUT_DIR/" "s3://$BUCKET/nightly/"
|
||||||
fi
|
fi
|
||||||
@@ -67,7 +69,7 @@ pure_version="${version%%+*}"
|
|||||||
echo "Pure version (without variant): $pure_version"
|
echo "Pure version (without variant): $pure_version"
|
||||||
|
|
||||||
# re-generate and copy to /<pure_version>/ only if it does not have "dev" in the version
|
# re-generate and copy to /<pure_version>/ only if it does not have "dev" in the version
|
||||||
if [[ "$version" != *"dev"* ]]; then
|
if [[ "${UPDATE_VERSION_INDEX:-1}" == "1" && "$version" != *"dev"* ]]; then
|
||||||
echo "Re-generating indices for /$pure_version/"
|
echo "Re-generating indices for /$pure_version/"
|
||||||
rm -rf "${INDICES_OUTPUT_DIR:?}"
|
rm -rf "${INDICES_OUTPUT_DIR:?}"
|
||||||
mkdir -p "$INDICES_OUTPUT_DIR"
|
mkdir -p "$INDICES_OUTPUT_DIR"
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
|
||||||
# This script runs tests inside the corresponding ROCm docker container.
|
# This script runs ROCm tests either directly in a native CI pod or inside the
|
||||||
# It handles both single-node and multi-node test configurations.
|
# corresponding Docker container. Multi-node tests continue to use Docker.
|
||||||
#
|
#
|
||||||
# Multi-node detection: Instead of matching on fragile group names, we detect
|
# Multi-node detection: Instead of matching on fragile group names, we detect
|
||||||
# multi-node jobs structurally by looking for the bracket command syntax
|
# multi-node jobs structurally by looking for the bracket command syntax
|
||||||
@@ -34,10 +34,27 @@ set -o pipefail
|
|||||||
: "${CLICOLOR_FORCE:=1}"
|
: "${CLICOLOR_FORCE:=1}"
|
||||||
: "${PY_COLORS:=1}"
|
: "${PY_COLORS:=1}"
|
||||||
: "${ROCM_DOCKER_TTY:=1}"
|
: "${ROCM_DOCKER_TTY:=1}"
|
||||||
|
: "${PYTHONFAULTHANDLER:=1}"
|
||||||
|
: "${PYTEST_TIMEOUT:=2400}"
|
||||||
if [[ " ${PYTEST_ADDOPTS:-} " != *" --color"* ]]; then
|
if [[ " ${PYTEST_ADDOPTS:-} " != *" --color"* ]]; then
|
||||||
PYTEST_ADDOPTS="${PYTEST_ADDOPTS:+${PYTEST_ADDOPTS} }--color=yes"
|
PYTEST_ADDOPTS="${PYTEST_ADDOPTS:+${PYTEST_ADDOPTS} }--color=yes"
|
||||||
fi
|
fi
|
||||||
export BUILDKIT_PROGRESS TERM FORCE_COLOR CLICOLOR_FORCE PY_COLORS PYTEST_ADDOPTS ROCM_DOCKER_TTY
|
if [[ " ${PYTEST_ADDOPTS:-} " != *" --durations="* ]]; then
|
||||||
|
PYTEST_ADDOPTS="${PYTEST_ADDOPTS:+${PYTEST_ADDOPTS} }--durations=25"
|
||||||
|
fi
|
||||||
|
if [[ " ${PYTEST_ADDOPTS:-} " != *" --durations-min="* ]]; then
|
||||||
|
PYTEST_ADDOPTS="${PYTEST_ADDOPTS:+${PYTEST_ADDOPTS} }--durations-min=1.0"
|
||||||
|
fi
|
||||||
|
# Dump stacks after 25 minutes, then stop an individual test after 40 minutes.
|
||||||
|
if [[ " ${PYTEST_ADDOPTS:-} " != *" faulthandler_timeout="* ]]; then
|
||||||
|
PYTEST_ADDOPTS="${PYTEST_ADDOPTS:+${PYTEST_ADDOPTS} }-o faulthandler_timeout=1500"
|
||||||
|
fi
|
||||||
|
if [[ " ${PYTEST_ADDOPTS:-} " != *" --timeout-method="* &&
|
||||||
|
" ${PYTEST_ADDOPTS:-} " != *" --timeout-method "* ]]; then
|
||||||
|
PYTEST_ADDOPTS="${PYTEST_ADDOPTS:+${PYTEST_ADDOPTS} }--timeout-method=thread"
|
||||||
|
fi
|
||||||
|
export BUILDKIT_PROGRESS TERM FORCE_COLOR CLICOLOR_FORCE PY_COLORS PYTEST_ADDOPTS PYTEST_TIMEOUT ROCM_DOCKER_TTY
|
||||||
|
export PYTHONFAULTHANDLER
|
||||||
|
|
||||||
# Export Python path for commands that run directly on the host. Containerized
|
# Export Python path for commands that run directly on the host. Containerized
|
||||||
# tests set this to /vllm-workspace below so spawned Python processes do not
|
# tests set this to /vllm-workspace below so spawned Python processes do not
|
||||||
@@ -53,6 +70,28 @@ report_docker_usage() {
|
|||||||
docker system df || true
|
docker system df || true
|
||||||
}
|
}
|
||||||
|
|
||||||
|
clear_ci_orchestration_env() {
|
||||||
|
unset -v \
|
||||||
|
VLLM_TEST_GROUP_NAME \
|
||||||
|
VLLM_CI_REQUIRE_PERSISTENT_HF_CACHE \
|
||||||
|
VLLM_CI_ARTIFACT_STEP \
|
||||||
|
VLLM_TEST_CACHE \
|
||||||
|
VLLM_CI_EXECUTION_MODE \
|
||||||
|
VLLM_CI_WORKSPACE \
|
||||||
|
VLLM_CI_REQUIRE_WORKSPACE_MOUNT \
|
||||||
|
VLLM_TEST_COMMANDS \
|
||||||
|
VLLM_CI_BRANCH \
|
||||||
|
VLLM_CI_BASE_IMAGE \
|
||||||
|
VLLM_CI_FALLBACK_IMAGE \
|
||||||
|
VLLM_CI_DOCKER_DISABLED \
|
||||||
|
VLLM_CI_ARTIFACT_GLOB \
|
||||||
|
VLLM_CI_ARTIFACT_CHECKSUM_GLOB \
|
||||||
|
VLLM_CI_EXPECTED_GPU_COUNT \
|
||||||
|
VLLM_CI_USE_ARTIFACTS \
|
||||||
|
VLLM_CI_RESULTS_ROOT \
|
||||||
|
VLLM_ALLOW_DEPRECATED_BEAM_SEARCH
|
||||||
|
}
|
||||||
|
|
||||||
cleanup_network() {
|
cleanup_network() {
|
||||||
local max_nodes=${NUM_NODES:-2}
|
local max_nodes=${NUM_NODES:-2}
|
||||||
for node in $(seq 0 $((max_nodes - 1))); do
|
for node in $(seq 0 $((max_nodes - 1))); do
|
||||||
@@ -145,7 +184,11 @@ prepare_artifact_image() {
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
cp "${wheel_dir}"/*.whl "${context_dir}/wheels/" || return 1
|
cp "${wheel_dir}"/*.whl "${context_dir}/wheels/" || return 1
|
||||||
tar -C "${wheel_dir}" --exclude='*.whl' -cf - . \
|
tar -C "${wheel_dir}" \
|
||||||
|
--exclude='*.whl' \
|
||||||
|
--exclude='.vllm-ci-artifact' \
|
||||||
|
--exclude='./.vllm-ci-artifact' \
|
||||||
|
-cf - . \
|
||||||
| tar -C "${workspace_dir}" -xf - || return 1
|
| tar -C "${workspace_dir}" -xf - || return 1
|
||||||
cat > "${context_dir}/Dockerfile" <<'EOF'
|
cat > "${context_dir}/Dockerfile" <<'EOF'
|
||||||
ARG BASE_IMAGE
|
ARG BASE_IMAGE
|
||||||
@@ -168,6 +211,276 @@ EOF
|
|||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
|
is_native_runtime() {
|
||||||
|
[[ "${AMD_CI_RUNTIME:-}" == "native" || "${NATIVE_CI:-}" == "true" ]]
|
||||||
|
}
|
||||||
|
|
||||||
|
validate_native_workspace() {
|
||||||
|
local workspace_dir="${VLLM_CI_WORKSPACE:-/vllm-workspace}"
|
||||||
|
local workspace_real=""
|
||||||
|
local checkout_real=""
|
||||||
|
local workspace_mount=""
|
||||||
|
|
||||||
|
mkdir -p "${workspace_dir}" || return 1
|
||||||
|
workspace_real=$(readlink -m "${workspace_dir}") || return 1
|
||||||
|
if [[ -n "${BUILDKITE_BUILD_CHECKOUT_PATH:-}" ]]; then
|
||||||
|
checkout_real=$(readlink -m "${BUILDKITE_BUILD_CHECKOUT_PATH}") || return 1
|
||||||
|
if [[ "${checkout_real}" == "${workspace_real}" \
|
||||||
|
|| "${checkout_real}" == "${workspace_real}/"* \
|
||||||
|
|| "${workspace_real}" == "${checkout_real}/"* ]]; then
|
||||||
|
echo "Refusing to replace ${workspace_real}; it overlaps the Buildkite checkout ${checkout_real}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
if [[ "${VLLM_CI_REQUIRE_WORKSPACE_MOUNT:-1}" == "1" ]]; then
|
||||||
|
if ! command -v findmnt >/dev/null 2>&1; then
|
||||||
|
echo "findmnt is required to verify the native workspace mount" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
workspace_mount=$(findmnt -n -T "${workspace_real}" -o TARGET 2>/dev/null || true)
|
||||||
|
if [[ "$(readlink -m "${workspace_mount:-/}")" != "${workspace_real}" ]]; then
|
||||||
|
echo "Native CI requires a dedicated volume mounted at ${workspace_real}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
prepare_native_workspace() {
|
||||||
|
if [[ "${VLLM_CI_USE_ARTIFACTS:-0}" != "1" ]]; then
|
||||||
|
echo "Native CI requires VLLM_CI_USE_ARTIFACTS=1"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
if ! command -v buildkite-agent >/dev/null 2>&1; then
|
||||||
|
echo "buildkite-agent not found; cannot download ROCm wheel artifact"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
validate_native_workspace || return 1
|
||||||
|
|
||||||
|
local artifact_glob="${VLLM_CI_ARTIFACT_GLOB:-artifacts/vllm-rocm-install/vllm-rocm-install.tar.gz}"
|
||||||
|
local artifact_checksum_glob="${VLLM_CI_ARTIFACT_CHECKSUM_GLOB:-${artifact_glob}.sha256}"
|
||||||
|
local artifact_step="${VLLM_CI_ARTIFACT_STEP:-image-build-amd}"
|
||||||
|
local archive=""
|
||||||
|
local checksum=""
|
||||||
|
local download_dir=""
|
||||||
|
local metadata_dir=""
|
||||||
|
local recorded_base=""
|
||||||
|
local recorded_commit=""
|
||||||
|
local recorded_wheel=""
|
||||||
|
local workspace_dir="${VLLM_CI_WORKSPACE:-/vllm-workspace}"
|
||||||
|
local wheel_dir=""
|
||||||
|
local attempt=0
|
||||||
|
local attempt_dir=""
|
||||||
|
local -a archives=()
|
||||||
|
local -a checksums=()
|
||||||
|
local -a wheels=()
|
||||||
|
|
||||||
|
artifact_work_dir=$(mktemp -d -t vllm-rocm-artifact.XXXXXX) || return 1
|
||||||
|
wheel_dir="${artifact_work_dir}/wheels"
|
||||||
|
mkdir -p "${wheel_dir}" || return 1
|
||||||
|
|
||||||
|
echo "--- Downloading ROCm wheel artifact from ${artifact_step} (native in-pod)"
|
||||||
|
for attempt in 1 2 3; do
|
||||||
|
attempt_dir="${artifact_work_dir}/download-${attempt}"
|
||||||
|
rm -rf "${attempt_dir}" || return 1
|
||||||
|
mkdir -p "${attempt_dir}" || return 1
|
||||||
|
if buildkite-agent artifact download \
|
||||||
|
"${artifact_glob}" "${attempt_dir}" --step "${artifact_step}" \
|
||||||
|
&& buildkite-agent artifact download \
|
||||||
|
"${artifact_checksum_glob}" "${attempt_dir}" --step "${artifact_step}"; then
|
||||||
|
download_dir="${attempt_dir}"
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
echo "Artifact download attempt ${attempt}/3 failed"
|
||||||
|
if [[ "${attempt}" -lt 3 ]]; then
|
||||||
|
sleep $((attempt * 2))
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
if [[ -z "${download_dir}" ]]; then
|
||||||
|
echo "Failed to download ${artifact_glob} and ${artifact_checksum_glob} from ${artifact_step}"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
mapfile -t archives < <(
|
||||||
|
find "${download_dir}" -name "vllm-rocm-install.tar.gz" -type f -print
|
||||||
|
)
|
||||||
|
mapfile -t checksums < <(
|
||||||
|
find "${download_dir}" -name "vllm-rocm-install.tar.gz.sha256" -type f -print
|
||||||
|
)
|
||||||
|
if [[ ${#archives[@]} -ne 1 || ${#checksums[@]} -ne 1 ]]; then
|
||||||
|
echo "Expected exactly one ROCm archive and checksum; found ${#archives[@]} archive(s) and ${#checksums[@]} checksum(s)" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
archive="${archives[0]}"
|
||||||
|
checksum="${checksums[0]}"
|
||||||
|
if [[ "$(dirname "${archive}")" != "$(dirname "${checksum}")" ]]; then
|
||||||
|
echo "ROCm archive and checksum were downloaded to different directories" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
(
|
||||||
|
cd "$(dirname "${archive}")"
|
||||||
|
sha256sum -c "$(basename "${checksum}")"
|
||||||
|
) || return 1
|
||||||
|
|
||||||
|
tar --no-same-owner -xzf "${archive}" -C "${wheel_dir}" || return 1
|
||||||
|
mapfile -t wheels < <(
|
||||||
|
find "${wheel_dir}" -maxdepth 1 -type f -name '*.whl' -print
|
||||||
|
)
|
||||||
|
if [[ ${#wheels[@]} -ne 1 ]]; then
|
||||||
|
echo "ROCm artifact must contain exactly one top-level wheel; found ${#wheels[@]}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
metadata_dir="${wheel_dir}/.vllm-ci-artifact"
|
||||||
|
for metadata_file in commit.txt native-base-image.txt wheel-filename.txt; do
|
||||||
|
if [[ ! -s "${metadata_dir}/${metadata_file}" ]]; then
|
||||||
|
echo "ROCm artifact metadata is missing ${metadata_file}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
for metadata_file in ci-base-image.txt fallback-image.txt; do
|
||||||
|
if [[ ! -f "${metadata_dir}/${metadata_file}" ]]; then
|
||||||
|
echo "ROCm artifact metadata is missing ${metadata_file}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
recorded_commit=$(tr -d '\r\n' < "${metadata_dir}/commit.txt")
|
||||||
|
recorded_base=$(tr -d '\r\n' < "${metadata_dir}/native-base-image.txt")
|
||||||
|
recorded_wheel=$(tr -d '\r\n' < "${metadata_dir}/wheel-filename.txt")
|
||||||
|
if [[ -z "${BUILDKITE_COMMIT:-}" || "${recorded_commit}" != "${BUILDKITE_COMMIT}" ]]; then
|
||||||
|
echo "ROCm artifact commit ${recorded_commit} does not match ${BUILDKITE_COMMIT:-unset}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
if [[ -z "${VLLM_CI_BASE_IMAGE:-}" || "${recorded_base}" != "${VLLM_CI_BASE_IMAGE}" ]]; then
|
||||||
|
echo "ROCm artifact base ${recorded_base} does not match ${VLLM_CI_BASE_IMAGE:-unset}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
if [[ "${recorded_wheel}" != "$(basename "${wheels[0]}")" ]]; then
|
||||||
|
echo "ROCm artifact wheel manifest ${recorded_wheel} does not match $(basename "${wheels[0]}")" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
for required_dir in tests .buildkite requirements; do
|
||||||
|
if [[ ! -d "${wheel_dir}/${required_dir}" ]]; then
|
||||||
|
echo "ROCm wheel artifact did not contain ${required_dir}/" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
echo "--- Installing ROCm wheel into pod environment"
|
||||||
|
python3 -m pip install --no-deps --force-reinstall "${wheels[0]}" || return 1
|
||||||
|
|
||||||
|
echo "--- Preparing ${workspace_dir} from artifact"
|
||||||
|
find "${workspace_dir}" -mindepth 1 -maxdepth 1 -exec rm -rf -- {} + || return 1
|
||||||
|
tar -C "${wheel_dir}" \
|
||||||
|
--exclude='*.whl' \
|
||||||
|
--exclude='.vllm-ci-artifact' \
|
||||||
|
--exclude='./.vllm-ci-artifact' \
|
||||||
|
-cf - . | tar --no-same-owner -C "${workspace_dir}" -xf - || return 1
|
||||||
|
if [[ ! -d "${workspace_dir}/tests" ]]; then
|
||||||
|
echo "Failed to stage the native test workspace" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
initialize_native_environment() {
|
||||||
|
local job_id="${BUILDKITE_JOB_ID:-${BUILDKITE_PARALLEL_JOB:-local}}"
|
||||||
|
local job_id_suffix=""
|
||||||
|
local native_root=""
|
||||||
|
local hf_fstype=""
|
||||||
|
local hf_mount=""
|
||||||
|
|
||||||
|
if [[ "$(id -u)" -ne 0 ]]; then
|
||||||
|
echo "Native ROCm CI currently requires the ci_base container to run as root" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
job_id="${job_id//[^A-Za-z0-9_.-]/_}"
|
||||||
|
job_id_suffix="${job_id##*-}"
|
||||||
|
job_id_suffix="${job_id_suffix:0:12}"
|
||||||
|
native_root="/tmp/vllm-native-${job_id}"
|
||||||
|
TMPDIR="/tmp/vllm-${job_id_suffix}/tmp"
|
||||||
|
VLLM_RPC_BASE_PATH="/tmp"
|
||||||
|
TORCHINDUCTOR_CACHE_DIR="${native_root}/cache/torchinductor"
|
||||||
|
TRITON_CACHE_DIR="${native_root}/cache/triton"
|
||||||
|
VLLM_CACHE_ROOT="${native_root}/cache/vllm"
|
||||||
|
XDG_CACHE_HOME="${native_root}/cache/xdg"
|
||||||
|
: "${HF_HOME:=/home/buildkite-agent/huggingface}"
|
||||||
|
# datasets uses POSIX locks that are unsupported by the shared HF NFS cache.
|
||||||
|
# Keep processed datasets job-local while retaining the persistent Hub cache.
|
||||||
|
HF_DATASETS_CACHE="${native_root}/cache/huggingface/datasets"
|
||||||
|
: "${HF_HUB_DOWNLOAD_TIMEOUT:=300}"
|
||||||
|
: "${HF_HUB_ETAG_TIMEOUT:=60}"
|
||||||
|
export TMPDIR VLLM_RPC_BASE_PATH
|
||||||
|
export TORCHINDUCTOR_CACHE_DIR TRITON_CACHE_DIR VLLM_CACHE_ROOT XDG_CACHE_HOME
|
||||||
|
export HF_HOME HF_DATASETS_CACHE HF_HUB_DOWNLOAD_TIMEOUT HF_HUB_ETAG_TIMEOUT
|
||||||
|
export PYTORCH_ROCM_ARCH=""
|
||||||
|
|
||||||
|
mkdir -p "${TMPDIR}" \
|
||||||
|
"${TORCHINDUCTOR_CACHE_DIR}" \
|
||||||
|
"${TRITON_CACHE_DIR}" \
|
||||||
|
"${VLLM_CACHE_ROOT}" \
|
||||||
|
"${XDG_CACHE_HOME}" \
|
||||||
|
"${HF_HOME}" \
|
||||||
|
"${HF_DATASETS_CACHE}" || return 1
|
||||||
|
|
||||||
|
echo "Native compile caches: VLLM_CACHE_ROOT=${VLLM_CACHE_ROOT} TORCHINDUCTOR_CACHE_DIR=${TORCHINDUCTOR_CACHE_DIR}"
|
||||||
|
|
||||||
|
if [[ "${VLLM_CI_REQUIRE_PERSISTENT_HF_CACHE:-0}" == "1" ]]; then
|
||||||
|
if ! command -v findmnt >/dev/null 2>&1; then
|
||||||
|
echo "findmnt is required to verify the native Hugging Face cache mount" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
hf_mount=$(findmnt -n -T "${HF_HOME}" -o TARGET 2>/dev/null || true)
|
||||||
|
if [[ -z "${hf_mount}" || "${hf_mount}" == "/" ]]; then
|
||||||
|
echo "Native CI requires a persistent volume mounted at or above ${HF_HOME}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
if command -v findmnt >/dev/null 2>&1; then
|
||||||
|
hf_fstype=$(findmnt -n -T "${HF_HOME}" -o FSTYPE 2>/dev/null || true)
|
||||||
|
fi
|
||||||
|
if [[ "${hf_fstype}" == nfs || "${hf_fstype}" == nfs4 ]]; then
|
||||||
|
# Keep hf-xet state local and avoid vectored writes on shared NFS.
|
||||||
|
export HF_XET_CACHE="${native_root}/cache/hf-xet"
|
||||||
|
export HF_XET_HIGH_PERFORMANCE=0
|
||||||
|
export HF_XET_RECONSTRUCTION_USE_VECTORED_WRITE=0
|
||||||
|
mkdir -p "${HF_XET_CACHE}" || return 1
|
||||||
|
echo "Configured hf-xet for shared ${hf_fstype} cache at ${HF_HOME}"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
run_native_preflight() {
|
||||||
|
local expected_gpus="${VLLM_CI_EXPECTED_GPU_COUNT:-1}"
|
||||||
|
|
||||||
|
if [[ ! "${expected_gpus}" =~ ^[0-9]+$ ]]; then
|
||||||
|
echo "Invalid VLLM_CI_EXPECTED_GPU_COUNT=${expected_gpus}" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
python3 -c "import encodings, importlib.metadata as im, importlib.util as iu; [im.version(d) for d in ('transformers', 'torch', 'ray', 'sympy', 'markupsafe', 'vllm')]; missing=[m for m in ('torch.utils.model_zoo', 'transformers.models.nomic_bert', 'ray.dag', 'sympy.physics', 'markupsafe._speedups') if iu.find_spec(m) is None]; assert not missing, missing" || return 1
|
||||||
|
|
||||||
|
if [[ "${expected_gpus}" == "0" ]]; then
|
||||||
|
echo "Native CPU-only AMD job: skipping ROCm device validation"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "--- ROCm info"
|
||||||
|
rocminfo || return 1
|
||||||
|
VLLM_CI_EXPECTED_GPU_COUNT="${expected_gpus}" python3 - <<'PY'
|
||||||
|
import os
|
||||||
|
|
||||||
|
import torch
|
||||||
|
|
||||||
|
expected = int(os.environ["VLLM_CI_EXPECTED_GPU_COUNT"])
|
||||||
|
assert torch.version.hip, "PyTorch is not a ROCm build"
|
||||||
|
assert torch.cuda.is_available(), "ROCm GPU is not available to PyTorch"
|
||||||
|
actual = torch.cuda.device_count()
|
||||||
|
assert actual == expected, f"Expected {expected} ROCm GPU(s), found {actual}"
|
||||||
|
PY
|
||||||
|
}
|
||||||
|
|
||||||
is_multi_node() {
|
is_multi_node() {
|
||||||
local cmds="$1"
|
local cmds="$1"
|
||||||
# Primary signal: NUM_NODES environment variable set by the pipeline
|
# Primary signal: NUM_NODES environment variable set by the pipeline
|
||||||
@@ -350,7 +663,58 @@ re_quote_pytest_markers() {
|
|||||||
# Main
|
# Main
|
||||||
###############################################################################
|
###############################################################################
|
||||||
|
|
||||||
# --- GPU initialization ---
|
if is_native_runtime; then
|
||||||
|
echo "--- Native in-pod ROCm CI (AMD_CI_RUNTIME=${AMD_CI_RUNTIME:-unset}, NATIVE_CI=${NATIVE_CI:-unset})"
|
||||||
|
artifact_work_dir=""
|
||||||
|
|
||||||
|
cleanup_native_workspace() {
|
||||||
|
if [[ -n "${artifact_work_dir}" ]]; then
|
||||||
|
rm -rf "${artifact_work_dir}"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
trap cleanup_native_workspace EXIT
|
||||||
|
|
||||||
|
if [[ -n "${VLLM_TEST_COMMANDS:-}" ]]; then
|
||||||
|
commands="${VLLM_TEST_COMMANDS}"
|
||||||
|
commands_source="env"
|
||||||
|
else
|
||||||
|
commands="$*"
|
||||||
|
commands_source="argv"
|
||||||
|
if [[ -z "$commands" ]]; then
|
||||||
|
echo "Error: No test commands provided for native CI." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ "$commands_source" == "argv" ]]; then
|
||||||
|
commands=$(re_quote_pytest_markers "$commands")
|
||||||
|
fi
|
||||||
|
|
||||||
|
if is_multi_node "$commands"; then
|
||||||
|
echo "Native CI does not support multi-node jobs yet."
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if ! initialize_native_environment; then
|
||||||
|
echo "Failed to initialize the native test environment"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
if ! prepare_native_workspace; then
|
||||||
|
echo "Failed to prepare native test workspace"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
export PYTHONPATH="${VLLM_CI_WORKSPACE:-/vllm-workspace}"
|
||||||
|
|
||||||
|
echo "Native test commands: $commands"
|
||||||
|
run_native_preflight || exit 1
|
||||||
|
# Keep AMD CI orchestration variables out of vLLM's runtime environment.
|
||||||
|
clear_ci_orchestration_env
|
||||||
|
/bin/bash -o pipefail -c "${commands}"
|
||||||
|
handle_pytest_exit "$?"
|
||||||
|
fi
|
||||||
|
|
||||||
|
# --- GPU initialization for legacy Docker execution ---
|
||||||
echo "--- ROCm info"
|
echo "--- ROCm info"
|
||||||
rocminfo
|
rocminfo
|
||||||
|
|
||||||
@@ -452,25 +816,27 @@ fi
|
|||||||
|
|
||||||
echo "Final commands: $commands"
|
echo "Final commands: $commands"
|
||||||
|
|
||||||
# The ROCm test image often ships /vllm-workspace without .git (artifact tarball unpack).
|
standalone_merge_base_env=()
|
||||||
# tests/standalone_tests/python_only_compile.sh uses merge-base(HEAD, origin/main) for
|
if [[ "$commands" == *python_only_compile.sh* ]]; then
|
||||||
# wheels.vllm.ai; compute on the agent (full git checkout) and pass into the container.
|
# The ROCm test image often ships /vllm-workspace without .git. Resolve the
|
||||||
vllm_standalone_merge_base=""
|
# wheels.vllm.ai commit from the agent checkout for this test only.
|
||||||
checkout="${BUILDKITE_BUILD_CHECKOUT_PATH:-}"
|
vllm_standalone_merge_base=""
|
||||||
if [[ -z "${checkout}" || ! -d "${checkout}" ]]; then
|
checkout="${BUILDKITE_BUILD_CHECKOUT_PATH:-}"
|
||||||
checkout="."
|
if [[ -z "${checkout}" || ! -d "${checkout}" ]]; then
|
||||||
|
checkout="."
|
||||||
|
fi
|
||||||
|
# Pass safe.directory per-command because Buildkite uses mixed user IDs.
|
||||||
|
if git -c "safe.directory=${checkout}" -C "${checkout}" rev-parse --is-inside-work-tree >/dev/null 2>&1; then
|
||||||
|
vllm_standalone_merge_base="$(
|
||||||
|
git -c "safe.directory=${checkout}" -C "${checkout}" merge-base HEAD origin/main 2>/dev/null || true
|
||||||
|
)"
|
||||||
|
fi
|
||||||
|
if [[ -z "${vllm_standalone_merge_base}" ]]; then
|
||||||
|
vllm_standalone_merge_base="${BUILDKITE_COMMIT:-}"
|
||||||
|
fi
|
||||||
|
echo "INFO: passing CI_STANDALONE_MERGE_BASE into container: ${vllm_standalone_merge_base}"
|
||||||
|
standalone_merge_base_env=(-e "CI_STANDALONE_MERGE_BASE=${vllm_standalone_merge_base}")
|
||||||
fi
|
fi
|
||||||
# Pass safe.directory per-command (-c) because buildkite runs will always fail
|
|
||||||
# the next check on git 2.35.2+ due to mixed uses of root and buildkite-agent/uids.
|
|
||||||
if git -c "safe.directory=${checkout}" -C "${checkout}" rev-parse --is-inside-work-tree >/dev/null 2>&1; then
|
|
||||||
vllm_standalone_merge_base="$(
|
|
||||||
git -c "safe.directory=${checkout}" -C "${checkout}" merge-base HEAD origin/main 2>/dev/null || true
|
|
||||||
)"
|
|
||||||
fi
|
|
||||||
if [[ -z "${vllm_standalone_merge_base}" ]]; then
|
|
||||||
vllm_standalone_merge_base="${BUILDKITE_COMMIT:-}"
|
|
||||||
fi
|
|
||||||
echo "INFO: passing VLLM_STANDALONE_MERGE_BASE into container: ${vllm_standalone_merge_base}"
|
|
||||||
|
|
||||||
MYPYTHONPATH="/vllm-workspace"
|
MYPYTHONPATH="/vllm-workspace"
|
||||||
|
|
||||||
@@ -501,6 +867,7 @@ else
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
# --- Route: multi-node vs single-node ---
|
# --- Route: multi-node vs single-node ---
|
||||||
|
clear_ci_orchestration_env
|
||||||
if is_multi_node "$commands"; then
|
if is_multi_node "$commands"; then
|
||||||
echo "--- Multi-node job detected"
|
echo "--- Multi-node job detected"
|
||||||
export DCKR_VER=$(docker --version | sed 's/Docker version \(.*\), build .*/\1/')
|
export DCKR_VER=$(docker --version | sed 's/Docker version \(.*\), build .*/\1/')
|
||||||
@@ -589,7 +956,9 @@ else
|
|||||||
-e FORCE_COLOR \
|
-e FORCE_COLOR \
|
||||||
-e CLICOLOR_FORCE \
|
-e CLICOLOR_FORCE \
|
||||||
-e PY_COLORS \
|
-e PY_COLORS \
|
||||||
|
-e PYTHONFAULTHANDLER \
|
||||||
-e PYTEST_ADDOPTS \
|
-e PYTEST_ADDOPTS \
|
||||||
|
-e PYTEST_TIMEOUT \
|
||||||
-v "${HF_CACHE}:${HF_MOUNT}" \
|
-v "${HF_CACHE}:${HF_MOUNT}" \
|
||||||
-e "HF_HOME=${HF_MOUNT}" \
|
-e "HF_HOME=${HF_MOUNT}" \
|
||||||
-e "PYTHONPATH=${MYPYTHONPATH}" \
|
-e "PYTHONPATH=${MYPYTHONPATH}" \
|
||||||
@@ -599,7 +968,7 @@ else
|
|||||||
-e "VLLM_CACHE_ROOT=${CONTAINER_CACHE_ROOT}/vllm" \
|
-e "VLLM_CACHE_ROOT=${CONTAINER_CACHE_ROOT}/vllm" \
|
||||||
-e "XDG_CACHE_HOME=${CONTAINER_CACHE_ROOT}/xdg" \
|
-e "XDG_CACHE_HOME=${CONTAINER_CACHE_ROOT}/xdg" \
|
||||||
-e "PYTORCH_ROCM_ARCH=" \
|
-e "PYTORCH_ROCM_ARCH=" \
|
||||||
-e "VLLM_STANDALONE_MERGE_BASE=${vllm_standalone_merge_base}" \
|
"${standalone_merge_base_env[@]}" \
|
||||||
--name "${container_name}" \
|
--name "${container_name}" \
|
||||||
"${image_name}" \
|
"${image_name}" \
|
||||||
/bin/bash -c "${CONTAINER_PREFLIGHT} && ${commands}"
|
/bin/bash -c "${CONTAINER_PREFLIGHT} && ${commands}"
|
||||||
|
|||||||
@@ -3,8 +3,9 @@ set -euox pipefail
|
|||||||
|
|
||||||
export VLLM_CPU_KVCACHE_SPACE=1
|
export VLLM_CPU_KVCACHE_SPACE=1
|
||||||
export VLLM_CPU_CI_ENV=1
|
export VLLM_CPU_CI_ENV=1
|
||||||
# Reduce sub-processes for acceleration
|
# Skip torch.compile via vLLM's --enforce-eager flag (passed below) instead of
|
||||||
export TORCH_COMPILE_DISABLE=1
|
# TORCH_COMPILE_DISABLE=1, which torch 2.12 no longer treats as a silent no-op
|
||||||
|
# when callers specify fullgraph=True.
|
||||||
export VLLM_ENABLE_V1_MULTIPROCESSING=0
|
export VLLM_ENABLE_V1_MULTIPROCESSING=0
|
||||||
|
|
||||||
SDE_ARCHIVE="sde-external-10.7.0-2026-02-18-lin.tar.xz"
|
SDE_ARCHIVE="sde-external-10.7.0-2026-02-18-lin.tar.xz"
|
||||||
@@ -49,15 +50,15 @@ wait_for_pid_and_check_log() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
# Test Sky Lake (AVX512F)
|
# Test Sky Lake (AVX512F)
|
||||||
./sde/sde64 -skl -- python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --dtype bfloat16 > test_0.log 2>&1 &
|
./sde/sde64 -skl -- python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --dtype bfloat16 --enforce-eager > test_0.log 2>&1 &
|
||||||
PID_TEST_0=$!
|
PID_TEST_0=$!
|
||||||
|
|
||||||
# Test Cascade Lake (AVX512F + VNNI)
|
# Test Cascade Lake (AVX512F + VNNI)
|
||||||
./sde/sde64 -clx -- python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --dtype bfloat16 > test_1.log 2>&1 &
|
./sde/sde64 -clx -- python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --dtype bfloat16 --enforce-eager > test_1.log 2>&1 &
|
||||||
PID_TEST_1=$!
|
PID_TEST_1=$!
|
||||||
|
|
||||||
# Test Cooper Lake (AVX512F + VNNI + BF16)
|
# Test Cooper Lake (AVX512F + VNNI + BF16)
|
||||||
./sde/sde64 -cpx -- python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --dtype bfloat16 > test_2.log 2>&1 &
|
./sde/sde64 -cpx -- python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --dtype bfloat16 --enforce-eager > test_2.log 2>&1 &
|
||||||
PID_TEST_2=$!
|
PID_TEST_2=$!
|
||||||
|
|
||||||
wait_for_pid_and_check_log $PID_TEST_0 test_0.log
|
wait_for_pid_and_check_log $PID_TEST_0 test_0.log
|
||||||
|
|||||||
@@ -40,7 +40,9 @@ function cpu_tests() {
|
|||||||
pytest -x -v -s tests/kernels/moe/test_cpu_fused_moe.py
|
pytest -x -v -s tests/kernels/moe/test_cpu_fused_moe.py
|
||||||
pytest -x -v -s tests/kernels/mamba/cpu/test_cpu_gdn_ops.py
|
pytest -x -v -s tests/kernels/mamba/cpu/test_cpu_gdn_ops.py
|
||||||
pytest -x -v -s tests/kernels/moe/test_cpu_int4_moe.py
|
pytest -x -v -s tests/kernels/moe/test_cpu_int4_moe.py
|
||||||
pytest -x -v -s tests/kernels/mamba/test_cpu_short_conv.py"
|
pytest -x -v -s tests/kernels/mamba/test_cpu_short_conv.py
|
||||||
|
pytest -x -v -s tests/kernels/mamba/test_causal_conv1d.py
|
||||||
|
pytest -x -v -s tests/kernels/mamba/test_mamba_ssm.py"
|
||||||
|
|
||||||
# skip tests requiring model downloads if HF_TOKEN is not set
|
# skip tests requiring model downloads if HF_TOKEN is not set
|
||||||
# due to rate-limits
|
# due to rate-limits
|
||||||
@@ -97,3 +99,4 @@ function cpu_tests() {
|
|||||||
# All of CPU tests are expected to be finished less than 40 mins.
|
# All of CPU tests are expected to be finished less than 40 mins.
|
||||||
export -f cpu_tests
|
export -f cpu_tests
|
||||||
timeout 2h bash -c cpu_tests
|
timeout 2h bash -c cpu_tests
|
||||||
|
|
||||||
|
|||||||
@@ -35,6 +35,7 @@ case "${test_suite}" in
|
|||||||
pytest -v -s v1/worker --ignore=v1/worker/test_gpu_model_runner.py --ignore=v1/worker/test_worker_memory_snapshot.py
|
pytest -v -s v1/worker --ignore=v1/worker/test_gpu_model_runner.py --ignore=v1/worker/test_worker_memory_snapshot.py
|
||||||
pytest -v -s v1/structured_output
|
pytest -v -s v1/structured_output
|
||||||
pytest -v -s v1/test_serial_utils.py
|
pytest -v -s v1/test_serial_utils.py
|
||||||
|
pytest -v -s v1/e2e/general/test_correctness_sliding_window.py --deselect="tests/v1/e2e/general/test_correctness_sliding_window.py::test_sliding_window_retrieval[True-1-5-google/gemma-3-1b-it]"
|
||||||
pytest -v -s v1/spec_decode --ignore=v1/spec_decode/test_max_len.py --ignore=v1/spec_decode/test_speculators_eagle3.py --ignore=v1/spec_decode/test_acceptance_length.py --ignore=v1/spec_decode/test_speculators_correctness.py
|
pytest -v -s v1/spec_decode --ignore=v1/spec_decode/test_max_len.py --ignore=v1/spec_decode/test_speculators_eagle3.py --ignore=v1/spec_decode/test_acceptance_length.py --ignore=v1/spec_decode/test_speculators_correctness.py
|
||||||
pytest -v -s v1/kv_connector/unit --ignore=v1/kv_connector/unit/test_multi_connector.py --ignore=v1/kv_connector/unit/test_example_connector.py --ignore=v1/kv_connector/unit/test_lmcache_integration.py --ignore=v1/kv_connector/unit/test_hf3fs_client.py --ignore=v1/kv_connector/unit/test_hf3fs_connector.py --ignore=v1/kv_connector/unit/test_hf3fs_metadata_server.py --ignore=v1/kv_connector/unit/test_offloading_connector.py
|
pytest -v -s v1/kv_connector/unit --ignore=v1/kv_connector/unit/test_multi_connector.py --ignore=v1/kv_connector/unit/test_example_connector.py --ignore=v1/kv_connector/unit/test_lmcache_integration.py --ignore=v1/kv_connector/unit/test_hf3fs_client.py --ignore=v1/kv_connector/unit/test_hf3fs_connector.py --ignore=v1/kv_connector/unit/test_hf3fs_metadata_server.py --ignore=v1/kv_connector/unit/test_offloading_connector.py
|
||||||
;;
|
;;
|
||||||
|
|||||||
@@ -369,7 +369,7 @@ export HF_TOKEN ZE_AFFINITY_MASK
|
|||||||
-e CMDS \
|
-e CMDS \
|
||||||
--name "${container_name}" \
|
--name "${container_name}" \
|
||||||
"${IMAGE}" \
|
"${IMAGE}" \
|
||||||
bash -c 'set -e; source /opt/intel/oneapi/setvars.sh --force; source /opt/intel/oneapi/ccl/2021.15/env/vars.sh --force; echo "ZE_AFFINITY_MASK is ${ZE_AFFINITY_MASK:-}"; eval "$CMDS"' \
|
bash -c 'set -e; echo "ZE_AFFINITY_MASK is ${ZE_AFFINITY_MASK:-}"; eval "$CMDS"' \
|
||||||
>/dev/null
|
>/dev/null
|
||||||
} 9>/tmp/docker-pull.lock
|
} 9>/tmp/docker-pull.lock
|
||||||
|
|
||||||
|
|||||||
@@ -13,6 +13,18 @@ metadata_get() {
|
|||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
|
use_ci_base_if_present() {
|
||||||
|
local ci_base_image=""
|
||||||
|
|
||||||
|
ci_base_image="$(metadata_get rocm-ci-base-image)"
|
||||||
|
if [[ -z "${ci_base_image}" ]]; then
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
export CI_BASE_IMAGE="${ci_base_image}"
|
||||||
|
echo "Using ROCm ci_base image selected by the preceding build step: ${CI_BASE_IMAGE}"
|
||||||
|
}
|
||||||
|
|
||||||
use_refreshed_base_if_present() {
|
use_refreshed_base_if_present() {
|
||||||
local base_refreshed=""
|
local base_refreshed=""
|
||||||
|
|
||||||
@@ -22,15 +34,12 @@ use_refreshed_base_if_present() {
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
export BASE_IMAGE
|
export BASE_IMAGE
|
||||||
export CI_BASE_IMAGE
|
|
||||||
export IMAGE_TAG_LATEST
|
export IMAGE_TAG_LATEST
|
||||||
|
|
||||||
BASE_IMAGE="$(metadata_get rocm-base-image)"
|
BASE_IMAGE="$(metadata_get rocm-base-image)"
|
||||||
CI_BASE_IMAGE="$(metadata_get rocm-ci-base-image)"
|
|
||||||
IMAGE_TAG_LATEST="$(metadata_get rocm-ci-image-descriptive)"
|
IMAGE_TAG_LATEST="$(metadata_get rocm-ci-image-descriptive)"
|
||||||
|
|
||||||
echo "Using refreshed ROCm base image for test image: ${BASE_IMAGE}"
|
echo "Using refreshed ROCm base image for test image: ${BASE_IMAGE}"
|
||||||
echo "Using refreshed ROCm ci_base image for test image: ${CI_BASE_IMAGE}"
|
|
||||||
if [[ -n "${IMAGE_TAG_LATEST}" ]]; then
|
if [[ -n "${IMAGE_TAG_LATEST}" ]]; then
|
||||||
echo "Also tagging full ROCm CI image as: ${IMAGE_TAG_LATEST}"
|
echo "Also tagging full ROCm CI image as: ${IMAGE_TAG_LATEST}"
|
||||||
fi
|
fi
|
||||||
@@ -41,6 +50,8 @@ use_refreshed_base_if_present() {
|
|||||||
main() {
|
main() {
|
||||||
local base_refreshed=0
|
local base_refreshed=0
|
||||||
|
|
||||||
|
use_ci_base_if_present || true
|
||||||
|
|
||||||
if use_refreshed_base_if_present; then
|
if use_refreshed_base_if_present; then
|
||||||
base_refreshed=1
|
base_refreshed=1
|
||||||
fi
|
fi
|
||||||
|
|||||||
@@ -6,8 +6,14 @@ set -ex
|
|||||||
# manylinux platform tag with auditwheel.
|
# manylinux platform tag with auditwheel.
|
||||||
# Index generation is handled separately by generate-and-upload-nightly-index.sh.
|
# Index generation is handled separately by generate-and-upload-nightly-index.sh.
|
||||||
|
|
||||||
# shellcheck source=lib/manylinux.sh
|
# auditwheel is Linux-only; macOS wheels already carry a valid tag, so skip the
|
||||||
source .buildkite/scripts/lib/manylinux.sh
|
# manylinux retag for them.
|
||||||
|
WHEEL_PLATFORM="${VLLM_WHEEL_PLATFORM:-linux}"
|
||||||
|
|
||||||
|
if [[ "$WHEEL_PLATFORM" == "linux" ]]; then
|
||||||
|
# shellcheck source=lib/manylinux.sh
|
||||||
|
source .buildkite/scripts/lib/manylinux.sh
|
||||||
|
fi
|
||||||
|
|
||||||
BUCKET="vllm-wheels"
|
BUCKET="vllm-wheels"
|
||||||
SUBPATH=$BUILDKITE_COMMIT
|
SUBPATH=$BUILDKITE_COMMIT
|
||||||
@@ -27,8 +33,10 @@ wheel="${wheel_files[0]}"
|
|||||||
|
|
||||||
# ========= detect manylinux tag and rename ==========
|
# ========= detect manylinux tag and rename ==========
|
||||||
|
|
||||||
wheel="$(apply_manylinux_tag "$wheel")"
|
if [[ "$WHEEL_PLATFORM" == "linux" ]]; then
|
||||||
echo "Renamed wheel to: $wheel"
|
wheel="$(apply_manylinux_tag "$wheel")"
|
||||||
|
echo "Renamed wheel to: $wheel"
|
||||||
|
fi
|
||||||
|
|
||||||
# Extract the version from the wheel
|
# Extract the version from the wheel
|
||||||
version=$(unzip -p "$wheel" '**/METADATA' | grep '^Version: ' | cut -d' ' -f2)
|
version=$(unzip -p "$wheel" '**/METADATA' | grep '^Version: ' | cut -d' ' -f2)
|
||||||
|
|||||||
@@ -113,8 +113,8 @@ $PYTHON .buildkite/scripts/generate-nightly-index.py \
|
|||||||
echo "Uploading indices to $S3_COMMIT_PREFIX"
|
echo "Uploading indices to $S3_COMMIT_PREFIX"
|
||||||
aws s3 cp --recursive "$INDICES_OUTPUT_DIR/" "$S3_COMMIT_PREFIX"
|
aws s3 cp --recursive "$INDICES_OUTPUT_DIR/" "$S3_COMMIT_PREFIX"
|
||||||
|
|
||||||
# Update rocm/nightly/ if on main branch and not a PR
|
# Only scheduled nightly builds should update the moving nightly index.
|
||||||
if [[ "$BUILDKITE_BRANCH" == "main" && "$BUILDKITE_PULL_REQUEST" == "false" ]] || [[ "$NIGHTLY" == "1" ]]; then
|
if [[ "${NIGHTLY:-0}" == "1" ]]; then
|
||||||
echo "Updating rocm/nightly/ index..."
|
echo "Updating rocm/nightly/ index..."
|
||||||
aws s3 cp --recursive "$INDICES_OUTPUT_DIR/" "s3://$BUCKET/rocm/nightly/"
|
aws s3 cp --recursive "$INDICES_OUTPUT_DIR/" "s3://$BUCKET/rocm/nightly/"
|
||||||
fi
|
fi
|
||||||
@@ -147,7 +147,7 @@ echo ""
|
|||||||
echo "Install command (by commit):"
|
echo "Install command (by commit):"
|
||||||
echo " pip install vllm --extra-index-url https://${BUCKET}.s3.amazonaws.com/$ROCM_SUBPATH/"
|
echo " pip install vllm --extra-index-url https://${BUCKET}.s3.amazonaws.com/$ROCM_SUBPATH/"
|
||||||
echo ""
|
echo ""
|
||||||
if [[ "$BUILDKITE_BRANCH" == "main" ]] || [[ "$NIGHTLY" == "1" ]]; then
|
if [[ "${NIGHTLY:-0}" == "1" ]]; then
|
||||||
echo "Install command (nightly):"
|
echo "Install command (nightly):"
|
||||||
echo " pip install vllm --extra-index-url https://${BUCKET}.s3.amazonaws.com/rocm/nightly/"
|
echo " pip install vllm --extra-index-url https://${BUCKET}.s3.amazonaws.com/rocm/nightly/"
|
||||||
fi
|
fi
|
||||||
|
|||||||
+451
-271
File diff suppressed because it is too large
Load Diff
@@ -16,8 +16,9 @@ steps:
|
|||||||
parallelism: 2
|
parallelism: 2
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 95
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 125
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Basic Correctness
|
- label: Basic Correctness
|
||||||
key: basic-correctness
|
key: basic-correctness
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 68
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -18,7 +18,8 @@ steps:
|
|||||||
- pytest -v -s basic_correctness/test_cpu_offload.py
|
- pytest -v -s basic_correctness/test_cpu_offload.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 70
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 60
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Benchmarks CLI Test
|
- label: Benchmarks CLI Test
|
||||||
key: benchmarks-cli-test
|
key: benchmarks-cli-test
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -13,7 +13,9 @@ steps:
|
|||||||
- pytest -v -s benchmarks/
|
- pytest -v -s benchmarks/
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_1
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 40
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
|
|||||||
@@ -16,8 +16,10 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- pytest -v -s cuda/test_cuda_context.py
|
- pytest -v -s cuda/test_cuda_context.py
|
||||||
- pytest -v -s cuda/test_platform_no_cuda_init.py
|
- pytest -v -s cuda/test_platform_no_cuda_init.py
|
||||||
|
- pytest -v -s cuda/test_cuda_compatibility_path.py
|
||||||
|
|
||||||
- label: Cudagraph
|
- label: Cudagraph
|
||||||
|
device: h200_35gb
|
||||||
key: cudagraph
|
key: cudagraph
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -25,7 +27,10 @@ steps:
|
|||||||
- vllm/v1/cudagraph_dispatcher.py
|
- vllm/v1/cudagraph_dispatcher.py
|
||||||
- vllm/config/compilation.py
|
- vllm/config/compilation.py
|
||||||
- vllm/compilation
|
- vllm/compilation
|
||||||
|
- vllm/v1/worker/encoder_cudagraph.py
|
||||||
|
- vllm/v1/worker/encoder_cudagraph_defs.py
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s v1/cudagraph/test_cudagraph_dispatch.py
|
- pytest -v -s v1/cudagraph/test_cudagraph_dispatch.py
|
||||||
- pytest -v -s v1/cudagraph/test_cudagraph_mode.py
|
- pytest -v -s v1/cudagraph/test_cudagraph_mode.py
|
||||||
- pytest -v -s v1/cudagraph/test_breakable_cudagraph.py
|
- pytest -v -s v1/cudagraph/test_breakable_cudagraph.py
|
||||||
|
- pytest -v -s v1/cudagraph/test_encoder_cudagraph.py
|
||||||
|
|||||||
@@ -15,8 +15,9 @@ steps:
|
|||||||
- bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
- bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_4
|
device: mi300_4
|
||||||
timeout_in_minutes: 85
|
timeout_in_minutes: 60
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -65,8 +66,9 @@ steps:
|
|||||||
- DP_EP=1 bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
- DP_EP=1 bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_4
|
device: mi300_4
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 40
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -90,8 +92,9 @@ steps:
|
|||||||
- CROSS_LAYERS_BLOCKS=True bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
- CROSS_LAYERS_BLOCKS=True bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_4
|
device: mi300_4
|
||||||
timeout_in_minutes: 85
|
timeout_in_minutes: 60
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -115,8 +118,9 @@ steps:
|
|||||||
- HYBRID_SSM=1 bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
- HYBRID_SSM=1 bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_4
|
device: mi300_4
|
||||||
timeout_in_minutes: 80
|
timeout_in_minutes: 55
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -127,6 +131,22 @@ steps:
|
|||||||
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||||
- HYBRID_SSM=1 ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
- HYBRID_SSM=1 ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
|
|
||||||
|
- label: NixlConnector PD edge case test (2 GPUs)
|
||||||
|
key: nixlconnector-pd-edge-cases-2-gpus
|
||||||
|
timeout_in_minutes: 40
|
||||||
|
working_dir: "/vllm-workspace/tests"
|
||||||
|
num_devices: 2
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/nixl/
|
||||||
|
- vllm/v1/core/sched/
|
||||||
|
- tests/v1/kv_connector/nixl_integration/
|
||||||
|
env:
|
||||||
|
PREFILL_GPU_ID: "0"
|
||||||
|
DECODE_GPU_ID: "1"
|
||||||
|
commands:
|
||||||
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
|
- bash v1/kv_connector/nixl_integration/run_edge_case_test.sh
|
||||||
|
|
||||||
- label: Hybrid SSM NixlConnector PD prefix cache test (2 GPUs)
|
- label: Hybrid SSM NixlConnector PD prefix cache test (2 GPUs)
|
||||||
key: hybrid-ssm-nixlconnector-pd-prefix-cache-2-gpus
|
key: hybrid-ssm-nixlconnector-pd-prefix-cache-2-gpus
|
||||||
timeout_in_minutes: 25
|
timeout_in_minutes: 25
|
||||||
@@ -171,8 +191,9 @@ steps:
|
|||||||
- bash v1/kv_connector/nixl_integration/config_sweep_spec_decode_test.sh
|
- bash v1/kv_connector/nixl_integration/config_sweep_spec_decode_test.sh
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_2
|
device: mi300_2
|
||||||
timeout_in_minutes: 70
|
timeout_in_minutes: 45
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -182,7 +203,7 @@ steps:
|
|||||||
- vllm/platforms/rocm.py
|
- vllm/platforms/rocm.py
|
||||||
commands:
|
commands:
|
||||||
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||||
- ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_spec_decode_test.sh
|
- KV_CACHE_MEMORY_BYTES=8G ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_spec_decode_test.sh
|
||||||
|
|
||||||
- label: MultiConnector (Nixl+Offloading) PD edge cases (2 GPUs)
|
- label: MultiConnector (Nixl+Offloading) PD edge cases (2 GPUs)
|
||||||
key: multiconnector-nixl-offloading-pd-edge-cases-2-gpus
|
key: multiconnector-nixl-offloading-pd-edge-cases-2-gpus
|
||||||
@@ -198,3 +219,25 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
- bash v1/kv_connector/nixl_integration/run_multi_connector_edge_case_test.sh
|
- bash v1/kv_connector/nixl_integration/run_multi_connector_edge_case_test.sh
|
||||||
|
|
||||||
|
# P TP 4 - D DPEP 4 test case for DSv4-Flash
|
||||||
|
- label: DSv4-Flash Disaggregated DP EP
|
||||||
|
key: dsv4-flash-disaggregated
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
device: h200
|
||||||
|
optional: true
|
||||||
|
working_dir: "/vllm-workspace/tests"
|
||||||
|
num_devices: 8
|
||||||
|
env:
|
||||||
|
ENABLE_HMA_FLAG: "1"
|
||||||
|
DP_EP: "1"
|
||||||
|
GPU_MEMORY_UTILIZATION: "0.85"
|
||||||
|
PREFILLER_TP_SIZE: "4"
|
||||||
|
DECODER_TP_SIZE: "4"
|
||||||
|
PREFILL_BLOCK_SIZE: "256"
|
||||||
|
DECODE_BLOCK_SIZE: "256"
|
||||||
|
MODEL_NAMES: "deepseek-ai/DeepSeek-V4-Flash"
|
||||||
|
VLLM_SERVE_EXTRA_ARGS: "--trust-remote-code,--kv-cache-dtype,fp8"
|
||||||
|
commands:
|
||||||
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
|
- bash v1/kv_connector/nixl_integration/run_accuracy_test.sh
|
||||||
|
|||||||
@@ -39,7 +39,9 @@ steps:
|
|||||||
- DP_SIZE=2 pytest -v -s entrypoints/openai/test_multi_api_servers.py
|
- DP_SIZE=2 pytest -v -s entrypoints/openai/test_multi_api_servers.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_2
|
device: mi300_2
|
||||||
|
timeout_in_minutes: 45
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -28,8 +28,9 @@ steps:
|
|||||||
- pytest -v -s engine test_sequence.py test_config.py test_logger.py test_vllm_port.py test_jit_monitor.py
|
- pytest -v -s engine test_sequence.py test_config.py test_logger.py test_vllm_port.py test_jit_monitor.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 50
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 40
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
@@ -39,19 +40,21 @@ steps:
|
|||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/engine/
|
- vllm/v1/engine/
|
||||||
- tests/v1/engine/
|
- tests/v1/engine/
|
||||||
|
- tests/v1/test_tensor_ipc_queue.py
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s v1/engine/test_preprocess_error_handling.py
|
- pytest -v -s v1/engine/test_preprocess_error_handling.py
|
||||||
- pytest -v -s v1/engine --ignore v1/engine/test_preprocess_error_handling.py
|
- pytest -v -s v1/engine --ignore v1/engine/test_preprocess_error_handling.py
|
||||||
|
- pytest -v -s v1/test_tensor_ipc_queue.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi250_1
|
||||||
timeout_in_minutes: 55
|
timeout_in_minutes: 45
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: e2e Scheduling (1 GPU)
|
- label: e2e Scheduling (1 GPU)
|
||||||
key: e2e-scheduling-1-gpu
|
key: e2e-scheduling-1-gpu
|
||||||
timeout_in_minutes: 35
|
timeout_in_minutes: 53
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/
|
- vllm/v1/
|
||||||
@@ -60,8 +63,8 @@ steps:
|
|||||||
- pytest -v -s v1/e2e/general/test_async_scheduling.py
|
- pytest -v -s v1/e2e/general/test_async_scheduling.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi250_1
|
||||||
timeout_in_minutes: 70
|
timeout_in_minutes: 55
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
@@ -76,8 +79,8 @@ steps:
|
|||||||
- pytest -v -s v1/e2e/general --ignore v1/e2e/general/test_async_scheduling.py
|
- pytest -v -s v1/e2e/general --ignore v1/e2e/general/test_async_scheduling.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi250_1
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 50
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -114,7 +117,9 @@ steps:
|
|||||||
- pytest -v -s v1/e2e/spec_decode/test_spec_decode.py -k "tensor_parallelism"
|
- pytest -v -s v1/e2e/spec_decode/test_spec_decode.py -k "tensor_parallelism"
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_2
|
device: mi300_2
|
||||||
|
timeout_in_minutes: 30
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ depends_on:
|
|||||||
- image-build
|
- image-build
|
||||||
steps:
|
steps:
|
||||||
- label: Entrypoints Unit Tests
|
- label: Entrypoints Unit Tests
|
||||||
|
device: h200_35gb
|
||||||
key: entrypoints-unit-tests
|
key: entrypoints-unit-tests
|
||||||
timeout_in_minutes: 25
|
timeout_in_minutes: 25
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
@@ -15,6 +16,7 @@ steps:
|
|||||||
- pytest -v -s entrypoints/weight_transfer
|
- pytest -v -s entrypoints/weight_transfer
|
||||||
|
|
||||||
- label: Entrypoints Integration (LLM)
|
- label: Entrypoints Integration (LLM)
|
||||||
|
device: h200_35gb
|
||||||
key: entrypoints-integration-llm
|
key: entrypoints-integration-llm
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 60
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
@@ -28,16 +30,16 @@ steps:
|
|||||||
- pytest -v -s entrypoints/llm/offline_mode # Needs to avoid interference with other tests
|
- pytest -v -s entrypoints/llm/offline_mode # Needs to avoid interference with other tests
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
# TODO(akaratza): Test after Torch >= 2.12 bump
|
device: mi300_1
|
||||||
soft_fail: true
|
timeout_in_minutes: 55
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: Entrypoints Integration (API Server)
|
- label: Entrypoints Integration (API Server)
|
||||||
key: entrypoints-integration-api-server
|
key: entrypoints-integration-api-server
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 75
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -50,13 +52,16 @@ steps:
|
|||||||
- pytest -v -s entrypoints/scale_out
|
- pytest -v -s entrypoints/scale_out
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 65
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: Entrypoints Integration (API Server OpenAI - Part 1)
|
- label: Entrypoints Integration (API Server OpenAI - Part 1)
|
||||||
|
device: h200_35gb
|
||||||
key: entrypoints-integration-api-server-openai-part-1
|
key: entrypoints-integration-api-server-openai-part-1
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 68
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -67,14 +72,16 @@ steps:
|
|||||||
- pytest -v -s entrypoints/openai --ignore=entrypoints/openai/completion --ignore=entrypoints/openai/chat_completion --ignore=entrypoints/openai/responses --ignore=entrypoints/openai/correctness
|
- pytest -v -s entrypoints/openai --ignore=entrypoints/openai/completion --ignore=entrypoints/openai/chat_completion --ignore=entrypoints/openai/responses --ignore=entrypoints/openai/correctness
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
timeout_in_minutes: 65
|
timeout_in_minutes: 65
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: Entrypoints Integration (API Server OpenAI - Part 2)
|
- label: Entrypoints Integration (API Server OpenAI - Part 2)
|
||||||
|
device: h200_35gb
|
||||||
key: entrypoints-integration-api-server-openai-part-2
|
key: entrypoints-integration-api-server-openai-part-2
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 83
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -86,12 +93,14 @@ steps:
|
|||||||
- pytest -v -s entrypoints/openai/completion --ignore=entrypoints/openai/completion/test_tensorizer_entrypoint.py
|
- pytest -v -s entrypoints/openai/completion --ignore=entrypoints/openai/completion/test_tensorizer_entrypoint.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 80
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 70
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: Entrypoints Integration (API Server Generate)
|
- label: Entrypoints Integration (API Server Generate)
|
||||||
|
device: h200_35gb
|
||||||
key: entrypoints-integration-api-server-generate
|
key: entrypoints-integration-api-server-generate
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 50
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
@@ -108,12 +117,14 @@ steps:
|
|||||||
- pytest -v -s entrypoints/anthropic
|
- pytest -v -s entrypoints/anthropic
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
timeout_in_minutes: 65
|
timeout_in_minutes: 65
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: Entrypoints Integration (Responses API)
|
- label: Entrypoints Integration (Responses API)
|
||||||
|
device: h200_35gb
|
||||||
key: entrypoints-integration-responses-api
|
key: entrypoints-integration-responses-api
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 50
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
@@ -148,8 +159,9 @@ steps:
|
|||||||
- pytest -v -s entrypoints/multimodal
|
- pytest -v -s entrypoints/multimodal
|
||||||
|
|
||||||
- label: Entrypoints Integration (Pooling)
|
- label: Entrypoints Integration (Pooling)
|
||||||
|
device: h200_35gb
|
||||||
key: entrypoints-integration-pooling
|
key: entrypoints-integration-pooling
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 75
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -169,7 +181,9 @@ steps:
|
|||||||
- pytest -s entrypoints/openai/correctness/
|
- pytest -s entrypoints/openai/correctness/
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 30
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -16,7 +16,9 @@ steps:
|
|||||||
- pytest -v -s distributed/test_eplb_utils.py
|
- pytest -v -s distributed/test_eplb_utils.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_1
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 30
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -50,4 +52,5 @@ steps:
|
|||||||
- vllm/compilation/
|
- vllm/compilation/
|
||||||
- tests/distributed/
|
- tests/distributed/
|
||||||
commands:
|
commands:
|
||||||
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
- pytest -v -s distributed/test_elastic_ep.py
|
- pytest -v -s distributed/test_elastic_ep.py
|
||||||
|
|||||||
@@ -0,0 +1,26 @@
|
|||||||
|
group: Fault Tolerance
|
||||||
|
depends_on:
|
||||||
|
- image-build
|
||||||
|
steps:
|
||||||
|
- label: Fault Tolerance E2E (2xH100)
|
||||||
|
key: fault-tolerance-e2e-2xh100
|
||||||
|
timeout_in_minutes: 35
|
||||||
|
device: h100
|
||||||
|
num_devices: 2
|
||||||
|
working_dir: "/vllm-workspace/tests"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/v1/fault_tolerance/
|
||||||
|
- vllm/v1/worker/sentinel/
|
||||||
|
- vllm/entrypoints/serve/fault_tolerance/
|
||||||
|
- vllm/distributed/elastic_ep/
|
||||||
|
- vllm/distributed/device_communicators/
|
||||||
|
- vllm/v1/engine/
|
||||||
|
- vllm/v1/worker/
|
||||||
|
- tests/v1/fault_tolerance/
|
||||||
|
- tests/v1/distributed/test_external_lb_dp.py
|
||||||
|
commands:
|
||||||
|
# Base image has no nixl; install it or has_nixl_ep() skips the tests.
|
||||||
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
|
# https://github.com/NVIDIA/nccl/issues/1838
|
||||||
|
- export NCCL_CUMEM_HOST_ENABLE=0
|
||||||
|
- pytest -v -s v1/fault_tolerance/test_fault_tolerance_e2e.py
|
||||||
@@ -15,6 +15,7 @@ steps:
|
|||||||
- pytest -v -s tests/kernels/ir
|
- pytest -v -s tests/kernels/ir
|
||||||
|
|
||||||
- label: Kernels Core Operation Test
|
- label: Kernels Core Operation Test
|
||||||
|
device: h200_35gb
|
||||||
key: kernels-core-operation-test
|
key: kernels-core-operation-test
|
||||||
timeout_in_minutes: 120
|
timeout_in_minutes: 120
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -60,9 +61,45 @@ steps:
|
|||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu
|
- csrc/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu
|
||||||
- vllm/models/deepseek_v4/common/ops/
|
- vllm/models/deepseek_v4/common/ops/
|
||||||
|
- vllm/models/deepseek_v4/nvidia/
|
||||||
- tests/kernels/test_fused_deepseek_v4_qnorm_rope_kv_insert.py
|
- tests/kernels/test_fused_deepseek_v4_qnorm_rope_kv_insert.py
|
||||||
|
- tests/models/test_deepseek_v4_mega_moe.py
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s kernels/test_fused_deepseek_v4_*.py
|
- pytest -v -s kernels/test_fused_deepseek_v4_*.py
|
||||||
|
- pytest -v -s models/test_deepseek_v4_mega_moe.py
|
||||||
|
|
||||||
|
# Catch-all for test files at the tests/kernels root. This job collects
|
||||||
|
# the whole root so new files are wired by default.
|
||||||
|
# Files with dedicated jobs elsewhere in this file are excluded via --ignore
|
||||||
|
# (test_kda, test_bf16x3_router_gemm_cutedsl and test_ll_bf16_gemm run in
|
||||||
|
# their own jobs / Kernels (B200)).
|
||||||
|
- label: Kernels Root Misc Test (B200)
|
||||||
|
key: kernels-root-misc-test-b200
|
||||||
|
timeout_in_minutes: 45
|
||||||
|
device: b200-k8s
|
||||||
|
source_file_dependencies:
|
||||||
|
- csrc/
|
||||||
|
- vllm/
|
||||||
|
- tests/kernels/
|
||||||
|
commands:
|
||||||
|
- pytest -v -s kernels/
|
||||||
|
--ignore=kernels/attention
|
||||||
|
--ignore=kernels/core
|
||||||
|
--ignore=kernels/helion
|
||||||
|
--ignore=kernels/ir
|
||||||
|
--ignore=kernels/mamba
|
||||||
|
--ignore=kernels/moe
|
||||||
|
--ignore=kernels/quantization
|
||||||
|
--ignore=kernels/test_concat_mla_q.py
|
||||||
|
--ignore=kernels/test_fused_qk_norm_rope_gate.py
|
||||||
|
--ignore=kernels/test_fused_deepseek_v4_qnorm_rope_kv_insert.py
|
||||||
|
--ignore=kernels/test_top_k_per_row.py
|
||||||
|
--ignore=kernels/test_kda.py
|
||||||
|
--ignore=kernels/test_bf16x3_router_gemm_cutedsl.py
|
||||||
|
--ignore=kernels/test_ll_bf16_gemm.py
|
||||||
|
--ignore=kernels/test_shuffle_rows.py
|
||||||
|
# BROKEN on main, pending kernel fixes (B200):
|
||||||
|
# test_shuffle_rows.py (1: test_shuffle_rows_edge_cases)
|
||||||
|
|
||||||
- label: Kernels Attention Test %N
|
- label: Kernels Attention Test %N
|
||||||
key: kernels-attention-test
|
key: kernels-attention-test
|
||||||
@@ -79,7 +116,8 @@ steps:
|
|||||||
parallelism: 2
|
parallelism: 2
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
timeout_in_minutes: 90
|
timeout_in_minutes: 90
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
@@ -117,7 +155,9 @@ steps:
|
|||||||
parallelism: 2
|
parallelism: 2
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 120
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/quantization/
|
- csrc/quantization/
|
||||||
- vllm/model_executor/layers/quantization
|
- vllm/model_executor/layers/quantization
|
||||||
@@ -147,8 +187,9 @@ steps:
|
|||||||
parallelism: 5
|
parallelism: 5
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 65
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 55
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/quantization/cutlass_w8a8/moe/
|
- csrc/quantization/cutlass_w8a8/moe/
|
||||||
- csrc/moe/
|
- csrc/moe/
|
||||||
@@ -163,6 +204,7 @@ steps:
|
|||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: Kernels Mamba Test
|
- label: Kernels Mamba Test
|
||||||
|
device: h200_35gb
|
||||||
key: kernels-mamba-test
|
key: kernels-mamba-test
|
||||||
timeout_in_minutes: 40
|
timeout_in_minutes: 40
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -172,17 +214,6 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- pytest -v -s kernels/mamba
|
- pytest -v -s kernels/mamba
|
||||||
|
|
||||||
- label: Kernels KDA Test
|
|
||||||
timeout_in_minutes: 25
|
|
||||||
device: h200_18gb
|
|
||||||
source_file_dependencies:
|
|
||||||
- vllm/model_executor/layers/fla/ops/kda.py
|
|
||||||
- vllm/model_executor/layers/fla/ops/chunk_delta_h.py
|
|
||||||
- vllm/model_executor/layers/fla/ops/l2norm.py
|
|
||||||
- tests/kernels/test_kda.py
|
|
||||||
commands:
|
|
||||||
- pytest -v -s kernels/test_kda.py
|
|
||||||
|
|
||||||
- label: Kernels DeepGEMM Test (H100)
|
- label: Kernels DeepGEMM Test (H100)
|
||||||
key: kernels-deepgemm-test-h100
|
key: kernels-deepgemm-test-h100
|
||||||
timeout_in_minutes: 35
|
timeout_in_minutes: 35
|
||||||
@@ -232,6 +263,15 @@ steps:
|
|||||||
- vllm/v1/attention/backends/mla/flashinfer_mla.py
|
- vllm/v1/attention/backends/mla/flashinfer_mla.py
|
||||||
- vllm/v1/attention/selector.py
|
- vllm/v1/attention/selector.py
|
||||||
- vllm/platforms/cuda.py
|
- vllm/platforms/cuda.py
|
||||||
|
- vllm/model_executor/kernels/linear/cute_dsl/ll_bf16.py
|
||||||
|
- vllm/model_executor/kernels/linear/cute_dsl/_ll_bf16_dotprod.py
|
||||||
|
- vllm/model_executor/kernels/linear/cute_dsl/_ll_bf16_splitk.py
|
||||||
|
- vllm/cute_utils/
|
||||||
|
- vllm/model_executor/layers/mamba/ops/gdn_chunk_cutedsl/
|
||||||
|
- vllm/model_executor/layers/fused_moe/router/bf16x3_router_gemm_cutedsl.py
|
||||||
|
- tests/kernels/mamba/test_gdn_prefill_cutedsl.py
|
||||||
|
- tests/kernels/test_bf16x3_router_gemm_cutedsl.py
|
||||||
|
- tests/kernels/test_ll_bf16_gemm.py
|
||||||
- tests/kernels/test_top_k_per_row.py
|
- tests/kernels/test_top_k_per_row.py
|
||||||
commands:
|
commands:
|
||||||
- nvidia-smi
|
- nvidia-smi
|
||||||
@@ -260,6 +300,9 @@ steps:
|
|||||||
- pytest -v -s tests/kernels/moe/test_flashinfer_moe.py
|
- pytest -v -s tests/kernels/moe/test_flashinfer_moe.py
|
||||||
- pytest -v -s tests/kernels/moe/test_trtllm_nvfp4_moe.py
|
- pytest -v -s tests/kernels/moe/test_trtllm_nvfp4_moe.py
|
||||||
- pytest -v -s tests/kernels/moe/test_cutedsl_moe.py
|
- pytest -v -s tests/kernels/moe/test_cutedsl_moe.py
|
||||||
|
- pytest -v -s tests/kernels/mamba/test_gdn_prefill_cutedsl.py
|
||||||
|
- pytest -v -s tests/kernels/test_bf16x3_router_gemm_cutedsl.py
|
||||||
|
- pytest -v -s tests/kernels/test_ll_bf16_gemm.py
|
||||||
# e2e
|
# e2e
|
||||||
- pytest -v -s tests/models/quantization/test_nvfp4.py
|
- pytest -v -s tests/models/quantization/test_nvfp4.py
|
||||||
|
|
||||||
|
|||||||
@@ -14,8 +14,9 @@ steps:
|
|||||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small.txt
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small.txt
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 55
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 45
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -78,6 +79,28 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small-tp.txt
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small-tp.txt
|
||||||
|
|
||||||
|
- label: LM Eval PCP (4xB200)
|
||||||
|
key: lm-eval-pcp-4xb200
|
||||||
|
timeout_in_minutes: 360
|
||||||
|
device: b200-k8s
|
||||||
|
num_devices: 4
|
||||||
|
optional: true
|
||||||
|
source_file_dependencies:
|
||||||
|
- csrc/
|
||||||
|
- tests/evals/gsm8k/configs/GLM-5.2-NVFP4-TP2-PCP2-EP.yaml
|
||||||
|
- tests/evals/gsm8k/configs/GLM-5.2-NVFP4-TP1-PCP4-EP.yaml
|
||||||
|
- tests/evals/gsm8k/configs/models-pcp.txt
|
||||||
|
- vllm/model_executor/layers/quantization
|
||||||
|
- vllm/config/parallel.py
|
||||||
|
- vllm/distributed/parallel_state.py
|
||||||
|
- vllm/model_executor/layers/attention/mla_attention.py
|
||||||
|
- vllm/model_executor/layers/attention/pcp.py
|
||||||
|
- vllm/v1/worker/gpu/model_runner.py
|
||||||
|
- vllm/v1/worker/gpu/pcp_manager.py
|
||||||
|
autorun_on_main: true
|
||||||
|
commands:
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-pcp.txt
|
||||||
|
|
||||||
- label: LM Eval Large Models EP (2xB200)
|
- label: LM Eval Large Models EP (2xB200)
|
||||||
key: lm-eval-large-models-ep-2xb200
|
key: lm-eval-large-models-ep-2xb200
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 60
|
||||||
@@ -103,7 +126,7 @@ steps:
|
|||||||
- vllm/transformers_utils/configs/qwen3_5_moe.py
|
- vllm/transformers_utils/configs/qwen3_5_moe.py
|
||||||
- vllm/model_executor/models/qwen3_next.py
|
- vllm/model_executor/models/qwen3_next.py
|
||||||
- vllm/model_executor/models/qwen3_next_mtp.py
|
- vllm/model_executor/models/qwen3_next_mtp.py
|
||||||
- vllm/model_executor/layers/fla/ops/
|
- vllm/third_party/flash_linear_attention/ops/
|
||||||
commands:
|
commands:
|
||||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-qwen35-blackwell.txt
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-qwen35-blackwell.txt
|
||||||
|
|
||||||
@@ -117,8 +140,9 @@ steps:
|
|||||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-h200.txt
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-h200.txt
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_8
|
device: mi300_8
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 40
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
commands:
|
commands:
|
||||||
@@ -313,7 +337,7 @@ steps:
|
|||||||
|
|
||||||
- label: LM Eval KV-Offload (2xH100)
|
- label: LM Eval KV-Offload (2xH100)
|
||||||
key: kv-offload-medium
|
key: kv-offload-medium
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -323,7 +347,7 @@ steps:
|
|||||||
- vllm/v1/simple_kv_offload/
|
- vllm/v1/simple_kv_offload/
|
||||||
- tests/evals/gsm8k/test_gsm8k_offloading.py
|
- tests/evals/gsm8k/test_gsm8k_offloading.py
|
||||||
commands:
|
commands:
|
||||||
- pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "qwen3.5-35b"
|
- pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "qwen3.5-35b or deepseek-v2-lite"
|
||||||
|
|
||||||
- label: LM Eval KV-Offload (4xH100)
|
- label: LM Eval KV-Offload (4xH100)
|
||||||
key: kv-offload-large
|
key: kv-offload-large
|
||||||
|
|||||||
@@ -14,9 +14,10 @@ steps:
|
|||||||
parallelism: 4
|
parallelism: 4
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
timeout_in_minutes: 65
|
timeout_in_minutes: 85
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/lora
|
- vllm/lora
|
||||||
- tests/lora
|
- tests/lora
|
||||||
|
|||||||
@@ -23,14 +23,15 @@ steps:
|
|||||||
- pytest -v -s -m 'not slow_test' v1/spec_decode
|
- pytest -v -s -m 'not slow_test' v1/spec_decode
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_1
|
device: mi300_1
|
||||||
timeout_in_minutes: 75
|
timeout_in_minutes: 50
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: V1 Sample + Logits
|
- label: V1 Sample + Logits
|
||||||
key: v1-sample-logits
|
key: v1-sample-logits
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 83
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/config/
|
- vllm/config/
|
||||||
@@ -58,13 +59,16 @@ steps:
|
|||||||
- pytest -v -s v1/test_outputs.py
|
- pytest -v -s v1/test_outputs.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 70
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: V1 Core + KV + Metrics
|
- label: V1 Core + KV + Metrics
|
||||||
|
device: h200_35gb
|
||||||
key: v1-core-kv-metrics
|
key: v1-core-kv-metrics
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 80
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/config/
|
- vllm/config/
|
||||||
- vllm/distributed/
|
- vllm/distributed/
|
||||||
@@ -88,6 +92,7 @@ steps:
|
|||||||
- tests/v1/kv_offload
|
- tests/v1/kv_offload
|
||||||
- tests/v1/simple_kv_offload
|
- tests/v1/simple_kv_offload
|
||||||
- tests/v1/worker
|
- tests/v1/worker
|
||||||
|
- tests/v1/streaming_input
|
||||||
- tests/v1/kv_connector/unit
|
- tests/v1/kv_connector/unit
|
||||||
- tests/v1/ec_connector/unit
|
- tests/v1/ec_connector/unit
|
||||||
- tests/v1/metrics
|
- tests/v1/metrics
|
||||||
@@ -101,6 +106,7 @@ steps:
|
|||||||
- pytest -v -s v1/kv_offload
|
- pytest -v -s v1/kv_offload
|
||||||
- pytest -v -s v1/simple_kv_offload
|
- pytest -v -s v1/simple_kv_offload
|
||||||
- pytest -v -s v1/worker
|
- pytest -v -s v1/worker
|
||||||
|
- pytest -v -s v1/streaming_input
|
||||||
- pytest -v -s -m 'not cpu_test' v1/kv_connector/unit
|
- pytest -v -s -m 'not cpu_test' v1/kv_connector/unit
|
||||||
- pytest -v -s -m 'not cpu_test' v1/ec_connector/unit
|
- pytest -v -s -m 'not cpu_test' v1/ec_connector/unit
|
||||||
- pytest -v -s -m 'not cpu_test' v1/metrics
|
- pytest -v -s -m 'not cpu_test' v1/metrics
|
||||||
@@ -109,8 +115,9 @@ steps:
|
|||||||
- pytest -v -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
|
- pytest -v -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 75
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 65
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
@@ -141,6 +148,8 @@ steps:
|
|||||||
- pytest -v -s -m 'cpu_test' v1/core
|
- pytest -v -s -m 'cpu_test' v1/core
|
||||||
- pytest -v -s v1/structured_output
|
- pytest -v -s v1/structured_output
|
||||||
- pytest -v -s v1/test_serial_utils.py
|
- pytest -v -s v1/test_serial_utils.py
|
||||||
|
- pytest -v -s v1/test_kv_cache_spec_registry.py
|
||||||
|
- pytest -v -s v1/cudagraph/test_cudagraph_manager.py
|
||||||
- pytest -v -s -m 'cpu_test' v1/kv_connector/unit
|
- pytest -v -s -m 'cpu_test' v1/kv_connector/unit
|
||||||
- pytest -v -s -m 'cpu_test' v1/metrics
|
- pytest -v -s -m 'cpu_test' v1/metrics
|
||||||
|
|
||||||
@@ -204,7 +213,7 @@ steps:
|
|||||||
- vllm/multimodal
|
- vllm/multimodal
|
||||||
- examples/
|
- examples/
|
||||||
commands:
|
commands:
|
||||||
- pip install tensorizer # for tensorizer test
|
- pip install --no-deps tensorizer # for tensorizer test
|
||||||
# for basic
|
# for basic
|
||||||
- python3 basic/offline_inference/chat.py
|
- python3 basic/offline_inference/chat.py
|
||||||
- python3 basic/offline_inference/generate.py --model facebook/opt-125m
|
- python3 basic/offline_inference/generate.py --model facebook/opt-125m
|
||||||
@@ -228,7 +237,9 @@ steps:
|
|||||||
- python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle3 --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 1536
|
- python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle3 --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 1536
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 75
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/entrypoints
|
- vllm/entrypoints
|
||||||
- vllm/multimodal
|
- vllm/multimodal
|
||||||
@@ -255,6 +266,7 @@ steps:
|
|||||||
- vllm/utils/
|
- vllm/utils/
|
||||||
- vllm/v1/
|
- vllm/v1/
|
||||||
- tests/v1/tracing
|
- tests/v1/tracing
|
||||||
|
- tests/tracing/
|
||||||
commands:
|
commands:
|
||||||
- "pip install \
|
- "pip install \
|
||||||
'opentelemetry-sdk>=1.26.0' \
|
'opentelemetry-sdk>=1.26.0' \
|
||||||
@@ -262,12 +274,14 @@ steps:
|
|||||||
'opentelemetry-exporter-otlp>=1.26.0' \
|
'opentelemetry-exporter-otlp>=1.26.0' \
|
||||||
'opentelemetry-semantic-conventions-ai>=0.4.1'"
|
'opentelemetry-semantic-conventions-ai>=0.4.1'"
|
||||||
- pytest -v -s v1/tracing
|
- pytest -v -s v1/tracing
|
||||||
|
- pytest -v -s tracing
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_2
|
dind: false
|
||||||
|
device: mi300_2
|
||||||
|
timeout_in_minutes: 30
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
optional: true
|
|
||||||
|
|
||||||
- label: Python-only Installation
|
- label: Python-only Installation
|
||||||
key: python-only-installation
|
key: python-only-installation
|
||||||
@@ -282,8 +296,9 @@ steps:
|
|||||||
- bash standalone_tests/python_only_compile.sh
|
- bash standalone_tests/python_only_compile.sh
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi250_1
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 55
|
||||||
|
soft_fail: true
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -383,7 +398,7 @@ steps:
|
|||||||
|
|
||||||
- label: Batch Invariance (A100)
|
- label: Batch Invariance (A100)
|
||||||
key: batch-invariance-a100
|
key: batch-invariance-a100
|
||||||
timeout_in_minutes: 40
|
timeout_in_minutes: 60
|
||||||
device: a100
|
device: a100
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/attention
|
- vllm/v1/attention
|
||||||
@@ -393,11 +408,11 @@ steps:
|
|||||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||||
- pip install pytest-timeout pytest-forked
|
- pip install pytest-timeout pytest-forked
|
||||||
- pytest -v -s v1/determinism/test_batch_invariance.py
|
- pytest -v -s v1/determinism/test_batch_invariance.py
|
||||||
- VLLM_TEST_MODEL=deepseek-ai/DeepSeek-V2-Lite-Chat pytest -v -s v1/determinism/test_batch_invariance.py::test_v1_generation_is_deterministic_across_batch_sizes_with_needle[TRITON_MLA]
|
- VLLM_TEST_MODEL=deepseek-ai/DeepSeek-V2-Lite-Chat pytest -v -s v1/determinism/test_batch_invariance.py::test_v1_generation_is_deterministic_across_batch_sizes_with_needle -k TRITON_MLA
|
||||||
|
|
||||||
- label: Batch Invariance (H100)
|
- label: Batch Invariance (H100)
|
||||||
key: batch-invariance-h100
|
key: batch-invariance-h100
|
||||||
timeout_in_minutes: 40
|
timeout_in_minutes: 60
|
||||||
device: h100
|
device: h100
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/attention
|
- vllm/v1/attention
|
||||||
@@ -408,12 +423,12 @@ steps:
|
|||||||
- pip install pytest-timeout pytest-forked
|
- pip install pytest-timeout pytest-forked
|
||||||
- pytest -v -s v1/determinism/test_batch_invariance.py
|
- pytest -v -s v1/determinism/test_batch_invariance.py
|
||||||
- pytest -v -s v1/determinism/test_rms_norm_batch_invariant.py
|
- pytest -v -s v1/determinism/test_rms_norm_batch_invariant.py
|
||||||
- VLLM_TEST_MODEL=deepseek-ai/DeepSeek-V2-Lite-Chat pytest -v -s v1/determinism/test_batch_invariance.py::test_v1_generation_is_deterministic_across_batch_sizes_with_needle[TRITON_MLA]
|
- VLLM_TEST_MODEL=deepseek-ai/DeepSeek-V2-Lite-Chat pytest -v -s v1/determinism/test_batch_invariance.py::test_v1_generation_is_deterministic_across_batch_sizes_with_needle -k TRITON_MLA
|
||||||
- VLLM_TEST_MODEL=Qwen/Qwen3-30B-A3B-Thinking-2507-FP8 pytest -v -s v1/determinism/test_batch_invariance.py::test_v1_generation_is_deterministic_across_batch_sizes_with_needle[FLASH_ATTN]
|
- VLLM_TEST_MODEL=Qwen/Qwen3-30B-A3B-Thinking-2507-FP8 pytest -v -s v1/determinism/test_batch_invariance.py::test_v1_generation_is_deterministic_across_batch_sizes_with_needle -k FLASH_ATTN
|
||||||
|
|
||||||
- label: Batch Invariance (B200)
|
- label: Batch Invariance (B200)
|
||||||
key: batch-invariance-b200
|
key: batch-invariance-b200
|
||||||
timeout_in_minutes: 35
|
timeout_in_minutes: 45
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/attention
|
- vllm/v1/attention
|
||||||
@@ -424,10 +439,13 @@ steps:
|
|||||||
- pip install pytest-timeout pytest-forked
|
- pip install pytest-timeout pytest-forked
|
||||||
- pytest -v -s v1/determinism/test_batch_invariance.py
|
- pytest -v -s v1/determinism/test_batch_invariance.py
|
||||||
- pytest -v -s v1/determinism/test_rms_norm_batch_invariant.py
|
- pytest -v -s v1/determinism/test_rms_norm_batch_invariant.py
|
||||||
- VLLM_TEST_MODEL=deepseek-ai/DeepSeek-V2-Lite-Chat pytest -v -s v1/determinism/test_batch_invariance.py::test_v1_generation_is_deterministic_across_batch_sizes_with_needle[TRITON_MLA]
|
- VLLM_TEST_MODEL=deepseek-ai/DeepSeek-V2-Lite-Chat pytest -v -s v1/determinism/test_batch_invariance.py::test_v1_generation_is_deterministic_across_batch_sizes_with_needle -k TRITON_MLA
|
||||||
- VLLM_TEST_MODEL=Qwen/Qwen3-30B-A3B-Thinking-2507-FP8 pytest -v -s v1/determinism/test_batch_invariance.py::test_v1_generation_is_deterministic_across_batch_sizes_with_needle[FLASH_ATTN]
|
- VLLM_TEST_MODEL=Qwen/Qwen3-30B-A3B-Thinking-2507-FP8 pytest -v -s v1/determinism/test_batch_invariance.py::test_v1_generation_is_deterministic_across_batch_sizes_with_needle -k FLASH_ATTN
|
||||||
- pytest -v -s v1/determinism/test_nvfp4_batch_invariant.py
|
- pytest -v -s v1/determinism/test_nvfp4_batch_invariant.py
|
||||||
- pytest -v -s v1/determinism/test_nvfp4_batch_invariant_scaled_mm.py
|
- pytest -v -s v1/determinism/test_nvfp4_batch_invariant_scaled_mm.py
|
||||||
|
- pytest -v -s v1/determinism/test_matmul_batch_invariant.py
|
||||||
|
- pytest -v -s v1/determinism/test_cutlass_batch_invariance.py
|
||||||
|
- pytest -v -s v1/determinism/test_online_batch_invariance.py
|
||||||
|
|
||||||
- label: Acceptance Length Test (Large Models) # optional
|
- label: Acceptance Length Test (Large Models) # optional
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
|
|||||||
@@ -3,13 +3,16 @@ depends_on:
|
|||||||
- image-build
|
- image-build
|
||||||
steps:
|
steps:
|
||||||
- label: Model Executor
|
- label: Model Executor
|
||||||
|
device: h200_35gb
|
||||||
key: model-executor
|
key: model-executor
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 60
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/engine/arg_utils.py
|
- vllm/engine/arg_utils.py
|
||||||
- vllm/config/model.py
|
- vllm/config/model.py
|
||||||
- vllm/model_executor
|
- vllm/model_executor
|
||||||
|
- vllm/model_executor/warmup
|
||||||
- tests/model_executor
|
- tests/model_executor
|
||||||
|
- tests/model_executor/test_jit_warmup.py
|
||||||
- tests/entrypoints/openai/completion/test_tensorizer_entrypoint.py
|
- tests/entrypoints/openai/completion/test_tensorizer_entrypoint.py
|
||||||
commands:
|
commands:
|
||||||
- apt-get update && apt-get install -y curl libsodium23
|
- apt-get update && apt-get install -y curl libsodium23
|
||||||
@@ -25,14 +28,18 @@ steps:
|
|||||||
- pytest -v -s entrypoints/openai/completion/test_tensorizer_entrypoint.py --timeout=900 --timeout-method=thread
|
- pytest -v -s entrypoints/openai/completion/test_tensorizer_entrypoint.py --timeout=900 --timeout-method=thread
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_1
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 60
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/engine/arg_utils.py
|
- vllm/engine/arg_utils.py
|
||||||
- vllm/config/model.py
|
- vllm/config/model.py
|
||||||
- vllm/model_executor
|
- vllm/model_executor
|
||||||
|
- vllm/model_executor/warmup
|
||||||
- tests/model_executor
|
- tests/model_executor
|
||||||
|
- tests/model_executor/test_jit_warmup.py
|
||||||
- tests/entrypoints/openai/completion/test_tensorizer_entrypoint.py
|
- tests/entrypoints/openai/completion/test_tensorizer_entrypoint.py
|
||||||
- vllm/_aiter_ops.py
|
- vllm/_aiter_ops.py
|
||||||
- vllm/platforms/rocm.py
|
- vllm/platforms/rocm.py
|
||||||
|
|||||||
@@ -41,7 +41,7 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- set -x
|
- set -x
|
||||||
- export VLLM_USE_V2_MODEL_RUNNER=1
|
- export VLLM_USE_V2_MODEL_RUNNER=1
|
||||||
- pip install tensorizer # for tensorizer test
|
- pip install --no-deps tensorizer # for tensorizer test
|
||||||
- python3 basic/offline_inference/chat.py # for basic
|
- python3 basic/offline_inference/chat.py # for basic
|
||||||
- python3 basic/offline_inference/generate.py --model facebook/opt-125m
|
- python3 basic/offline_inference/generate.py --model facebook/opt-125m
|
||||||
#- python3 basic/offline_inference/generate.py --model meta-llama/Llama-2-13b-chat-hf --cpu-offload-gb 10 # TODO
|
#- python3 basic/offline_inference/generate.py --model meta-llama/Llama-2-13b-chat-hf --cpu-offload-gb 10 # TODO
|
||||||
|
|||||||
@@ -42,10 +42,39 @@ steps:
|
|||||||
- pytest -v -s models/test_terratorch.py models/transformers/test_backend.py models/test_registry.py
|
- pytest -v -s models/test_terratorch.py models/transformers/test_backend.py models/test_registry.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 50
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
|
- label: Inkling Unit Tests (B200)
|
||||||
|
key: inkling-unit-tests-b200
|
||||||
|
timeout_in_minutes: 40
|
||||||
|
device: b200-k8s
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/models/inkling/
|
||||||
|
- vllm/cute_utils/
|
||||||
|
- cmake/external_projects/tml_fa4.cmake
|
||||||
|
- tests/models/inkling/
|
||||||
|
commands:
|
||||||
|
# FA4 kernel tests require SM100; the suite skips them elsewhere.
|
||||||
|
- pytest -v -s models/inkling
|
||||||
|
|
||||||
|
- label: Kimi K3 Unit Tests (B200)
|
||||||
|
key: kimi-k3-unit-tests-b200
|
||||||
|
timeout_in_minutes: 40
|
||||||
|
device: b200-k8s
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/models/kimi_k3/
|
||||||
|
- csrc/libtorch_stable/kimi_k3/
|
||||||
|
- tests/models/kimi_k3/
|
||||||
|
- tests/kernels/attention/test_kimi_k3_mla_fused_epilogue.py
|
||||||
|
- tests/kernels/test_bf16_skinny_gemm.py
|
||||||
|
commands:
|
||||||
|
# The native NVIDIA Kimi K3 kernels require the SM100 family.
|
||||||
|
- pytest -v -s models/kimi_k3 kernels/attention/test_kimi_k3_mla_fused_epilogue.py kernels/test_bf16_skinny_gemm.py
|
||||||
|
|
||||||
- label: Basic Models Test (Other CPU) # 5min
|
- label: Basic Models Test (Other CPU) # 5min
|
||||||
key: basic-models-test-other-cpu
|
key: basic-models-test-other-cpu
|
||||||
depends_on:
|
depends_on:
|
||||||
@@ -55,7 +84,8 @@ steps:
|
|||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/test_utils.py
|
- tests/models/test_utils.py
|
||||||
- tests/models/test_vision.py
|
- tests/models/test_vision.py
|
||||||
|
- tests/models/test_adapters.py
|
||||||
- tests/models/transformers/fusers/
|
- tests/models/transformers/fusers/
|
||||||
device: cpu-small
|
device: cpu-small
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s models/test_utils.py models/test_vision.py models/transformers/fusers/
|
- pytest -v -s models/test_utils.py models/test_vision.py models/test_adapters.py models/transformers/fusers/
|
||||||
|
|||||||
@@ -15,11 +15,14 @@ steps:
|
|||||||
- pytest -v -s models/language -m 'core_model and (not slow_test)'
|
- pytest -v -s models/language -m 'core_model and (not slow_test)'
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_1
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 45
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: Language Models Tests (Extra Standard) %N
|
- label: Language Models Tests (Extra Standard) %N
|
||||||
|
device: h200_35gb
|
||||||
key: language-models-tests-extra-standard
|
key: language-models-tests-extra-standard
|
||||||
timeout_in_minutes: 40
|
timeout_in_minutes: 40
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -35,7 +38,9 @@ steps:
|
|||||||
parallelism: 2
|
parallelism: 2
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_1
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 40
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -49,8 +54,8 @@ steps:
|
|||||||
- tests/models/language/pooling/test_classification.py
|
- tests/models/language/pooling/test_classification.py
|
||||||
- vllm/_aiter_ops.py
|
- vllm/_aiter_ops.py
|
||||||
- vllm/platforms/rocm.py
|
- vllm/platforms/rocm.py
|
||||||
|
|
||||||
- label: Language Models Tests (Hybrid) %N
|
- label: Language Models Tests (Hybrid) %N
|
||||||
|
device: h200_35gb
|
||||||
key: language-models-tests-hybrid
|
key: language-models-tests-hybrid
|
||||||
timeout_in_minutes: 65
|
timeout_in_minutes: 65
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -58,16 +63,16 @@ steps:
|
|||||||
- tests/models/language/generation
|
- tests/models/language/generation
|
||||||
commands:
|
commands:
|
||||||
# Install fast path packages for testing against transformers
|
# Install fast path packages for testing against transformers
|
||||||
# Note: also needed to run plamo2 model in vLLM
|
|
||||||
- uv pip install --system --no-build-isolation 'git+https://github.com/state-spaces/mamba@v2.3.0'
|
- uv pip install --system --no-build-isolation 'git+https://github.com/state-spaces/mamba@v2.3.0'
|
||||||
- uv pip install --system --no-build-isolation 'git+https://github.com/Dao-AILab/causal-conv1d@v1.6.0'
|
- uv pip install --system --no-build-isolation 'git+https://github.com/Dao-AILab/causal-conv1d@v1.6.0'
|
||||||
# Shard hybrid language model tests
|
# Shard the hybrid language model tests that are numerically stable on Hopper.
|
||||||
- pytest -v -s models/language/generation -m hybrid_model --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
|
- pytest -v -s models/language/generation -m hybrid_model -k 'not granite-4.0-tiny-preview' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
|
||||||
parallelism: 2
|
parallelism: 2
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 70
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 60
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
commands:
|
commands:
|
||||||
@@ -75,6 +80,20 @@ steps:
|
|||||||
- uv pip install --system --no-build-isolation 'git+https://github.com/Dao-AILab/causal-conv1d@v1.6.0'
|
- uv pip install --system --no-build-isolation 'git+https://github.com/Dao-AILab/causal-conv1d@v1.6.0'
|
||||||
- pytest -v -s models/language/generation -m hybrid_model --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
|
- pytest -v -s models/language/generation -m hybrid_model --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
|
||||||
|
|
||||||
|
# Granite 4 hybrid generation is sensitive to hardware-specific Triton SSD
|
||||||
|
# autotuning (https://github.com/vllm-project/vllm/issues/25194). Keep this one
|
||||||
|
# correctness test on L4 until its H200 output matches the Transformers reference.
|
||||||
|
- label: Language Models Tests (Granite L4 Compatibility)
|
||||||
|
key: language-models-tests-granite-l4-compatibility
|
||||||
|
timeout_in_minutes: 65
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/
|
||||||
|
- tests/models/language/generation
|
||||||
|
commands:
|
||||||
|
- uv pip install --system --no-build-isolation 'git+https://github.com/state-spaces/mamba@v2.3.0'
|
||||||
|
- uv pip install --system --no-build-isolation 'git+https://github.com/Dao-AILab/causal-conv1d@v1.6.0'
|
||||||
|
- pytest -v -s models/language/generation -m hybrid_model -k 'granite-4.0-tiny-preview'
|
||||||
|
|
||||||
- label: Language Models Test (Extended Generation) # 80min
|
- label: Language Models Test (Extended Generation) # 80min
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: language-models-test-extended-generation
|
key: language-models-test-extended-generation
|
||||||
@@ -85,7 +104,6 @@ steps:
|
|||||||
- tests/models/language/generation
|
- tests/models/language/generation
|
||||||
commands:
|
commands:
|
||||||
# Install fast path packages for testing against transformers
|
# Install fast path packages for testing against transformers
|
||||||
# Note: also needed to run plamo2 model in vLLM
|
|
||||||
- uv pip install --system --no-build-isolation 'git+https://github.com/state-spaces/mamba@v2.3.0'
|
- uv pip install --system --no-build-isolation 'git+https://github.com/state-spaces/mamba@v2.3.0'
|
||||||
- uv pip install --system --no-build-isolation 'git+https://github.com/Dao-AILab/causal-conv1d@v1.6.0'
|
- uv pip install --system --no-build-isolation 'git+https://github.com/Dao-AILab/causal-conv1d@v1.6.0'
|
||||||
- pytest -v -s models/language/generation -m '(not core_model) and (not hybrid_model)'
|
- pytest -v -s models/language/generation -m '(not core_model) and (not hybrid_model)'
|
||||||
@@ -101,10 +119,10 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- pytest -v -s models/language/generation_ppl_test
|
- pytest -v -s models/language/generation_ppl_test
|
||||||
|
|
||||||
- label: Language Models Test (Extended Pooling) # 36min
|
- label: Language Models Test (Extended Pooling)
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: language-models-test-extended-pooling
|
key: language-models-test-extended-pooling
|
||||||
timeout_in_minutes: 70
|
timeout_in_minutes: 120
|
||||||
optional: true
|
optional: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -113,14 +131,15 @@ steps:
|
|||||||
- pytest -v -s models/language/pooling -m 'not core_model'
|
- pytest -v -s models/language/pooling -m 'not core_model'
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 100
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 95
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: Language Models Test (MTEB)
|
- label: Language Models Test (MTEB)
|
||||||
key: language-models-test-mteb
|
key: language-models-test-mteb
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 68
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
optional: true
|
optional: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: "Multi-Modal Models (Standard) 1: qwen2"
|
- label: "Multi-Modal Models (Standard) 1: qwen2"
|
||||||
key: multi-modal-models-standard-1-qwen2
|
key: multi-modal-models-standard-1-qwen2
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 68
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -14,13 +14,15 @@ steps:
|
|||||||
- pytest -v -s models/multimodal/generation/test_ultravox.py -m core_model
|
- pytest -v -s models/multimodal/generation/test_ultravox.py -m core_model
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 65
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: "Multi-Modal Models (Standard) 2: qwen3 + gemma"
|
- label: "Multi-Modal Models (Standard) 2: qwen3 + gemma"
|
||||||
key: multi-modal-models-standard-2-qwen3-gemma
|
key: multi-modal-models-standard-2-qwen3-gemma
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 75
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -31,7 +33,9 @@ steps:
|
|||||||
- pytest -v -s models/multimodal/generation/test_qwen2_5_vl.py -m core_model
|
- pytest -v -s models/multimodal/generation/test_qwen2_5_vl.py -m core_model
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 55
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
@@ -47,14 +51,15 @@ steps:
|
|||||||
- pytest -v -s models/multimodal/generation/test_qwen2_vl.py -m core_model
|
- pytest -v -s models/multimodal/generation/test_qwen2_vl.py -m core_model
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi250_1
|
||||||
|
timeout_in_minutes: 55
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: "Multi-Modal Models (Standard) 4: other + whisper"
|
- label: "Multi-Modal Models (Standard) 4: other + whisper"
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: multi-modal-models-standard-4-other-whisper
|
key: multi-modal-models-standard-4-other-whisper
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 75
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/multimodal
|
- tests/models/multimodal
|
||||||
@@ -65,7 +70,9 @@ steps:
|
|||||||
- cd .. && VLLM_WORKER_MULTIPROC_METHOD=spawn pytest -v -s tests/models/multimodal/generation/test_whisper.py -m core_model # Otherwise, mp_method="spawn" doesn't work
|
- cd .. && VLLM_WORKER_MULTIPROC_METHOD=spawn pytest -v -s tests/models/multimodal/generation/test_whisper.py -m core_model # Otherwise, mp_method="spawn" doesn't work
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 50
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
@@ -85,7 +92,7 @@ steps:
|
|||||||
|
|
||||||
- label: Multi-Modal Processor # 44min
|
- label: Multi-Modal Processor # 44min
|
||||||
key: multi-modal-processor
|
key: multi-modal-processor
|
||||||
timeout_in_minutes: 65
|
timeout_in_minutes: 98
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -107,7 +114,9 @@ steps:
|
|||||||
- pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-mm-small.txt --tp-size=1
|
- pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-mm-small.txt --tp-size=1
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_1
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 35
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -118,6 +127,7 @@ steps:
|
|||||||
- vllm/model_executor/model_loader/
|
- vllm/model_executor/model_loader/
|
||||||
|
|
||||||
- label: Multi-Modal Models (Extended Generation 1)
|
- label: Multi-Modal Models (Extended Generation 1)
|
||||||
|
device: h200_35gb
|
||||||
key: multi-modal-models-extended-generation-1
|
key: multi-modal-models-extended-generation-1
|
||||||
optional: true
|
optional: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -129,7 +139,9 @@ steps:
|
|||||||
- pytest -v -s models/multimodal/test_mapping.py
|
- pytest -v -s models/multimodal/test_mapping.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 90
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
@@ -164,8 +176,9 @@ steps:
|
|||||||
- pytest -v -s models/multimodal/pooling -m 'not core_model'
|
- pytest -v -s models/multimodal/pooling -m 'not core_model'
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 75
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 60
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ steps:
|
|||||||
- label: PyTorch Compilation Unit Tests
|
- label: PyTorch Compilation Unit Tests
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: pytorch-compilation-unit-tests
|
key: pytorch-compilation-unit-tests
|
||||||
timeout_in_minutes: 90
|
timeout_in_minutes: 150
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/__init__.py
|
- vllm/__init__.py
|
||||||
- vllm/_aiter_ops.py
|
- vllm/_aiter_ops.py
|
||||||
@@ -107,16 +107,11 @@ steps:
|
|||||||
- tests/compile/passes
|
- tests/compile/passes
|
||||||
commands:
|
commands:
|
||||||
- pytest -s -v compile/passes --ignore compile/passes/distributed
|
- pytest -s -v compile/passes --ignore compile/passes/distributed
|
||||||
mirror:
|
|
||||||
amd:
|
|
||||||
device: mi300_1
|
|
||||||
timeout_in_minutes: 65
|
|
||||||
depends_on:
|
|
||||||
- image-build-amd
|
|
||||||
|
|
||||||
- label: PyTorch Fullgraph Smoke Test
|
- label: PyTorch Fullgraph Smoke Test
|
||||||
|
device: h200_35gb
|
||||||
key: pytorch-fullgraph-smoke-test
|
key: pytorch-fullgraph-smoke-test
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 90
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/__init__.py
|
- vllm/__init__.py
|
||||||
- vllm/_aiter_ops.py
|
- vllm/_aiter_ops.py
|
||||||
@@ -148,7 +143,42 @@ steps:
|
|||||||
# as it is a heavy test that is covered in other steps.
|
# as it is a heavy test that is covered in other steps.
|
||||||
# Use `find` to launch multiple instances of pytest so that
|
# Use `find` to launch multiple instances of pytest so that
|
||||||
# they do not suffer from https://github.com/vllm-project/vllm/issues/28965
|
# they do not suffer from https://github.com/vllm-project/vllm/issues/28965
|
||||||
- "find compile/fullgraph/ -name 'test_*.py' -not -name 'test_full_graph.py' -print0 | xargs -0 -n1 -I{} pytest -s -v '{}'"
|
- "find compile/fullgraph/ -name 'test_*.py' -not -name 'test_full_cudagraph.py' -not -name 'test_full_graph.py' -print0 | xargs -0 -n1 -I{} pytest -s -v '{}'"
|
||||||
|
|
||||||
|
# Hopper-only DeepSeek-V2-Lite cases in this file require two 29.3-GiB model
|
||||||
|
# instances and cannot fit a 35GB MIG slice. L4 retains the original coverage:
|
||||||
|
# those SM90 cases skip while the architecture-compatible cases still run.
|
||||||
|
- label: PyTorch Fullgraph CUDAGraph (L4 Compatibility)
|
||||||
|
key: pytorch-fullgraph-cudagraph-l4-compatibility
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/__init__.py
|
||||||
|
- vllm/_aiter_ops.py
|
||||||
|
- vllm/_custom_ops.py
|
||||||
|
- vllm/compilation/
|
||||||
|
- vllm/config/
|
||||||
|
- vllm/distributed/
|
||||||
|
- vllm/engine/
|
||||||
|
- vllm/env_override.py
|
||||||
|
- vllm/envs.py
|
||||||
|
- vllm/forward_context.py
|
||||||
|
- vllm/inputs/
|
||||||
|
- vllm/ir/
|
||||||
|
- vllm/kernels/
|
||||||
|
- vllm/logger.py
|
||||||
|
- vllm/model_executor/
|
||||||
|
- vllm/multimodal/
|
||||||
|
- vllm/platforms/
|
||||||
|
- vllm/plugins/
|
||||||
|
- vllm/sampling_params.py
|
||||||
|
- vllm/sequence.py
|
||||||
|
- vllm/transformers_utils/
|
||||||
|
- vllm/triton_utils/
|
||||||
|
- vllm/utils/
|
||||||
|
- vllm/v1/
|
||||||
|
- tests/compile
|
||||||
|
commands:
|
||||||
|
- pytest -s -v compile/fullgraph/test_full_cudagraph.py
|
||||||
|
|
||||||
- label: PyTorch Fullgraph
|
- label: PyTorch Fullgraph
|
||||||
key: pytorch-fullgraph
|
key: pytorch-fullgraph
|
||||||
@@ -197,7 +227,9 @@ steps:
|
|||||||
- bash standalone_tests/pytorch_nightly_dependency.sh
|
- bash standalone_tests/pytorch_nightly_dependency.sh
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_1
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 30
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -3,8 +3,11 @@ depends_on:
|
|||||||
- image-build
|
- image-build
|
||||||
steps:
|
steps:
|
||||||
- label: Quantization
|
- label: Quantization
|
||||||
|
device: h200_35gb
|
||||||
key: quantization
|
key: quantization
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 75
|
||||||
|
env:
|
||||||
|
VLLM_USE_V2_MODEL_RUNNER: "0"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/
|
- csrc/
|
||||||
- vllm/model_executor/layers/quantization
|
- vllm/model_executor/layers/quantization
|
||||||
@@ -19,9 +22,12 @@ steps:
|
|||||||
# TODO(jerryzh168): resolve the above comment
|
# TODO(jerryzh168): resolve the above comment
|
||||||
- uv pip install --system torchao==0.17.0 --index-url https://download.pytorch.org/whl/cu130
|
- uv pip install --system torchao==0.17.0 --index-url https://download.pytorch.org/whl/cu130
|
||||||
- uv pip install --system conch-triton-kernels
|
- uv pip install --system conch-triton-kernels
|
||||||
- VLLM_TEST_FORCE_LOAD_FORMAT=auto pytest -v -s quantization/ --ignore quantization/test_blackwell_moe.py
|
# The SM90-only checkpoint currently contains a removed weight_chan_scale
|
||||||
|
# parameter. It was not exercised by the previous L4 job.
|
||||||
|
- VLLM_TEST_FORCE_LOAD_FORMAT=auto pytest -v -s quantization/ --ignore quantization/test_blackwell_moe.py -k 'not test_compressed_tensors_w4a8_fp8'
|
||||||
|
|
||||||
- label: Quantized Fusions
|
- label: Quantized Fusions
|
||||||
|
device: h200_35gb
|
||||||
key: quantized-fusions
|
key: quantized-fusions
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 20
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -52,8 +58,11 @@ steps:
|
|||||||
- pytest -s -v tests/quantization/test_blackwell_moe.py
|
- pytest -s -v tests/quantization/test_blackwell_moe.py
|
||||||
|
|
||||||
- label: Quantized Models Test
|
- label: Quantized Models Test
|
||||||
|
device: h200_35gb
|
||||||
key: quantized-models-test
|
key: quantized-models-test
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 65
|
||||||
|
env:
|
||||||
|
VLLM_USE_V2_MODEL_RUNNER: "0"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/model_executor/layers/quantization
|
- vllm/model_executor/layers/quantization
|
||||||
- tests/models/quantization
|
- tests/models/quantization
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ steps:
|
|||||||
- export VLLM_USE_RUST_FRONTEND=1
|
- export VLLM_USE_RUST_FRONTEND=1
|
||||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||||
- pytest -v -s benchmarks/test_serve_cli.py -k "not insecure and not (test_bench_serve and not test_bench_serve_chat)"
|
- pytest -v -s benchmarks/test_serve_cli.py -k "not insecure and not (test_bench_serve and not test_bench_serve_chat)"
|
||||||
- pytest -v -s entrypoints/openai/chat_completion/test_chat_completion.py -k "not test_invalid_json_schema and not test_invalid_regex"
|
- pytest -v -s entrypoints/openai/chat_completion/test_chat_completion.py -k "not test_invalid_json_schema and not test_invalid_regex and not test_kv_transfer_prompt_token_ids_round_trip and not test_kv_transfer_prompt_token_ids_streaming"
|
||||||
- pytest -v -s entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py -k "not multiple"
|
- pytest -v -s entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py -k "not multiple"
|
||||||
|
|
||||||
# - pytest -v -s entrypoints/openai/completion/test_prompt_validation.py -k "not prompt_embeds"
|
# - pytest -v -s entrypoints/openai/completion/test_prompt_validation.py -k "not prompt_embeds"
|
||||||
@@ -81,6 +81,7 @@ steps:
|
|||||||
- pytest -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
|
- pytest -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
|
||||||
|
|
||||||
- label: Rust Frontend Tool Use
|
- label: Rust Frontend Tool Use
|
||||||
|
device: h200_35gb
|
||||||
timeout_in_minutes: 25
|
timeout_in_minutes: 25
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -19,8 +19,18 @@ steps:
|
|||||||
- VLLM_USE_FLASHINFER_SAMPLER=1 pytest -v -s samplers
|
- VLLM_USE_FLASHINFER_SAMPLER=1 pytest -v -s samplers
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi250_1
|
||||||
|
timeout_in_minutes: 40
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/model_executor/layers
|
||||||
|
- vllm/sampling_metadata.py
|
||||||
|
- vllm/v1/sample/
|
||||||
|
- vllm/entrypoints/generate/beam_search/
|
||||||
|
- tests/samplers
|
||||||
|
- tests/conftest.py
|
||||||
|
- vllm/_aiter_ops.py
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s samplers
|
- pytest -v -s samplers
|
||||||
|
|||||||
@@ -14,8 +14,9 @@ steps:
|
|||||||
- pytest -v -s v1/e2e/spec_decode -k "eagle_correctness"
|
- pytest -v -s v1/e2e/spec_decode -k "eagle_correctness"
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 60
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 55
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -53,8 +54,9 @@ steps:
|
|||||||
- pytest -v -s v1/e2e/spec_decode -k "speculators or mtp_correctness"
|
- pytest -v -s v1/e2e/spec_decode -k "speculators or mtp_correctness"
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 65
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 75
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -88,14 +90,15 @@ steps:
|
|||||||
- vllm/v1/spec_decode/
|
- vllm/v1/spec_decode/
|
||||||
- vllm/v1/worker/gpu/spec_decode/
|
- vllm/v1/worker/gpu/spec_decode/
|
||||||
- tests/v1/e2e/spec_decode/
|
- tests/v1/e2e/spec_decode/
|
||||||
|
- tests/spec_decode/
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s v1/e2e/spec_decode -k "ngram or suffix"
|
- pytest -v -s v1/e2e/spec_decode -k "ngram or suffix"
|
||||||
|
- python3 spec_decode/test_custom_proposer.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
timeout_in_minutes: 55
|
device: mi300_1
|
||||||
# TODO(akaratza): Test after Torch >= 2.12 bump
|
timeout_in_minutes: 35
|
||||||
soft_fail: true
|
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -119,7 +122,8 @@ steps:
|
|||||||
- pytest -v -s v1/e2e/spec_decode -k "draft_model or no_sync or batch_inference"
|
- pytest -v -s v1/e2e/spec_decode -k "draft_model or no_sync or batch_inference"
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
dind: false
|
||||||
|
device: mi300_1
|
||||||
timeout_in_minutes: 55
|
timeout_in_minutes: 55
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
@@ -170,3 +174,31 @@ steps:
|
|||||||
- tests/v1/e2e/spec_decode/
|
- tests/v1/e2e/spec_decode/
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s v1/e2e/spec_decode -k "qwen3_5-hybrid"
|
- pytest -v -s v1/e2e/spec_decode -k "qwen3_5-hybrid"
|
||||||
|
|
||||||
|
- label: Spec Decode DeepSeek MTP Parallel Load (B200)
|
||||||
|
key: spec-decode-deepseek-mtp-parallel-load-b200
|
||||||
|
timeout_in_minutes: 30
|
||||||
|
device: b200-k8s
|
||||||
|
optional: true
|
||||||
|
num_devices: 2
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/v1/spec_decode/llm_base_proposer.py
|
||||||
|
- vllm/v1/spec_decode/eagle.py
|
||||||
|
- vllm/v1/worker/gpu/spec_decode/eagle/
|
||||||
|
- vllm/model_executor/models/deepseek_mtp.py
|
||||||
|
- vllm/model_executor/models/deepseek_v2.py
|
||||||
|
- tests/v1/e2e/spec_decode/test_mtp_parallel_load.py
|
||||||
|
commands:
|
||||||
|
- pytest -v -s v1/e2e/spec_decode/test_mtp_parallel_load.py
|
||||||
|
|
||||||
|
- label: Spec Decode Acceptance Rates Nightly
|
||||||
|
key: spec-decode-acceptance-rates-nightly
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
device: h200_35gb
|
||||||
|
optional: true
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/v1/spec_decode/
|
||||||
|
- vllm/v1/worker/gpu/spec_decode/
|
||||||
|
- tests/v1/e2e/spec_decode/
|
||||||
|
commands:
|
||||||
|
- pytest -v -s v1/e2e/spec_decode/test_spec_decode.py -k "acceptance_rates"
|
||||||
|
|||||||
@@ -0,0 +1,14 @@
|
|||||||
|
group: Torch ABI
|
||||||
|
depends_on:
|
||||||
|
- image-build
|
||||||
|
steps:
|
||||||
|
- label: Torch Stable ABI Audit
|
||||||
|
key: torch-stable-abi-audit
|
||||||
|
timeout_in_minutes: 5
|
||||||
|
source_file_dependencies:
|
||||||
|
- .buildkite/check-torch-abi.py
|
||||||
|
- csrc/
|
||||||
|
- cmake/
|
||||||
|
- setup.py
|
||||||
|
commands:
|
||||||
|
- python3 /vllm-workspace/.buildkite/check-torch-abi.py
|
||||||
@@ -15,7 +15,9 @@ steps:
|
|||||||
- bash weight_loading/run_model_weight_loading_test.sh -c weight_loading/models.txt
|
- bash weight_loading/run_model_weight_loading_test.sh -c weight_loading/models.txt
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
|
dind: false
|
||||||
device: mi300_2
|
device: mi300_2
|
||||||
|
timeout_in_minutes: 35
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
commands:
|
commands:
|
||||||
|
|||||||
+2
-1
@@ -47,6 +47,7 @@
|
|||||||
|
|
||||||
# Rust Frontend
|
# Rust Frontend
|
||||||
/rust/ @BugenZhao @njhill
|
/rust/ @BugenZhao @njhill
|
||||||
|
/rust/src/bench @esmeetu
|
||||||
/build_rust.sh @BugenZhao @njhill
|
/build_rust.sh @BugenZhao @njhill
|
||||||
/rust-toolchain.toml @BugenZhao @njhill
|
/rust-toolchain.toml @BugenZhao @njhill
|
||||||
/.buildkite/test_areas/rust* @BugenZhao @njhill
|
/.buildkite/test_areas/rust* @BugenZhao @njhill
|
||||||
@@ -172,7 +173,7 @@ mkdocs.yaml @hmellor
|
|||||||
# Kernels
|
# Kernels
|
||||||
/vllm/v1/attention/ops/chunked_prefill_paged_decode.py @tdoublep
|
/vllm/v1/attention/ops/chunked_prefill_paged_decode.py @tdoublep
|
||||||
/vllm/v1/attention/ops/triton_unified_attention.py @tdoublep
|
/vllm/v1/attention/ops/triton_unified_attention.py @tdoublep
|
||||||
/vllm/model_executor/layers/fla @ZJY0516 @vadiklyutiy
|
/vllm/third_party/flash_linear_attention @ZJY0516 @vadiklyutiy
|
||||||
|
|
||||||
# ROCm related: specify owner with write access to notify AMD folks for careful code review
|
# ROCm related: specify owner with write access to notify AMD folks for careful code review
|
||||||
/vllm/**/*rocm* @tjtanaa @dllehr-amd
|
/vllm/**/*rocm* @tjtanaa @dllehr-amd
|
||||||
|
|||||||
@@ -19,6 +19,7 @@ pull_request_rules:
|
|||||||
description: Comment on PR when pre-commit check fails
|
description: Comment on PR when pre-commit check fails
|
||||||
conditions:
|
conditions:
|
||||||
- check-failure=pre-commit
|
- check-failure=pre-commit
|
||||||
|
- -check-cancelled=pre-commit
|
||||||
- -closed
|
- -closed
|
||||||
- -draft
|
- -draft
|
||||||
- or:
|
- or:
|
||||||
@@ -181,6 +182,18 @@ pull_request_rules:
|
|||||||
add:
|
add:
|
||||||
- performance
|
- performance
|
||||||
|
|
||||||
|
- name: label-quantization
|
||||||
|
description: Automatically apply quantization label
|
||||||
|
conditions:
|
||||||
|
- label != stale
|
||||||
|
- or:
|
||||||
|
- files~=^vllm/model_executor/layers/quantization/
|
||||||
|
- title~=(?i)quant
|
||||||
|
actions:
|
||||||
|
label:
|
||||||
|
add:
|
||||||
|
- quantization
|
||||||
|
|
||||||
- name: label-qwen
|
- name: label-qwen
|
||||||
description: Automatically apply qwen label
|
description: Automatically apply qwen label
|
||||||
conditions:
|
conditions:
|
||||||
@@ -220,6 +233,31 @@ pull_request_rules:
|
|||||||
add:
|
add:
|
||||||
- gpt-oss
|
- gpt-oss
|
||||||
|
|
||||||
|
- name: label-kimi
|
||||||
|
description: Automatically apply kimi label
|
||||||
|
conditions:
|
||||||
|
- label != stale
|
||||||
|
- or:
|
||||||
|
- files~=(?i)kimi
|
||||||
|
- files~=(?i)moonshot
|
||||||
|
- title~=(?i)(?:kimi|moonshot)
|
||||||
|
actions:
|
||||||
|
label:
|
||||||
|
add:
|
||||||
|
- kimi
|
||||||
|
|
||||||
|
- name: label-k3
|
||||||
|
description: Automatically apply k3 label (launch triage; retire after ramp-down)
|
||||||
|
conditions:
|
||||||
|
- label != stale
|
||||||
|
- or:
|
||||||
|
- files~=(?i)kimi[-_]?k3
|
||||||
|
- title~=(?i)(?:kimi[-\s]?k3|\bk3\b)
|
||||||
|
actions:
|
||||||
|
label:
|
||||||
|
add:
|
||||||
|
- k3
|
||||||
|
|
||||||
- name: label-nvidia
|
- name: label-nvidia
|
||||||
description: Automatically apply nvidia label
|
description: Automatically apply nvidia label
|
||||||
conditions:
|
conditions:
|
||||||
|
|||||||
@@ -130,6 +130,66 @@ jobs:
|
|||||||
},
|
},
|
||||||
],
|
],
|
||||||
},
|
},
|
||||||
|
kimi: {
|
||||||
|
keywords: [
|
||||||
|
{ term: "Kimi", searchIn: "both" },
|
||||||
|
{ term: "Moonshot", searchIn: "both" },
|
||||||
|
],
|
||||||
|
substrings: [
|
||||||
|
{ term: "moonshotai/", searchIn: "both" },
|
||||||
|
{ term: "kimi", searchIn: "title" },
|
||||||
|
],
|
||||||
|
},
|
||||||
|
k3: {
|
||||||
|
keywords: [
|
||||||
|
{ term: "Kimi K3", searchIn: "both" },
|
||||||
|
{ term: "K3", searchIn: "title" },
|
||||||
|
],
|
||||||
|
substrings: [
|
||||||
|
{ term: "moonshotai/kimi-k3", searchIn: "both" },
|
||||||
|
],
|
||||||
|
},
|
||||||
|
quantization: {
|
||||||
|
keywords: [
|
||||||
|
{
|
||||||
|
term: "quantization",
|
||||||
|
searchIn: "both"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
term: "quantized",
|
||||||
|
searchIn: "both"
|
||||||
|
},
|
||||||
|
],
|
||||||
|
},
|
||||||
|
"intel-gpu": {
|
||||||
|
// Keyword search - matches whole words only (with word boundaries)
|
||||||
|
keywords: [
|
||||||
|
{
|
||||||
|
term: "B50",
|
||||||
|
searchIn: "both"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
term: "B60",
|
||||||
|
searchIn: "both"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
term: "B70",
|
||||||
|
searchIn: "both"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
term: "intel gpu",
|
||||||
|
searchIn: "both"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
term: "Arc GPU",
|
||||||
|
searchIn: "both"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
term: "BMG",
|
||||||
|
searchIn: "both"
|
||||||
|
},
|
||||||
|
],
|
||||||
|
},
|
||||||
// Add more label configurations here as needed
|
// Add more label configurations here as needed
|
||||||
// example: {
|
// example: {
|
||||||
// keywords: [...],
|
// keywords: [...],
|
||||||
@@ -323,7 +383,7 @@ jobs:
|
|||||||
// {users} will be replaced with @mentions
|
// {users} will be replaced with @mentions
|
||||||
const ccConfig = {
|
const ccConfig = {
|
||||||
rocm: {
|
rocm: {
|
||||||
users: ['hongxiayang', 'tjtanaa', 'vllmellm'],
|
users: ['hongxiayang', 'tjtanaa', 'vllmellm', 'giuseppegrossi'],
|
||||||
message: 'CC {users} for ROCm-related issue',
|
message: 'CC {users} for ROCm-related issue',
|
||||||
},
|
},
|
||||||
mistral: {
|
mistral: {
|
||||||
|
|||||||
@@ -80,9 +80,9 @@ jobs:
|
|||||||
'',
|
'',
|
||||||
'\u{1f4ac} Join our developer Slack at https://slack.vllm.ai to discuss your PR in `#pr-reviews`, coordinate on features in `#feat-` channels, or join special interest groups in `#sig-` channels.',
|
'\u{1f4ac} Join our developer Slack at https://slack.vllm.ai to discuss your PR in `#pr-reviews`, coordinate on features in `#feat-` channels, or join special interest groups in `#sig-` channels.',
|
||||||
'',
|
'',
|
||||||
'PRs do not trigger a full CI run by default. Once the PR is approved and ready to go, your PR reviewer(s) can run CI to test the changes comprehensively before merging.',
|
'PRs do not trigger a full CI run by default. Reviewers with write access and configured trusted contributors can comment `/ci run` whenever CI signals are needed.',
|
||||||
'',
|
'',
|
||||||
'To run CI, PR reviewers can either: Add `ready` label to the PR or enable auto-merge.',
|
'Once the PR is approved or has the `ready` label, the PR author can also use `/ci run` or `/ci retry`. New commits do not start CI automatically.',
|
||||||
'',
|
'',
|
||||||
'If you have any questions, please reach out to us on Slack at https://slack.vllm.ai.',
|
'If you have any questions, please reach out to us on Slack at https://slack.vllm.ai.',
|
||||||
'',
|
'',
|
||||||
|
|||||||
@@ -41,7 +41,7 @@ jobs:
|
|||||||
if (hasReadyLabel || hasVerifiedLabel || mergedCount >= 4) {
|
if (hasReadyLabel || hasVerifiedLabel || mergedCount >= 4) {
|
||||||
core.info(`Check passed: verified label=${hasVerifiedLabel}, ready label=${hasReadyLabel}, 4+ merged PRs=${mergedCount >= 4}`);
|
core.info(`Check passed: verified label=${hasVerifiedLabel}, ready label=${hasReadyLabel}, 4+ merged PRs=${mergedCount >= 4}`);
|
||||||
} else {
|
} else {
|
||||||
core.setFailed(`PR must have the 'verified', 'ready', or 'ready-run-all-tests' label (the ready labels also trigger tests) or the author must have at least 4 merged PRs (found ${mergedCount}).`);
|
core.setFailed(`PR must have the 'verified', 'ready', or 'ready-run-all-tests' label to run pre-commit, or the author must have at least 4 merged PRs (found ${mergedCount}).`);
|
||||||
}
|
}
|
||||||
|
|
||||||
pre-commit:
|
pre-commit:
|
||||||
|
|||||||
@@ -0,0 +1,40 @@
|
|||||||
|
name: Run CI from PR comment
|
||||||
|
|
||||||
|
on:
|
||||||
|
issue_comment:
|
||||||
|
types: [created]
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: run-ci-comment-${{ github.event.issue.number }}
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
issues: write
|
||||||
|
pull-requests: write
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
run-ci-command:
|
||||||
|
if: >-
|
||||||
|
github.event.issue.pull_request &&
|
||||||
|
(github.event.comment.body == '/ci run' ||
|
||||||
|
github.event.comment.body == '/ci retry')
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||||
|
with:
|
||||||
|
ref: ${{ github.event.repository.default_branch }}
|
||||||
|
persist-credentials: false
|
||||||
|
- uses: astral-sh/setup-uv@37802adc94f370d6bfd71619e3f0bf239e1f3b78 # v7.6.0
|
||||||
|
with:
|
||||||
|
python-version: "3.12"
|
||||||
|
- name: Authorize and run CI command
|
||||||
|
run: >-
|
||||||
|
uv run --no-project --python 3.12
|
||||||
|
.github/workflows/scripts/run_ci_command.py
|
||||||
|
env:
|
||||||
|
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||||
|
BUILDKITE_ORGANIZATION: vllm
|
||||||
|
BUILDKITE_PIPELINE: ci
|
||||||
|
CI_TRUSTED_USERS: ${{ vars.CI_TRUSTED_USERS }}
|
||||||
|
GH_TOKEN: ${{ github.token }}
|
||||||
@@ -0,0 +1,581 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import urllib.error
|
||||||
|
import urllib.parse
|
||||||
|
import urllib.request
|
||||||
|
from collections.abc import Mapping, Sequence
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
COMMAND_RUN_CI = "/ci run"
|
||||||
|
COMMAND_RETRY_FAILED = "/ci retry"
|
||||||
|
READY_LABELS = {"ready", "ready-run-all-tests"}
|
||||||
|
TRUSTED_PERMISSIONS = {"admin", "maintain", "write"}
|
||||||
|
ACTIVE_BUILD_STATES = {
|
||||||
|
"blocked",
|
||||||
|
"creating",
|
||||||
|
"scheduled",
|
||||||
|
"running",
|
||||||
|
"failing",
|
||||||
|
"canceling",
|
||||||
|
"waiting",
|
||||||
|
"waiting_failed",
|
||||||
|
}
|
||||||
|
RETRY_STATES = "failed,timed_out,expired"
|
||||||
|
|
||||||
|
|
||||||
|
class ApiError(RuntimeError):
|
||||||
|
def __init__(self, status: int | None, message: str) -> None:
|
||||||
|
super().__init__(message)
|
||||||
|
self.status = status
|
||||||
|
|
||||||
|
|
||||||
|
class HttpTransport:
|
||||||
|
def request(
|
||||||
|
self,
|
||||||
|
url: str,
|
||||||
|
*,
|
||||||
|
body: Mapping[str, Any] | None = None,
|
||||||
|
headers: Mapping[str, str] | None = None,
|
||||||
|
method: str = "GET",
|
||||||
|
) -> Any:
|
||||||
|
data = None if body is None else json.dumps(body).encode()
|
||||||
|
request = urllib.request.Request(
|
||||||
|
url,
|
||||||
|
data=data,
|
||||||
|
headers=dict(headers or {}),
|
||||||
|
method=method,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
with urllib.request.urlopen(request, timeout=30) as response:
|
||||||
|
response_body = response.read().decode()
|
||||||
|
except urllib.error.HTTPError as error:
|
||||||
|
response_body = error.read().decode()
|
||||||
|
message = self._error_message(response_body, error.reason)
|
||||||
|
raise ApiError(
|
||||||
|
error.code,
|
||||||
|
f"API returned {error.code}: {message}",
|
||||||
|
) from error
|
||||||
|
except urllib.error.URLError as error:
|
||||||
|
raise ApiError(None, f"API request failed: {error.reason}") from error
|
||||||
|
|
||||||
|
if not response_body:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return json.loads(response_body)
|
||||||
|
except json.JSONDecodeError as error:
|
||||||
|
raise ApiError(None, "API returned a non-JSON response.") from error
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _error_message(response_body: str, fallback: str) -> str:
|
||||||
|
try:
|
||||||
|
parsed = json.loads(response_body)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
return fallback
|
||||||
|
return str(parsed.get("message", fallback))
|
||||||
|
|
||||||
|
|
||||||
|
class GitHubClient:
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
token: str,
|
||||||
|
repository: str,
|
||||||
|
transport: HttpTransport | None = None,
|
||||||
|
) -> None:
|
||||||
|
if not token:
|
||||||
|
raise RuntimeError("GH_TOKEN is not set.")
|
||||||
|
self.owner, self.repo = repository.split("/", maxsplit=1)
|
||||||
|
self.transport = transport or HttpTransport()
|
||||||
|
self.headers = {
|
||||||
|
"Accept": "application/vnd.github+json",
|
||||||
|
"Authorization": f"Bearer {token}",
|
||||||
|
"Content-Type": "application/json",
|
||||||
|
"User-Agent": "vllm-ci-command",
|
||||||
|
"X-GitHub-Api-Version": "2022-11-28",
|
||||||
|
}
|
||||||
|
|
||||||
|
def _request(
|
||||||
|
self,
|
||||||
|
path: str,
|
||||||
|
*,
|
||||||
|
body: Mapping[str, Any] | None = None,
|
||||||
|
method: str = "GET",
|
||||||
|
) -> Any:
|
||||||
|
return self.transport.request(
|
||||||
|
f"https://api.github.com{path}",
|
||||||
|
body=body,
|
||||||
|
headers=self.headers,
|
||||||
|
method=method,
|
||||||
|
)
|
||||||
|
|
||||||
|
def _repo_path(self, suffix: str) -> str:
|
||||||
|
owner = urllib.parse.quote(self.owner, safe="")
|
||||||
|
repo = urllib.parse.quote(self.repo, safe="")
|
||||||
|
return f"/repos/{owner}/{repo}{suffix}"
|
||||||
|
|
||||||
|
def _paginate(self, path: str) -> list[dict[str, Any]]:
|
||||||
|
results: list[dict[str, Any]] = []
|
||||||
|
separator = "&" if "?" in path else "?"
|
||||||
|
for page in range(1, 101):
|
||||||
|
response = self._request(f"{path}{separator}per_page=100&page={page}")
|
||||||
|
if not isinstance(response, list):
|
||||||
|
raise ApiError(None, "GitHub API returned an invalid list response.")
|
||||||
|
results.extend(response)
|
||||||
|
if len(response) < 100:
|
||||||
|
return results
|
||||||
|
raise ApiError(None, "GitHub API pagination exceeded 10,000 results.")
|
||||||
|
|
||||||
|
def get_pr(self, number: int) -> dict[str, Any]:
|
||||||
|
return self._request(self._repo_path(f"/pulls/{number}"))
|
||||||
|
|
||||||
|
def get_permission(self, actor: str) -> str:
|
||||||
|
username = urllib.parse.quote(actor, safe="")
|
||||||
|
try:
|
||||||
|
response = self._request(
|
||||||
|
self._repo_path(f"/collaborators/{username}/permission")
|
||||||
|
)
|
||||||
|
except ApiError as error:
|
||||||
|
if error.status == 404:
|
||||||
|
return "none"
|
||||||
|
raise
|
||||||
|
return str(response["permission"])
|
||||||
|
|
||||||
|
def get_review_decision(self, number: int) -> str | None:
|
||||||
|
query = """
|
||||||
|
query($owner: String!, $repo: String!, $number: Int!) {
|
||||||
|
repository(owner: $owner, name: $repo) {
|
||||||
|
pullRequest(number: $number) {
|
||||||
|
reviewDecision
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
"""
|
||||||
|
response = self._request(
|
||||||
|
"/graphql",
|
||||||
|
body={
|
||||||
|
"query": query,
|
||||||
|
"variables": {
|
||||||
|
"number": number,
|
||||||
|
"owner": self.owner,
|
||||||
|
"repo": self.repo,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
method="POST",
|
||||||
|
)
|
||||||
|
return response["data"]["repository"]["pullRequest"]["reviewDecision"]
|
||||||
|
|
||||||
|
def list_reviews(self, number: int) -> list[dict[str, Any]]:
|
||||||
|
return self._paginate(self._repo_path(f"/pulls/{number}/reviews"))
|
||||||
|
|
||||||
|
def list_reactions(self, comment_id: int) -> list[dict[str, Any]]:
|
||||||
|
return self._paginate(
|
||||||
|
self._repo_path(f"/issues/comments/{comment_id}/reactions")
|
||||||
|
)
|
||||||
|
|
||||||
|
def add_reaction(self, comment_id: int, content: str) -> None:
|
||||||
|
self._request(
|
||||||
|
self._repo_path(f"/issues/comments/{comment_id}/reactions"),
|
||||||
|
body={"content": content},
|
||||||
|
method="POST",
|
||||||
|
)
|
||||||
|
|
||||||
|
def add_comment(self, issue_number: int, body: str) -> None:
|
||||||
|
self._request(
|
||||||
|
self._repo_path(f"/issues/{issue_number}/comments"),
|
||||||
|
body={"body": body},
|
||||||
|
method="POST",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class BuildkiteClient:
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
token: str,
|
||||||
|
organization: str,
|
||||||
|
pipeline: str,
|
||||||
|
transport: HttpTransport | None = None,
|
||||||
|
) -> None:
|
||||||
|
self.token = token
|
||||||
|
self.transport = transport or HttpTransport()
|
||||||
|
organization = urllib.parse.quote(organization, safe="")
|
||||||
|
pipeline = urllib.parse.quote(pipeline, safe="")
|
||||||
|
self.base_url = (
|
||||||
|
"https://api.buildkite.com/v2/organizations/"
|
||||||
|
f"{organization}/pipelines/{pipeline}/builds"
|
||||||
|
)
|
||||||
|
|
||||||
|
def _request(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
body: Mapping[str, Any] | None = None,
|
||||||
|
method: str = "GET",
|
||||||
|
path: str = "",
|
||||||
|
query: Sequence[tuple[str, str]] = (),
|
||||||
|
) -> Any:
|
||||||
|
if not self.token:
|
||||||
|
raise RuntimeError("The BUILDKITE_API_TOKEN repository secret is not set.")
|
||||||
|
url = f"{self.base_url}{path}"
|
||||||
|
if query:
|
||||||
|
url = f"{url}?{urllib.parse.urlencode(query)}"
|
||||||
|
return self.transport.request(
|
||||||
|
url,
|
||||||
|
body=body,
|
||||||
|
headers={
|
||||||
|
"Authorization": f"Bearer {self.token}",
|
||||||
|
"Content-Type": "application/json",
|
||||||
|
"User-Agent": "vllm-ci-command",
|
||||||
|
},
|
||||||
|
method=method,
|
||||||
|
)
|
||||||
|
|
||||||
|
def list_builds(
|
||||||
|
self,
|
||||||
|
commit: str,
|
||||||
|
*,
|
||||||
|
metadata: tuple[str, str] | None = None,
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
query = [
|
||||||
|
("commit", commit),
|
||||||
|
("exclude_jobs", "true"),
|
||||||
|
("exclude_pipeline", "true"),
|
||||||
|
("per_page", "100"),
|
||||||
|
]
|
||||||
|
if metadata:
|
||||||
|
key, value = metadata
|
||||||
|
query.append((f"meta_data[{key}]", value))
|
||||||
|
response = self._request(query=query)
|
||||||
|
if not isinstance(response, list):
|
||||||
|
raise ApiError(None, "Buildkite API returned an invalid build list.")
|
||||||
|
return response
|
||||||
|
|
||||||
|
def create_build(self, body: Mapping[str, Any]) -> dict[str, Any]:
|
||||||
|
return self._request(body=body, method="POST")
|
||||||
|
|
||||||
|
def retry_failed_jobs(
|
||||||
|
self,
|
||||||
|
build_number: int,
|
||||||
|
states: str,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
number = urllib.parse.quote(str(build_number), safe="")
|
||||||
|
return self._request(
|
||||||
|
body={"states": states},
|
||||||
|
method="PUT",
|
||||||
|
path=f"/{number}/retry_failed_jobs",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_command(body: str) -> str | None:
|
||||||
|
if body in {COMMAND_RUN_CI, COMMAND_RETRY_FAILED}:
|
||||||
|
return body
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def parse_trusted_users(value: str = "") -> set[str]:
|
||||||
|
return {
|
||||||
|
user.casefold() for item in value.split(",") for user in item.split() if user
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def has_ready_label(pr: Mapping[str, Any]) -> bool:
|
||||||
|
return any(label["name"] in READY_LABELS for label in pr["labels"])
|
||||||
|
|
||||||
|
|
||||||
|
def is_trusted_permission(permission: str) -> bool:
|
||||||
|
return permission in TRUSTED_PERMISSIONS
|
||||||
|
|
||||||
|
|
||||||
|
def authorize(
|
||||||
|
*,
|
||||||
|
actor: str,
|
||||||
|
permission: str,
|
||||||
|
pr: Mapping[str, Any],
|
||||||
|
trusted_approval: bool = False,
|
||||||
|
trusted_users: set[str] | None = None,
|
||||||
|
) -> tuple[bool, str]:
|
||||||
|
trusted_users = trusted_users or set()
|
||||||
|
if is_trusted_permission(permission):
|
||||||
|
return True, f"repository {permission} permission"
|
||||||
|
if actor.casefold() in trusted_users:
|
||||||
|
return True, "configured trusted contributor"
|
||||||
|
if actor.casefold() != pr["user"]["login"].casefold():
|
||||||
|
return (
|
||||||
|
False,
|
||||||
|
"Only reviewers with write access can run CI before it is "
|
||||||
|
"delegated to the PR author.",
|
||||||
|
)
|
||||||
|
if pr["draft"]:
|
||||||
|
return False, "PR authors cannot run CI while the PR is a draft."
|
||||||
|
if has_ready_label(pr):
|
||||||
|
return True, "ready label"
|
||||||
|
if trusted_approval:
|
||||||
|
return True, "approval from a trusted reviewer"
|
||||||
|
return (
|
||||||
|
False,
|
||||||
|
"A reviewer with write access must run `/ci run`, approve the PR, "
|
||||||
|
"or add the `ready` label first.",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def has_trusted_approval(
|
||||||
|
github: GitHubClient,
|
||||||
|
number: int,
|
||||||
|
trusted_users: set[str],
|
||||||
|
) -> bool:
|
||||||
|
if github.get_review_decision(number) != "APPROVED":
|
||||||
|
return False
|
||||||
|
|
||||||
|
latest_review_states: dict[str, tuple[str, str]] = {}
|
||||||
|
for review in github.list_reviews(number):
|
||||||
|
user = review.get("user") or {}
|
||||||
|
login = user.get("login")
|
||||||
|
state = review.get("state")
|
||||||
|
if login and state in {"APPROVED", "CHANGES_REQUESTED", "DISMISSED"}:
|
||||||
|
latest_review_states[login.casefold()] = (login, state)
|
||||||
|
|
||||||
|
for login, state in latest_review_states.values():
|
||||||
|
if state != "APPROVED":
|
||||||
|
continue
|
||||||
|
if login.casefold() in trusted_users:
|
||||||
|
return True
|
||||||
|
if is_trusted_permission(github.get_permission(login)):
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def is_build_for_pr(build: Mapping[str, Any], pr_number: int) -> bool:
|
||||||
|
pull_request = build.get("pull_request")
|
||||||
|
if isinstance(pull_request, Mapping):
|
||||||
|
build_pr_number = pull_request.get("id", pull_request.get("number"))
|
||||||
|
if build_pr_number is not None:
|
||||||
|
return str(build_pr_number) == str(pr_number)
|
||||||
|
metadata = build.get("meta_data") or {}
|
||||||
|
return str(metadata.get("github-pr-number")) == str(pr_number)
|
||||||
|
|
||||||
|
|
||||||
|
def is_active_build(build: Mapping[str, Any]) -> bool:
|
||||||
|
return bool(build.get("blocked")) or build.get("state") in ACTIVE_BUILD_STATES
|
||||||
|
|
||||||
|
|
||||||
|
def select_latest_build(
|
||||||
|
builds: Sequence[dict[str, Any]],
|
||||||
|
pr_number: int,
|
||||||
|
) -> dict[str, Any] | None:
|
||||||
|
matching = [build for build in builds if is_build_for_pr(build, pr_number)]
|
||||||
|
return max(matching, key=lambda build: build.get("created_at", ""), default=None)
|
||||||
|
|
||||||
|
|
||||||
|
def create_build_payload(
|
||||||
|
*,
|
||||||
|
actor: str,
|
||||||
|
comment_id: int,
|
||||||
|
pr: Mapping[str, Any],
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"commit": pr["head"]["sha"],
|
||||||
|
"branch": pr["head"]["ref"],
|
||||||
|
"message": f"PR #{pr['number']} {COMMAND_RUN_CI} by @{actor}",
|
||||||
|
"pull_request_id": pr["number"],
|
||||||
|
"pull_request_base_branch": pr["base"]["ref"],
|
||||||
|
"pull_request_repository": pr["head"]["repo"]["clone_url"],
|
||||||
|
"pull_request_labels": [label["name"] for label in pr["labels"]],
|
||||||
|
"ignore_pipeline_branch_filters": True,
|
||||||
|
"env": {
|
||||||
|
"VLLM_CI_GITHUB_COMMENT_ID": str(comment_id),
|
||||||
|
"VLLM_CI_TRIGGERED_BY": actor,
|
||||||
|
},
|
||||||
|
"meta_data": {
|
||||||
|
"github-comment-id": str(comment_id),
|
||||||
|
"github-pr-number": str(pr["number"]),
|
||||||
|
"github-triggered-by": actor,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def add_reaction_safely(
|
||||||
|
github: GitHubClient,
|
||||||
|
comment_id: int,
|
||||||
|
content: str,
|
||||||
|
) -> None:
|
||||||
|
try:
|
||||||
|
github.add_reaction(comment_id, content)
|
||||||
|
except Exception as error:
|
||||||
|
print(f"Could not add {content} reaction: {error}", file=sys.stderr)
|
||||||
|
|
||||||
|
|
||||||
|
def is_already_handled(github: GitHubClient, comment_id: int) -> bool:
|
||||||
|
return any(
|
||||||
|
reaction.get("content") in {"rocket", "-1"}
|
||||||
|
and (reaction.get("user") or {}).get("login") == "github-actions[bot]"
|
||||||
|
for reaction in github.list_reactions(comment_id)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def handle_run_ci(
|
||||||
|
*,
|
||||||
|
actor: str,
|
||||||
|
buildkite: BuildkiteClient,
|
||||||
|
comment_id: int,
|
||||||
|
github: GitHubClient,
|
||||||
|
pr: Mapping[str, Any],
|
||||||
|
) -> str:
|
||||||
|
duplicate_builds = buildkite.list_builds(
|
||||||
|
pr["head"]["sha"],
|
||||||
|
metadata=("github-comment-id", str(comment_id)),
|
||||||
|
)
|
||||||
|
duplicate = select_latest_build(duplicate_builds, pr["number"])
|
||||||
|
if duplicate:
|
||||||
|
return f"CI was already requested by this comment: {duplicate['web_url']}"
|
||||||
|
|
||||||
|
current_builds = buildkite.list_builds(pr["head"]["sha"])
|
||||||
|
active_build = next(
|
||||||
|
(
|
||||||
|
build
|
||||||
|
for build in current_builds
|
||||||
|
if is_build_for_pr(build, pr["number"]) and is_active_build(build)
|
||||||
|
),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
if active_build:
|
||||||
|
return f"CI is already running for this commit: {active_build['web_url']}"
|
||||||
|
|
||||||
|
current_pr = github.get_pr(pr["number"])
|
||||||
|
if current_pr["state"] != "open" or current_pr["head"]["sha"] != pr["head"]["sha"]:
|
||||||
|
return (
|
||||||
|
"The PR head changed while processing the command. Comment `/ci run` again."
|
||||||
|
)
|
||||||
|
|
||||||
|
build = buildkite.create_build(
|
||||||
|
create_build_payload(
|
||||||
|
actor=actor,
|
||||||
|
comment_id=comment_id,
|
||||||
|
pr=current_pr,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return (
|
||||||
|
f"Triggered [Buildkite CI #{build['number']}]({build['web_url']}) "
|
||||||
|
f"for commit `{current_pr['head']['sha'][:12]}`."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def handle_retry_failed(
|
||||||
|
*,
|
||||||
|
buildkite: BuildkiteClient,
|
||||||
|
pr: Mapping[str, Any],
|
||||||
|
) -> str:
|
||||||
|
builds = buildkite.list_builds(pr["head"]["sha"])
|
||||||
|
build = select_latest_build(builds, pr["number"])
|
||||||
|
if not build:
|
||||||
|
return "No CI build exists for the current PR commit. Use `/ci run` first."
|
||||||
|
if not build.get("finished_at") or is_active_build(build):
|
||||||
|
return f"CI is still running for this commit: {build['web_url']}"
|
||||||
|
|
||||||
|
retried = buildkite.retry_failed_jobs(build["number"], RETRY_STATES)
|
||||||
|
if retried["retried_jobs_count"] == 0:
|
||||||
|
return (
|
||||||
|
f"No failed, timed-out, or expired jobs need retrying: {build['web_url']}"
|
||||||
|
)
|
||||||
|
return (
|
||||||
|
f"Queued {retried['retried_jobs_count']} failed job(s) for retry in "
|
||||||
|
f"[Buildkite CI #{build['number']}]({build['web_url']})."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def run(
|
||||||
|
event: Mapping[str, Any],
|
||||||
|
github: GitHubClient,
|
||||||
|
buildkite: BuildkiteClient,
|
||||||
|
trusted_users_value: str = "",
|
||||||
|
) -> None:
|
||||||
|
command = parse_command(event["comment"]["body"])
|
||||||
|
if not command or "pull_request" not in event["issue"]:
|
||||||
|
return
|
||||||
|
|
||||||
|
issue_number = event["issue"]["number"]
|
||||||
|
comment_id = event["comment"]["id"]
|
||||||
|
actor = event["comment"]["user"]["login"]
|
||||||
|
|
||||||
|
if is_already_handled(github, comment_id):
|
||||||
|
print(f"Comment {comment_id} was already handled.")
|
||||||
|
return
|
||||||
|
add_reaction_safely(github, comment_id, "eyes")
|
||||||
|
|
||||||
|
try:
|
||||||
|
pr = github.get_pr(issue_number)
|
||||||
|
permission = github.get_permission(actor)
|
||||||
|
if pr["state"] != "open":
|
||||||
|
github.add_comment(issue_number, "CI commands require an open PR.")
|
||||||
|
return
|
||||||
|
|
||||||
|
trusted_users = parse_trusted_users(trusted_users_value)
|
||||||
|
should_check_approval = (
|
||||||
|
not is_trusted_permission(permission)
|
||||||
|
and actor.casefold() not in trusted_users
|
||||||
|
and actor.casefold() == pr["user"]["login"].casefold()
|
||||||
|
and not pr["draft"]
|
||||||
|
and not has_ready_label(pr)
|
||||||
|
)
|
||||||
|
trusted_approval = should_check_approval and has_trusted_approval(
|
||||||
|
github,
|
||||||
|
issue_number,
|
||||||
|
trusted_users,
|
||||||
|
)
|
||||||
|
allowed, reason = authorize(
|
||||||
|
actor=actor,
|
||||||
|
permission=permission,
|
||||||
|
pr=pr,
|
||||||
|
trusted_approval=trusted_approval,
|
||||||
|
trusted_users=trusted_users,
|
||||||
|
)
|
||||||
|
if not allowed:
|
||||||
|
add_reaction_safely(github, comment_id, "-1")
|
||||||
|
github.add_comment(issue_number, f"@{actor}, {reason}")
|
||||||
|
return
|
||||||
|
|
||||||
|
print(f"Authorized @{actor}: {reason}")
|
||||||
|
if command == COMMAND_RUN_CI:
|
||||||
|
message = handle_run_ci(
|
||||||
|
actor=actor,
|
||||||
|
buildkite=buildkite,
|
||||||
|
comment_id=comment_id,
|
||||||
|
github=github,
|
||||||
|
pr=pr,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
message = handle_retry_failed(buildkite=buildkite, pr=pr)
|
||||||
|
add_reaction_safely(github, comment_id, "rocket")
|
||||||
|
github.add_comment(issue_number, message)
|
||||||
|
except Exception:
|
||||||
|
add_reaction_safely(github, comment_id, "confused")
|
||||||
|
raise
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
event_path = os.environ["GITHUB_EVENT_PATH"]
|
||||||
|
with open(event_path, encoding="utf-8") as event_file:
|
||||||
|
event = json.load(event_file)
|
||||||
|
|
||||||
|
if not parse_command(event["comment"]["body"]):
|
||||||
|
return
|
||||||
|
|
||||||
|
github = GitHubClient(
|
||||||
|
os.environ.get("GH_TOKEN", ""),
|
||||||
|
os.environ["GITHUB_REPOSITORY"],
|
||||||
|
)
|
||||||
|
buildkite = BuildkiteClient(
|
||||||
|
os.environ.get("BUILDKITE_API_TOKEN", ""),
|
||||||
|
os.environ.get("BUILDKITE_ORGANIZATION", "vllm"),
|
||||||
|
os.environ.get("BUILDKITE_PIPELINE", "ci"),
|
||||||
|
)
|
||||||
|
run(
|
||||||
|
event,
|
||||||
|
github,
|
||||||
|
buildkite,
|
||||||
|
os.environ.get("CI_TRUSTED_USERS", ""),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,363 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
|
||||||
|
import unittest
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from run_ci_command import (
|
||||||
|
COMMAND_RETRY_FAILED,
|
||||||
|
COMMAND_RUN_CI,
|
||||||
|
RETRY_STATES,
|
||||||
|
BuildkiteClient,
|
||||||
|
authorize,
|
||||||
|
create_build_payload,
|
||||||
|
has_trusted_approval,
|
||||||
|
is_active_build,
|
||||||
|
is_build_for_pr,
|
||||||
|
parse_command,
|
||||||
|
parse_trusted_users,
|
||||||
|
run,
|
||||||
|
select_latest_build,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def make_pr(**overrides: Any) -> dict[str, Any]:
|
||||||
|
pr = {
|
||||||
|
"base": {"ref": "main"},
|
||||||
|
"draft": False,
|
||||||
|
"head": {
|
||||||
|
"ref": "feature",
|
||||||
|
"repo": {"clone_url": "https://github.com/contributor/vllm.git"},
|
||||||
|
"sha": "0123456789abcdef",
|
||||||
|
},
|
||||||
|
"labels": [],
|
||||||
|
"number": 42,
|
||||||
|
"state": "open",
|
||||||
|
"user": {"login": "author"},
|
||||||
|
}
|
||||||
|
pr.update(overrides)
|
||||||
|
return pr
|
||||||
|
|
||||||
|
|
||||||
|
def make_event(command: str, actor: str = "reviewer") -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"comment": {
|
||||||
|
"body": command,
|
||||||
|
"id": 99,
|
||||||
|
"user": {"login": actor},
|
||||||
|
},
|
||||||
|
"issue": {
|
||||||
|
"number": 42,
|
||||||
|
"pull_request": {},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class FakeGitHub:
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
permission: str = "write",
|
||||||
|
permissions: dict[str, str] | None = None,
|
||||||
|
pr: dict[str, Any] | None = None,
|
||||||
|
review_decision: str = "REVIEW_REQUIRED",
|
||||||
|
reviews: list[dict[str, Any]] | None = None,
|
||||||
|
) -> None:
|
||||||
|
self.comments: list[str] = []
|
||||||
|
self.permission = permission
|
||||||
|
self.permissions = permissions or {}
|
||||||
|
self.pr = pr or make_pr()
|
||||||
|
self.reactions: list[str] = []
|
||||||
|
self.review_decision = review_decision
|
||||||
|
self.reviews = reviews or []
|
||||||
|
|
||||||
|
def get_pr(self, number: int) -> dict[str, Any]:
|
||||||
|
return self.pr
|
||||||
|
|
||||||
|
def get_permission(self, actor: str) -> str:
|
||||||
|
return self.permissions.get(actor, self.permission)
|
||||||
|
|
||||||
|
def get_review_decision(self, number: int) -> str:
|
||||||
|
return self.review_decision
|
||||||
|
|
||||||
|
def list_reviews(self, number: int) -> list[dict[str, Any]]:
|
||||||
|
return self.reviews
|
||||||
|
|
||||||
|
def list_reactions(self, comment_id: int) -> list[dict[str, Any]]:
|
||||||
|
return []
|
||||||
|
|
||||||
|
def add_reaction(self, comment_id: int, content: str) -> None:
|
||||||
|
self.reactions.append(content)
|
||||||
|
|
||||||
|
def add_comment(self, issue_number: int, body: str) -> None:
|
||||||
|
self.comments.append(body)
|
||||||
|
|
||||||
|
|
||||||
|
class FakeBuildkite:
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
build_lists: list[list[dict[str, Any]]] | None = None,
|
||||||
|
) -> None:
|
||||||
|
self.build_lists = build_lists or []
|
||||||
|
self.created_builds: list[dict[str, Any]] = []
|
||||||
|
self.list_calls: list[tuple[str, tuple[str, str] | None]] = []
|
||||||
|
self.retry_calls: list[tuple[int, str]] = []
|
||||||
|
|
||||||
|
def list_builds(
|
||||||
|
self,
|
||||||
|
commit: str,
|
||||||
|
*,
|
||||||
|
metadata: tuple[str, str] | None = None,
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
self.list_calls.append((commit, metadata))
|
||||||
|
return self.build_lists.pop(0)
|
||||||
|
|
||||||
|
def create_build(self, body: dict[str, Any]) -> dict[str, Any]:
|
||||||
|
self.created_builds.append(body)
|
||||||
|
return {
|
||||||
|
"number": 123,
|
||||||
|
"web_url": "https://buildkite.example/builds/123",
|
||||||
|
}
|
||||||
|
|
||||||
|
def retry_failed_jobs(
|
||||||
|
self,
|
||||||
|
build_number: int,
|
||||||
|
states: str,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
self.retry_calls.append((build_number, states))
|
||||||
|
return {"retried_jobs_count": 3}
|
||||||
|
|
||||||
|
|
||||||
|
class FakeTransport:
|
||||||
|
def __init__(self, response: Any) -> None:
|
||||||
|
self.calls: list[dict[str, Any]] = []
|
||||||
|
self.response = response
|
||||||
|
|
||||||
|
def request(self, url: str, **kwargs: Any) -> Any:
|
||||||
|
self.calls.append({"url": url, **kwargs})
|
||||||
|
return self.response
|
||||||
|
|
||||||
|
|
||||||
|
class RunCiCommandTest(unittest.TestCase):
|
||||||
|
def test_only_exact_ci_commands_are_accepted(self) -> None:
|
||||||
|
self.assertEqual(parse_command(COMMAND_RUN_CI), COMMAND_RUN_CI)
|
||||||
|
self.assertEqual(
|
||||||
|
parse_command(COMMAND_RETRY_FAILED),
|
||||||
|
COMMAND_RETRY_FAILED,
|
||||||
|
)
|
||||||
|
self.assertIsNone(parse_command("/ci run please"))
|
||||||
|
self.assertIsNone(parse_command(" /ci run"))
|
||||||
|
|
||||||
|
def test_write_access_authorizes_reviewers_and_authors(self) -> None:
|
||||||
|
allowed, _ = authorize(
|
||||||
|
actor="reviewer",
|
||||||
|
permission="write",
|
||||||
|
pr=make_pr(),
|
||||||
|
)
|
||||||
|
self.assertTrue(allowed)
|
||||||
|
|
||||||
|
def test_configured_trusted_contributors_can_run_ci(self) -> None:
|
||||||
|
trusted_users = parse_trusted_users("trusted-one, TRUSTED-TWO")
|
||||||
|
allowed, _ = authorize(
|
||||||
|
actor="trusted-two",
|
||||||
|
permission="read",
|
||||||
|
pr=make_pr(),
|
||||||
|
trusted_users=trusted_users,
|
||||||
|
)
|
||||||
|
self.assertTrue(allowed)
|
||||||
|
|
||||||
|
def test_authors_need_an_approval_or_ready_label(self) -> None:
|
||||||
|
pending, _ = authorize(
|
||||||
|
actor="author",
|
||||||
|
permission="read",
|
||||||
|
pr=make_pr(),
|
||||||
|
)
|
||||||
|
approved, _ = authorize(
|
||||||
|
actor="author",
|
||||||
|
permission="read",
|
||||||
|
pr=make_pr(),
|
||||||
|
trusted_approval=True,
|
||||||
|
)
|
||||||
|
ready, _ = authorize(
|
||||||
|
actor="author",
|
||||||
|
permission="read",
|
||||||
|
pr=make_pr(labels=[{"name": "ready"}]),
|
||||||
|
)
|
||||||
|
self.assertFalse(pending)
|
||||||
|
self.assertTrue(approved)
|
||||||
|
self.assertTrue(ready)
|
||||||
|
|
||||||
|
def test_non_author_contributors_without_write_are_denied(self) -> None:
|
||||||
|
allowed, _ = authorize(
|
||||||
|
actor="contributor",
|
||||||
|
permission="read",
|
||||||
|
pr=make_pr(),
|
||||||
|
trusted_approval=True,
|
||||||
|
)
|
||||||
|
self.assertFalse(allowed)
|
||||||
|
|
||||||
|
def test_authors_cannot_use_ready_state_on_draft_prs(self) -> None:
|
||||||
|
allowed, _ = authorize(
|
||||||
|
actor="author",
|
||||||
|
permission="read",
|
||||||
|
pr=make_pr(draft=True, labels=[{"name": "ready"}]),
|
||||||
|
trusted_approval=True,
|
||||||
|
)
|
||||||
|
self.assertFalse(allowed)
|
||||||
|
|
||||||
|
def test_only_trusted_reviewers_can_delegate_through_approval(self) -> None:
|
||||||
|
approved_review = {
|
||||||
|
"state": "APPROVED",
|
||||||
|
"user": {"login": "reviewer"},
|
||||||
|
}
|
||||||
|
trusted = FakeGitHub(
|
||||||
|
permission="read",
|
||||||
|
permissions={"reviewer": "write"},
|
||||||
|
review_decision="APPROVED",
|
||||||
|
reviews=[approved_review],
|
||||||
|
)
|
||||||
|
untrusted = FakeGitHub(
|
||||||
|
permission="read",
|
||||||
|
review_decision="APPROVED",
|
||||||
|
reviews=[approved_review],
|
||||||
|
)
|
||||||
|
self.assertTrue(has_trusted_approval(trusted, 42, set()))
|
||||||
|
self.assertFalse(has_trusted_approval(untrusted, 42, set()))
|
||||||
|
|
||||||
|
def test_build_matching_is_scoped_to_the_pr(self) -> None:
|
||||||
|
self.assertTrue(is_build_for_pr({"pull_request": {"id": 42}}, 42))
|
||||||
|
self.assertFalse(is_build_for_pr({"pull_request": {"id": 43}}, 42))
|
||||||
|
self.assertTrue(
|
||||||
|
is_build_for_pr(
|
||||||
|
{"meta_data": {"github-pr-number": "42"}},
|
||||||
|
42,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_latest_build_selection_ignores_other_prs(self) -> None:
|
||||||
|
latest = select_latest_build(
|
||||||
|
[
|
||||||
|
{
|
||||||
|
"created_at": "2026-07-28T02:00:00Z",
|
||||||
|
"number": 3,
|
||||||
|
"pull_request": {"id": 43},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"created_at": "2026-07-28T01:00:00Z",
|
||||||
|
"number": 2,
|
||||||
|
"pull_request": {"id": 42},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"created_at": "2026-07-28T00:00:00Z",
|
||||||
|
"number": 1,
|
||||||
|
"pull_request": {"id": 42},
|
||||||
|
},
|
||||||
|
],
|
||||||
|
42,
|
||||||
|
)
|
||||||
|
self.assertEqual(latest["number"], 2)
|
||||||
|
|
||||||
|
def test_active_build_states_prevent_duplicate_runs(self) -> None:
|
||||||
|
self.assertTrue(is_active_build({"state": "scheduled"}))
|
||||||
|
self.assertTrue(is_active_build({"state": "running"}))
|
||||||
|
self.assertTrue(is_active_build({"state": "waiting"}))
|
||||||
|
self.assertTrue(is_active_build({"blocked": True, "state": "passed"}))
|
||||||
|
self.assertFalse(is_active_build({"state": "failed"}))
|
||||||
|
|
||||||
|
def test_build_payload_preserves_pr_context(self) -> None:
|
||||||
|
payload = create_build_payload(
|
||||||
|
actor="reviewer",
|
||||||
|
comment_id=99,
|
||||||
|
pr=make_pr(labels=[{"name": "ready"}, {"name": "v1"}]),
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
payload,
|
||||||
|
{
|
||||||
|
"commit": "0123456789abcdef",
|
||||||
|
"branch": "feature",
|
||||||
|
"message": "PR #42 /ci run by @reviewer",
|
||||||
|
"pull_request_id": 42,
|
||||||
|
"pull_request_base_branch": "main",
|
||||||
|
"pull_request_repository": ("https://github.com/contributor/vllm.git"),
|
||||||
|
"pull_request_labels": ["ready", "v1"],
|
||||||
|
"ignore_pipeline_branch_filters": True,
|
||||||
|
"env": {
|
||||||
|
"VLLM_CI_GITHUB_COMMENT_ID": "99",
|
||||||
|
"VLLM_CI_TRIGGERED_BY": "reviewer",
|
||||||
|
},
|
||||||
|
"meta_data": {
|
||||||
|
"github-comment-id": "99",
|
||||||
|
"github-pr-number": "42",
|
||||||
|
"github-triggered-by": "reviewer",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_ci_run_dispatches_build_with_current_pr_metadata(self) -> None:
|
||||||
|
github = FakeGitHub()
|
||||||
|
buildkite = FakeBuildkite([[], []])
|
||||||
|
run(make_event(COMMAND_RUN_CI), github, buildkite)
|
||||||
|
|
||||||
|
self.assertEqual(len(buildkite.created_builds), 1)
|
||||||
|
self.assertEqual(
|
||||||
|
buildkite.created_builds[0]["message"],
|
||||||
|
"PR #42 /ci run by @reviewer",
|
||||||
|
)
|
||||||
|
self.assertEqual(github.reactions, ["eyes", "rocket"])
|
||||||
|
self.assertIn("Buildkite CI #123", github.comments[0])
|
||||||
|
|
||||||
|
def test_unapproved_authors_are_denied_without_buildkite(self) -> None:
|
||||||
|
github = FakeGitHub(
|
||||||
|
permission="read",
|
||||||
|
pr=make_pr(),
|
||||||
|
review_decision="REVIEW_REQUIRED",
|
||||||
|
)
|
||||||
|
buildkite = FakeBuildkite()
|
||||||
|
run(make_event(COMMAND_RUN_CI, "author"), github, buildkite)
|
||||||
|
|
||||||
|
self.assertEqual(buildkite.list_calls, [])
|
||||||
|
self.assertEqual(github.reactions, ["eyes", "-1"])
|
||||||
|
self.assertIn("approve the PR", github.comments[0])
|
||||||
|
|
||||||
|
def test_ci_retry_uses_latest_current_sha_build(self) -> None:
|
||||||
|
github = FakeGitHub(
|
||||||
|
permission="read",
|
||||||
|
pr=make_pr(labels=[{"name": "ready"}]),
|
||||||
|
)
|
||||||
|
buildkite = FakeBuildkite(
|
||||||
|
[
|
||||||
|
[
|
||||||
|
{
|
||||||
|
"created_at": "2026-07-28T01:00:00Z",
|
||||||
|
"finished_at": "2026-07-28T02:00:00Z",
|
||||||
|
"number": 123,
|
||||||
|
"pull_request": {"id": 42},
|
||||||
|
"state": "failed",
|
||||||
|
"web_url": "https://buildkite.example/builds/123",
|
||||||
|
}
|
||||||
|
]
|
||||||
|
]
|
||||||
|
)
|
||||||
|
run(make_event(COMMAND_RETRY_FAILED, "author"), github, buildkite)
|
||||||
|
|
||||||
|
self.assertEqual(buildkite.retry_calls, [(123, RETRY_STATES)])
|
||||||
|
self.assertIn("Queued 3 failed job", github.comments[0])
|
||||||
|
|
||||||
|
def test_buildkite_retry_uses_retry_failed_jobs_endpoint(self) -> None:
|
||||||
|
transport = FakeTransport({"retried_jobs_count": 2})
|
||||||
|
client = BuildkiteClient(
|
||||||
|
"secret",
|
||||||
|
"vllm",
|
||||||
|
"ci",
|
||||||
|
transport=transport,
|
||||||
|
)
|
||||||
|
client.retry_failed_jobs(123, RETRY_STATES)
|
||||||
|
|
||||||
|
call = transport.calls[0]
|
||||||
|
self.assertEqual(call["method"], "PUT")
|
||||||
|
self.assertTrue(call["url"].endswith("/123/retry_failed_jobs"))
|
||||||
|
self.assertEqual(call["body"], {"states": RETRY_STATES})
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
+3
-3
@@ -18,6 +18,9 @@ vllm/third_party/deep_gemm/
|
|||||||
# fmha_sm100 vendored package built from source
|
# fmha_sm100 vendored package built from source
|
||||||
vllm/third_party/fmha_sm100/
|
vllm/third_party/fmha_sm100/
|
||||||
|
|
||||||
|
# tml-fa4 vendored package built from source
|
||||||
|
vllm/third_party/tml_fa4/
|
||||||
|
|
||||||
# triton jit
|
# triton jit
|
||||||
.triton
|
.triton
|
||||||
|
|
||||||
@@ -170,9 +173,6 @@ venv.bak/
|
|||||||
|
|
||||||
# mkdocs documentation
|
# mkdocs documentation
|
||||||
/site
|
/site
|
||||||
docs/argparse
|
|
||||||
docs/examples/*
|
|
||||||
!docs/examples/README.md
|
|
||||||
|
|
||||||
# mypy
|
# mypy
|
||||||
.mypy_cache/
|
.mypy_cache/
|
||||||
|
|||||||
@@ -3,6 +3,9 @@ MD007:
|
|||||||
MD013: false
|
MD013: false
|
||||||
MD024:
|
MD024:
|
||||||
siblings_only: true
|
siblings_only: true
|
||||||
|
MD025:
|
||||||
|
# Allow front matter title to be different from the first heading in the document.
|
||||||
|
front_matter_title: ""
|
||||||
MD031:
|
MD031:
|
||||||
list_items: false
|
list_items: false
|
||||||
MD033: false
|
MD033: false
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ default_install_hook_types:
|
|||||||
default_stages:
|
default_stages:
|
||||||
- pre-commit # Run locally
|
- pre-commit # Run locally
|
||||||
- manual # Run in CI
|
- manual # Run in CI
|
||||||
exclude: 'vllm/third_party/.*'
|
exclude: 'vllm/third_party/.*|vllm/models/kimi_k3/nvidia/ops/third_party/.*|vllm/models/kimi_k3/amd/ops/third_party/.*'
|
||||||
repos:
|
repos:
|
||||||
- repo: https://github.com/astral-sh/ruff-pre-commit
|
- repo: https://github.com/astral-sh/ruff-pre-commit
|
||||||
rev: v0.14.0
|
rev: v0.14.0
|
||||||
@@ -30,7 +30,7 @@ repos:
|
|||||||
- id: markdownlint-cli2
|
- id: markdownlint-cli2
|
||||||
language_version: lts
|
language_version: lts
|
||||||
args: [--fix]
|
args: [--fix]
|
||||||
exclude: ^CLAUDE\.md$
|
exclude: (^|/)CLAUDE\.md$
|
||||||
- repo: https://github.com/rhysd/actionlint
|
- repo: https://github.com/rhysd/actionlint
|
||||||
rev: v1.7.7
|
rev: v1.7.7
|
||||||
hooks:
|
hooks:
|
||||||
@@ -210,7 +210,7 @@ repos:
|
|||||||
name: Check SPDX headers
|
name: Check SPDX headers
|
||||||
entry: python tools/pre_commit/check_spdx_header.py
|
entry: python tools/pre_commit/check_spdx_header.py
|
||||||
language: python
|
language: python
|
||||||
types: [python]
|
types_or: [python, rust, proto]
|
||||||
- id: check-root-lazy-imports
|
- id: check-root-lazy-imports
|
||||||
name: Check root lazy imports
|
name: Check root lazy imports
|
||||||
entry: python tools/pre_commit/check_init_lazy_imports.py
|
entry: python tools/pre_commit/check_init_lazy_imports.py
|
||||||
@@ -260,10 +260,6 @@ repos:
|
|||||||
files: ^docker/(Dockerfile|versions\.json)$
|
files: ^docker/(Dockerfile|versions\.json)$
|
||||||
pass_filenames: false
|
pass_filenames: false
|
||||||
additional_dependencies: [dockerfile-parse]
|
additional_dependencies: [dockerfile-parse]
|
||||||
- id: attention-backend-docs
|
|
||||||
name: Check attention backend documentation is up to date
|
|
||||||
entry: python tools/pre_commit/generate_attention_backend_docs.py --check
|
|
||||||
language: python
|
|
||||||
- id: check-boolean-context-manager
|
- id: check-boolean-context-manager
|
||||||
name: Check for boolean ops in with-statements
|
name: Check for boolean ops in with-statements
|
||||||
entry: python tools/pre_commit/check_boolean_context_manager.py
|
entry: python tools/pre_commit/check_boolean_context_manager.py
|
||||||
|
|||||||
@@ -1,2 +0,0 @@
|
|||||||
collect_env.py
|
|
||||||
vllm/model_executor/layers/fla/ops/*.py
|
|
||||||
+73
-14
@@ -68,8 +68,8 @@ endif()
|
|||||||
# requirements.txt files and should be kept consistent. The ROCm torch
|
# requirements.txt files and should be kept consistent. The ROCm torch
|
||||||
# versions are derived from docker/Dockerfile.rocm
|
# versions are derived from docker/Dockerfile.rocm
|
||||||
#
|
#
|
||||||
set(TORCH_SUPPORTED_VERSION_CUDA "2.11.0")
|
set(TORCH_SUPPORTED_VERSION_CUDA "2.13.0")
|
||||||
set(TORCH_SUPPORTED_VERSION_ROCM "2.11.0")
|
set(TORCH_SUPPORTED_VERSION_ROCM "2.13.0")
|
||||||
# TORCH_NIGHTLY=1 builds run against unpinned nightly wheels, so the supported-
|
# TORCH_NIGHTLY=1 builds run against unpinned nightly wheels, so the supported-
|
||||||
# version check would always warn. Only treat it as a nightly build when the
|
# version check would always warn. Only treat it as a nightly build when the
|
||||||
# value is exactly "1" (the bootstrap exports TORCH_NIGHTLY=0 by default, which
|
# value is exactly "1" (the bootstrap exports TORCH_NIGHTLY=0 by default, which
|
||||||
@@ -114,6 +114,11 @@ find_package(Torch REQUIRED)
|
|||||||
# Supported NVIDIA architectures.
|
# Supported NVIDIA architectures.
|
||||||
# This check must happen after find_package(Torch) because that's when CMAKE_CUDA_COMPILER_VERSION gets defined
|
# This check must happen after find_package(Torch) because that's when CMAKE_CUDA_COMPILER_VERSION gets defined
|
||||||
if(DEFINED CMAKE_CUDA_COMPILER_VERSION AND
|
if(DEFINED CMAKE_CUDA_COMPILER_VERSION AND
|
||||||
|
CMAKE_CUDA_COMPILER_VERSION VERSION_GREATER_EQUAL 13.4)
|
||||||
|
# Rubin (10.7) can run SM100 family code, but CUDA 13.4 also supports
|
||||||
|
# targeting it directly.
|
||||||
|
set(CUDA_SUPPORTED_ARCHS "7.5;8.0;8.6;8.7;8.9;9.0;10.0;10.7;11.0;12.0")
|
||||||
|
elseif(DEFINED CMAKE_CUDA_COMPILER_VERSION AND
|
||||||
CMAKE_CUDA_COMPILER_VERSION VERSION_GREATER_EQUAL 13.0)
|
CMAKE_CUDA_COMPILER_VERSION VERSION_GREATER_EQUAL 13.0)
|
||||||
# starting from CUDA 12.9 and Blackwell (10.0), we use family-specific targets (10.0f, 12.0f, etc)
|
# starting from CUDA 12.9 and Blackwell (10.0), we use family-specific targets (10.0f, 12.0f, etc)
|
||||||
# to support the whole generation without specifying all sub-architectures
|
# to support the whole generation without specifying all sub-architectures
|
||||||
@@ -214,10 +219,8 @@ if(VLLM_GPU_LANG STREQUAL "CUDA")
|
|||||||
# the set of architectures we want to compile for and remove the from the
|
# the set of architectures we want to compile for and remove the from the
|
||||||
# CMAKE_CUDA_FLAGS so that they are not applied globally.
|
# CMAKE_CUDA_FLAGS so that they are not applied globally.
|
||||||
#
|
#
|
||||||
# `+PTX` in TORCH_CUDA_ARCH_LIST is not preserved here. It is emitted by torch
|
# `+PTX` in TORCH_CUDA_ARCH_LIST is not preserved here. If a kernel really
|
||||||
# as `code=compute_*`, while extract_unique_cuda_archs_ascending() records only
|
# needs PTX, add `+PTX` to that kernel's component-specific arch list below.
|
||||||
# `arch=compute_*`. If a kernel really needs PTX, add `+PTX` to that kernel's
|
|
||||||
# component-specific arch list below.
|
|
||||||
#
|
#
|
||||||
clear_cuda_arches(CUDA_ARCH_FLAGS)
|
clear_cuda_arches(CUDA_ARCH_FLAGS)
|
||||||
extract_unique_cuda_archs_ascending(CUDA_ARCHS "${CUDA_ARCH_FLAGS}")
|
extract_unique_cuda_archs_ascending(CUDA_ARCHS "${CUDA_ARCH_FLAGS}")
|
||||||
@@ -227,6 +230,13 @@ if(VLLM_GPU_LANG STREQUAL "CUDA")
|
|||||||
cuda_archs_loose_intersection(CUDA_ARCHS
|
cuda_archs_loose_intersection(CUDA_ARCHS
|
||||||
"${CUDA_SUPPORTED_ARCHS}" "${CUDA_ARCHS}")
|
"${CUDA_SUPPORTED_ARCHS}" "${CUDA_ARCHS}")
|
||||||
message(STATUS "CUDA supported target architectures: ${CUDA_ARCHS}")
|
message(STATUS "CUDA supported target architectures: ${CUDA_ARCHS}")
|
||||||
|
if(NOT CUDA_ARCHS)
|
||||||
|
message(FATAL_ERROR
|
||||||
|
"No supported CUDA architectures; the build would produce a binary "
|
||||||
|
"with no usable kernels. Detected gencode flags: ${CUDA_ARCH_FLAGS}; "
|
||||||
|
"supported: ${CUDA_SUPPORTED_ARCHS}. "
|
||||||
|
"Set TORCH_CUDA_ARCH_LIST for your GPU (e.g. 12.0).")
|
||||||
|
endif()
|
||||||
else()
|
else()
|
||||||
#
|
#
|
||||||
# For other GPU targets override the GPU architectures detected by cmake/torch
|
# For other GPU targets override the GPU architectures detected by cmake/torch
|
||||||
@@ -411,8 +421,11 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
"csrc/libtorch_stable/mamba/selective_scan_fwd.cu"
|
"csrc/libtorch_stable/mamba/selective_scan_fwd.cu"
|
||||||
"csrc/libtorch_stable/cache_kernels.cu"
|
"csrc/libtorch_stable/cache_kernels.cu"
|
||||||
"csrc/libtorch_stable/cache_kernels_fused.cu"
|
"csrc/libtorch_stable/cache_kernels_fused.cu"
|
||||||
|
"csrc/libtorch_stable/custom_all_gather_reduce_scatter.cu"
|
||||||
|
"csrc/libtorch_stable/custom_all_gather_reduce_scatter_ops.cpp"
|
||||||
"csrc/libtorch_stable/custom_all_reduce.cu"
|
"csrc/libtorch_stable/custom_all_reduce.cu"
|
||||||
"csrc/libtorch_stable/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu")
|
"csrc/libtorch_stable/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu"
|
||||||
|
"csrc/libtorch_stable/fused_kimi_k3_mla_key_concat_kv_cache_kernel.cu")
|
||||||
|
|
||||||
if(VLLM_GPU_LANG STREQUAL "CUDA" AND
|
if(VLLM_GPU_LANG STREQUAL "CUDA" AND
|
||||||
DEFINED CMAKE_CUDA_COMPILER_VERSION AND
|
DEFINED CMAKE_CUDA_COMPILER_VERSION AND
|
||||||
@@ -420,7 +433,7 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
|
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
cuda_archs_loose_intersection(COOPERATIVE_TOPK_ARCHS
|
cuda_archs_loose_intersection(COOPERATIVE_TOPK_ARCHS
|
||||||
"9.0a;10.0f;10.1f;10.3f;11.0f;12.0f;12.1f" "${CUDA_ARCHS}")
|
"9.0a;10.0f;10.1f;10.3f;10.7f;11.0f;12.0f;12.1f" "${CUDA_ARCHS}")
|
||||||
else()
|
else()
|
||||||
cuda_archs_loose_intersection(COOPERATIVE_TOPK_ARCHS
|
cuda_archs_loose_intersection(COOPERATIVE_TOPK_ARCHS
|
||||||
"9.0a;10.0a;10.1a;10.3a;12.0a;12.1a" "${CUDA_ARCHS}")
|
"9.0a;10.0a;10.1a;10.3a;12.0a;12.1a" "${CUDA_ARCHS}")
|
||||||
@@ -695,7 +708,7 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
|
|
||||||
# DeepSeek V3 fused A GEMM kernel (requires SM 9.0+, Hopper and later)
|
# DeepSeek V3 fused A GEMM kernel (requires SM 9.0+, Hopper and later)
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
cuda_archs_loose_intersection(DSV3_FUSED_A_GEMM_ARCHS "9.0a;10.0f;11.0f;12.0f" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(DSV3_FUSED_A_GEMM_ARCHS "9.0a;10.0f;10.7f;11.0f;12.0f" "${CUDA_ARCHS}")
|
||||||
else()
|
else()
|
||||||
cuda_archs_loose_intersection(DSV3_FUSED_A_GEMM_ARCHS "9.0a;10.0a;10.1a;10.3a;12.0a;12.1a" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(DSV3_FUSED_A_GEMM_ARCHS "9.0a;10.0a;10.1a;10.3a;12.0a;12.1a" "${CUDA_ARCHS}")
|
||||||
endif()
|
endif()
|
||||||
@@ -815,7 +828,7 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
# The cutlass_scaled_mm kernels for Blackwell SM100 (c3x, i.e. CUTLASS 3.x)
|
# The cutlass_scaled_mm kernels for Blackwell SM100 (c3x, i.e. CUTLASS 3.x)
|
||||||
# require CUDA 12.8 or later
|
# require CUDA 12.8 or later
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0f;11.0f" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0f;10.7f;11.0f" "${CUDA_ARCHS}")
|
||||||
else()
|
else()
|
||||||
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0a;10.1a;10.3a" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0a;10.1a;10.3a" "${CUDA_ARCHS}")
|
||||||
endif()
|
endif()
|
||||||
@@ -899,7 +912,7 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
endif()
|
endif()
|
||||||
|
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0f;11.0f" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0f;10.7f;11.0f" "${CUDA_ARCHS}")
|
||||||
else()
|
else()
|
||||||
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0a;10.1a;10.3a" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0a;10.1a;10.3a" "${CUDA_ARCHS}")
|
||||||
endif()
|
endif()
|
||||||
@@ -924,7 +937,7 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
|
|
||||||
# moe_data.cu is used by all CUTLASS MoE kernels.
|
# moe_data.cu is used by all CUTLASS MoE kernels.
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
cuda_archs_loose_intersection(CUTLASS_MOE_DATA_ARCHS "9.0a;10.0f;11.0f;12.0f" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(CUTLASS_MOE_DATA_ARCHS "9.0a;10.0f;10.7f;11.0f;12.0f" "${CUDA_ARCHS}")
|
||||||
else()
|
else()
|
||||||
cuda_archs_loose_intersection(CUTLASS_MOE_DATA_ARCHS "9.0a;10.0a;10.1a;10.3a;12.0a;12.1a" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(CUTLASS_MOE_DATA_ARCHS "9.0a;10.0a;10.1a;10.3a;12.0a;12.1a" "${CUDA_ARCHS}")
|
||||||
endif()
|
endif()
|
||||||
@@ -981,7 +994,7 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
# SM10x/11x FP4 kernels. MXFP4 experts quantization is currently compiled
|
# SM10x/11x FP4 kernels. MXFP4 experts quantization is currently compiled
|
||||||
# only in this block; SM12x has separate NVFP4 matmul/MoE kernels above.
|
# only in this block; SM12x has separate NVFP4 matmul/MoE kernels above.
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
cuda_archs_loose_intersection(FP4_SM100_ARCHS "10.0f;11.0f" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(FP4_SM100_ARCHS "10.0f;10.7f;11.0f" "${CUDA_ARCHS}")
|
||||||
else()
|
else()
|
||||||
cuda_archs_loose_intersection(FP4_SM100_ARCHS "10.0a;10.1a;10.3a" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(FP4_SM100_ARCHS "10.0a;10.1a;10.3a" "${CUDA_ARCHS}")
|
||||||
endif()
|
endif()
|
||||||
@@ -1047,7 +1060,7 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
# Runtime dispatch is gated in
|
# Runtime dispatch is gated in
|
||||||
# vllm/v1/attention/backends/mla/cutlass_mla.py.
|
# vllm/v1/attention/backends/mla/cutlass_mla.py.
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
cuda_archs_loose_intersection(MLA_ARCHS "10.0f;11.0f" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(MLA_ARCHS "10.0f;10.7f;11.0f" "${CUDA_ARCHS}")
|
||||||
else()
|
else()
|
||||||
cuda_archs_loose_intersection(MLA_ARCHS "10.0a;10.1a;10.3a" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(MLA_ARCHS "10.0a;10.1a;10.3a" "${CUDA_ARCHS}")
|
||||||
endif()
|
endif()
|
||||||
@@ -1069,6 +1082,41 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
set(MLA_ARCHS)
|
set(MLA_ARCHS)
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
|
cuda_archs_loose_intersection(FUSED_KDA_DECODE_ARCHS
|
||||||
|
"9.0a;10.0f;12.0f" "${CUDA_ARCHS}")
|
||||||
|
endif()
|
||||||
|
if(FUSED_KDA_DECODE_ARCHS)
|
||||||
|
set(FUSED_KDA_DECODE_SRC
|
||||||
|
"csrc/libtorch_stable/kimi_k3/fused_kda_decode_kernel.cu")
|
||||||
|
set_gencode_flags_for_srcs(
|
||||||
|
SRCS "${FUSED_KDA_DECODE_SRC}"
|
||||||
|
CUDA_ARCHS "${FUSED_KDA_DECODE_ARCHS}")
|
||||||
|
set_property(SOURCE ${FUSED_KDA_DECODE_SRC} APPEND PROPERTY
|
||||||
|
COMPILE_OPTIONS "$<$<COMPILE_LANGUAGE:CUDA>:--use_fast_math>")
|
||||||
|
list(APPEND VLLM_STABLE_EXT_SRC "${FUSED_KDA_DECODE_SRC}")
|
||||||
|
message(STATUS
|
||||||
|
"Building fused KDA decode for archs: ${FUSED_KDA_DECODE_ARCHS}")
|
||||||
|
endif()
|
||||||
|
|
||||||
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
|
cuda_archs_loose_intersection(KIMI_K3_ATTN_RES_ARCHS
|
||||||
|
"10.0f" "${CUDA_ARCHS}")
|
||||||
|
endif()
|
||||||
|
if(KIMI_K3_ATTN_RES_ARCHS)
|
||||||
|
set(KIMI_K3_ATTN_RES_SRC
|
||||||
|
"csrc/libtorch_stable/kimi_k3/attn_res_kernel.cu")
|
||||||
|
set_gencode_flags_for_srcs(
|
||||||
|
SRCS "${KIMI_K3_ATTN_RES_SRC}"
|
||||||
|
CUDA_ARCHS "${KIMI_K3_ATTN_RES_ARCHS}")
|
||||||
|
set_property(SOURCE ${KIMI_K3_ATTN_RES_SRC} APPEND PROPERTY
|
||||||
|
COMPILE_OPTIONS
|
||||||
|
"$<$<COMPILE_LANGUAGE:CUDA>:--expt-relaxed-constexpr;--expt-extended-lambda;--use_fast_math>")
|
||||||
|
list(APPEND VLLM_STABLE_EXT_SRC "${KIMI_K3_ATTN_RES_SRC}")
|
||||||
|
message(STATUS
|
||||||
|
"Building Kimi K3 AttnRes for archs: ${KIMI_K3_ATTN_RES_ARCHS}")
|
||||||
|
endif()
|
||||||
|
|
||||||
# Hadacore kernels
|
# Hadacore kernels
|
||||||
cuda_archs_loose_intersection(HADACORE_ARCHS "8.0+PTX;9.0+PTX" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(HADACORE_ARCHS "8.0+PTX;9.0+PTX" "${CUDA_ARCHS}")
|
||||||
if(HADACORE_ARCHS)
|
if(HADACORE_ARCHS)
|
||||||
@@ -1110,6 +1158,14 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
target_compile_definitions(_C_stable_libtorch PRIVATE
|
target_compile_definitions(_C_stable_libtorch PRIVATE
|
||||||
VLLM_ENABLE_COOPERATIVE_TOPK=1)
|
VLLM_ENABLE_COOPERATIVE_TOPK=1)
|
||||||
endif()
|
endif()
|
||||||
|
if(FUSED_KDA_DECODE_ARCHS)
|
||||||
|
target_compile_definitions(_C_stable_libtorch PRIVATE
|
||||||
|
VLLM_ENABLE_FUSED_KDA_DECODE=1)
|
||||||
|
endif()
|
||||||
|
if(KIMI_K3_ATTN_RES_ARCHS)
|
||||||
|
target_compile_definitions(_C_stable_libtorch PRIVATE
|
||||||
|
VLLM_ENABLE_KIMI_K3_ATTN_RES=1)
|
||||||
|
endif()
|
||||||
# Needed by CUTLASS kernels
|
# Needed by CUTLASS kernels
|
||||||
target_compile_definitions(_C_stable_libtorch PRIVATE
|
target_compile_definitions(_C_stable_libtorch PRIVATE
|
||||||
CUTLASS_ENABLE_DIRECT_CUDA_DRIVER_CALL=1)
|
CUTLASS_ENABLE_DIRECT_CUDA_DRIVER_CALL=1)
|
||||||
@@ -1364,6 +1420,7 @@ if(VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
set(VLLM_ROCM_EXT_SRC
|
set(VLLM_ROCM_EXT_SRC
|
||||||
"csrc/rocm/torch_bindings.cpp"
|
"csrc/rocm/torch_bindings.cpp"
|
||||||
"csrc/rocm/skinny_gemms.cu"
|
"csrc/rocm/skinny_gemms.cu"
|
||||||
|
"csrc/rocm/skinny_gemms_int4.cu"
|
||||||
"csrc/rocm/attention.cu")
|
"csrc/rocm/attention.cu")
|
||||||
|
|
||||||
set(VLLM_ROCM_HAS_GFX1100 OFF)
|
set(VLLM_ROCM_HAS_GFX1100 OFF)
|
||||||
@@ -1406,7 +1463,9 @@ if (VLLM_GPU_LANG STREQUAL "CUDA")
|
|||||||
include(cmake/external_projects/deepgemm.cmake)
|
include(cmake/external_projects/deepgemm.cmake)
|
||||||
include(cmake/external_projects/fmha_sm100.cmake)
|
include(cmake/external_projects/fmha_sm100.cmake)
|
||||||
include(cmake/external_projects/flashmla.cmake)
|
include(cmake/external_projects/flashmla.cmake)
|
||||||
|
include(cmake/external_projects/flashkda.cmake)
|
||||||
include(cmake/external_projects/qutlass.cmake)
|
include(cmake/external_projects/qutlass.cmake)
|
||||||
|
include(cmake/external_projects/tml_fa4.cmake)
|
||||||
|
|
||||||
# vllm-flash-attn should be last as it overwrites some CMake functions
|
# vllm-flash-attn should be last as it overwrites some CMake functions
|
||||||
include(cmake/external_projects/vllm_flash_attn.cmake)
|
include(cmake/external_projects/vllm_flash_attn.cmake)
|
||||||
|
|||||||
@@ -48,7 +48,7 @@ vLLM is flexible and easy to use with:
|
|||||||
- Tool calling and reasoning parsers
|
- Tool calling and reasoning parsers
|
||||||
- OpenAI-compatible API server, plus Anthropic Messages API and gRPC support
|
- OpenAI-compatible API server, plus Anthropic Messages API and gRPC support
|
||||||
- Efficient multi-LoRA support for dense and MoE layers
|
- Efficient multi-LoRA support for dense and MoE layers
|
||||||
- Support for NVIDIA GPUs, AMD GPUs, and x86/ARM/PowerPC CPUs. Additionally, diverse hardware plugins such as Google TPUs, Intel Gaudi, IBM Spyre, Huawei Ascend, Rebellions NPU, Apple Silicon, MetaX GPU, and more.
|
- Support for NVIDIA GPUs, AMD GPUs, Intel GPUs, and x86/ARM/PowerPC CPUs. Additionally, diverse hardware plugins such as Google TPUs, Intel Gaudi, IBM Spyre, Huawei Ascend, Rebellions NPU, Apple Silicon, MetaX GPU, and more.
|
||||||
|
|
||||||
vLLM seamlessly supports 200+ model architectures on Hugging Face, including:
|
vLLM seamlessly supports 200+ model architectures on Hugging Face, including:
|
||||||
|
|
||||||
|
|||||||
@@ -75,7 +75,11 @@ def run_mla_benchmark(config: BenchmarkConfig, **kwargs) -> BenchmarkResult:
|
|||||||
from mla_runner import run_mla_benchmark as run_mla
|
from mla_runner import run_mla_benchmark as run_mla
|
||||||
|
|
||||||
return run_mla(
|
return run_mla(
|
||||||
config.backend, config, prefill_backend=config.prefill_backend, **kwargs
|
config.backend,
|
||||||
|
config,
|
||||||
|
prefill_backend=config.prefill_backend,
|
||||||
|
sparse_mla_force_mqa=config.sparse_mla_force_mqa,
|
||||||
|
**kwargs,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -592,6 +596,30 @@ def main():
|
|||||||
default="profile",
|
default="profile",
|
||||||
help="Output file name for ncu profile (default: 'profile').",
|
help="Output file name for ncu profile (default: 'profile').",
|
||||||
)
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--torch-profile",
|
||||||
|
action="store_true",
|
||||||
|
default=False,
|
||||||
|
help="Collect a PyTorch profiler Chrome trace for each benchmark run.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--torch-profile-dir",
|
||||||
|
default=None,
|
||||||
|
help="Directory for PyTorch profiler traces.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--torch-profile-iters",
|
||||||
|
type=int,
|
||||||
|
default=3,
|
||||||
|
help="Number of forward passes to record per PyTorch profiler trace.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--sparse-mla-mha-variants",
|
||||||
|
nargs="+",
|
||||||
|
default=None,
|
||||||
|
choices=["dense_mha", "mqa"],
|
||||||
|
help="Sparse MLA variants to run in mha_vs_mqa mode. Defaults to both.",
|
||||||
|
)
|
||||||
|
|
||||||
# Parameter sweep (use YAML config for advanced sweeps)
|
# Parameter sweep (use YAML config for advanced sweeps)
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
@@ -641,6 +669,7 @@ def main():
|
|||||||
|
|
||||||
# Prefill backends (e.g., ["fa3", "fa4"])
|
# Prefill backends (e.g., ["fa3", "fa4"])
|
||||||
args.prefill_backends = yaml_config.get("prefill_backends", None)
|
args.prefill_backends = yaml_config.get("prefill_backends", None)
|
||||||
|
args.prefill_backend = yaml_config.get("prefill_backend", None)
|
||||||
|
|
||||||
# FP8 output benchmark knobs; CLI wins.
|
# FP8 output benchmark knobs; CLI wins.
|
||||||
if args.fp8_output_scale is None:
|
if args.fp8_output_scale is None:
|
||||||
@@ -683,6 +712,9 @@ def main():
|
|||||||
args.num_q_heads = model.get("num_q_heads", args.num_q_heads)
|
args.num_q_heads = model.get("num_q_heads", args.num_q_heads)
|
||||||
args.num_kv_heads = model.get("num_kv_heads", args.num_kv_heads)
|
args.num_kv_heads = model.get("num_kv_heads", args.num_kv_heads)
|
||||||
args.block_size = model.get("block_size", args.block_size)
|
args.block_size = model.get("block_size", args.block_size)
|
||||||
|
args.max_model_len = model.get(
|
||||||
|
"max_model_len", getattr(args, "max_model_len", None)
|
||||||
|
)
|
||||||
# MLA-specific dimensions
|
# MLA-specific dimensions
|
||||||
args.kv_lora_rank = model.get("kv_lora_rank", args.kv_lora_rank)
|
args.kv_lora_rank = model.get("kv_lora_rank", args.kv_lora_rank)
|
||||||
args.qk_nope_head_dim = model.get("qk_nope_head_dim", args.qk_nope_head_dim)
|
args.qk_nope_head_dim = model.get("qk_nope_head_dim", args.qk_nope_head_dim)
|
||||||
@@ -701,6 +733,21 @@ def main():
|
|||||||
args.cuda_graphs = yaml_config["cuda_graphs"]
|
args.cuda_graphs = yaml_config["cuda_graphs"]
|
||||||
if "ncu_profile" in yaml_config:
|
if "ncu_profile" in yaml_config:
|
||||||
args.ncu_profile = yaml_config["ncu_profile"]
|
args.ncu_profile = yaml_config["ncu_profile"]
|
||||||
|
if "torch_profile" in yaml_config:
|
||||||
|
args.torch_profile = yaml_config["torch_profile"]
|
||||||
|
if "torch_profile_dir" in yaml_config:
|
||||||
|
args.torch_profile_dir = yaml_config["torch_profile_dir"]
|
||||||
|
if "torch_profile_iters" in yaml_config:
|
||||||
|
args.torch_profile_iters = yaml_config["torch_profile_iters"]
|
||||||
|
args.sparse_mla_topk_pattern = yaml_config.get(
|
||||||
|
"sparse_mla_topk_pattern", "random"
|
||||||
|
)
|
||||||
|
args.sparse_mla_dense_mha_max_seq_len = yaml_config.get(
|
||||||
|
"sparse_mla_dense_mha_max_seq_len", None
|
||||||
|
)
|
||||||
|
args.sparse_mla_mha_variants = yaml_config.get(
|
||||||
|
"sparse_mla_mha_variants", args.sparse_mla_mha_variants
|
||||||
|
)
|
||||||
|
|
||||||
# Parameter sweep configuration
|
# Parameter sweep configuration
|
||||||
if "parameter_sweep" in yaml_config:
|
if "parameter_sweep" in yaml_config:
|
||||||
@@ -842,8 +889,6 @@ def main():
|
|||||||
num_kv_heads=args.num_kv_heads,
|
num_kv_heads=args.num_kv_heads,
|
||||||
block_size=args.block_size,
|
block_size=args.block_size,
|
||||||
device=args.device,
|
device=args.device,
|
||||||
repeats=args.repeats,
|
|
||||||
warmup_iters=args.warmup_iters,
|
|
||||||
profile_memory=args.profile_memory,
|
profile_memory=args.profile_memory,
|
||||||
kv_cache_dtype=args.kv_cache_dtype,
|
kv_cache_dtype=args.kv_cache_dtype,
|
||||||
use_cuda_graphs=args.cuda_graphs,
|
use_cuda_graphs=args.cuda_graphs,
|
||||||
@@ -1063,6 +1108,133 @@ def main():
|
|||||||
f"\n [yellow]Prefill always faster for batch_size={bs}[/]"
|
f"\n [yellow]Prefill always faster for batch_size={bs}[/]"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Handle MHA vs MQA comparison mode for sparse MLA
|
||||||
|
elif hasattr(args, "mode") and args.mode == "mha_vs_mqa":
|
||||||
|
console.print("[yellow]Mode: MHA vs MQA comparison for sparse MLA[/]")
|
||||||
|
|
||||||
|
sparse_mla_topk_pattern = getattr(args, "sparse_mla_topk_pattern", "random")
|
||||||
|
dense_mha_max_seq_len = getattr(args, "sparse_mla_dense_mha_max_seq_len", None)
|
||||||
|
prefill_backend = getattr(args, "prefill_backend", None)
|
||||||
|
if prefill_backend:
|
||||||
|
console.print(f"Prefill backend: {prefill_backend}")
|
||||||
|
available_variants = [
|
||||||
|
("dense_mha", False, "dense"),
|
||||||
|
("mqa", True, "auto"),
|
||||||
|
]
|
||||||
|
requested_variants = getattr(args, "sparse_mla_mha_variants", None)
|
||||||
|
if requested_variants is not None:
|
||||||
|
valid_variants = {label for label, _, _ in available_variants}
|
||||||
|
invalid_variants = sorted(set(requested_variants) - valid_variants)
|
||||||
|
if invalid_variants:
|
||||||
|
raise ValueError(
|
||||||
|
"Invalid sparse_mla_mha_variants entries: "
|
||||||
|
f"{invalid_variants}. Valid variants are: "
|
||||||
|
f"{sorted(valid_variants)}"
|
||||||
|
)
|
||||||
|
requested_variant_set = set(requested_variants)
|
||||||
|
variants = [
|
||||||
|
variant
|
||||||
|
for variant in available_variants
|
||||||
|
if variant[0] in requested_variant_set
|
||||||
|
]
|
||||||
|
else:
|
||||||
|
variants = available_variants
|
||||||
|
formatter = ResultsFormatter(console)
|
||||||
|
total = 0
|
||||||
|
for spec in args.batch_specs:
|
||||||
|
q_len = max(request.q_len for request in parse_batch_spec(spec))
|
||||||
|
for variant_label, _, _ in variants:
|
||||||
|
if (
|
||||||
|
variant_label == "dense_mha"
|
||||||
|
and dense_mha_max_seq_len is not None
|
||||||
|
and q_len > dense_mha_max_seq_len
|
||||||
|
):
|
||||||
|
continue
|
||||||
|
total += len(backends)
|
||||||
|
|
||||||
|
with tqdm(total=total, desc="Benchmarking") as pbar:
|
||||||
|
for spec in args.batch_specs:
|
||||||
|
q_len = max(request.q_len for request in parse_batch_spec(spec))
|
||||||
|
for backend in backends:
|
||||||
|
for variant_label, force_mqa, mha_mode in variants:
|
||||||
|
if (
|
||||||
|
variant_label == "dense_mha"
|
||||||
|
and dense_mha_max_seq_len is not None
|
||||||
|
and q_len > dense_mha_max_seq_len
|
||||||
|
):
|
||||||
|
continue
|
||||||
|
config = BenchmarkConfig(
|
||||||
|
backend=f"{backend}_{variant_label}",
|
||||||
|
batch_spec=spec,
|
||||||
|
num_layers=args.num_layers,
|
||||||
|
head_dim=args.head_dim,
|
||||||
|
num_q_heads=args.num_q_heads,
|
||||||
|
num_kv_heads=args.num_kv_heads,
|
||||||
|
block_size=args.block_size,
|
||||||
|
device=args.device,
|
||||||
|
max_model_len=getattr(args, "max_model_len", None),
|
||||||
|
kv_cache_dtype=args.kv_cache_dtype,
|
||||||
|
profile_memory=args.profile_memory,
|
||||||
|
use_cuda_graphs=args.cuda_graphs,
|
||||||
|
ncu_profile=args.ncu_profile,
|
||||||
|
torch_profile=args.torch_profile,
|
||||||
|
torch_profile_dir=args.torch_profile_dir,
|
||||||
|
torch_profile_iters=args.torch_profile_iters,
|
||||||
|
warmup_ms=args.warmup_ms,
|
||||||
|
kv_lora_rank=getattr(args, "kv_lora_rank", None),
|
||||||
|
qk_nope_head_dim=getattr(args, "qk_nope_head_dim", None),
|
||||||
|
qk_rope_head_dim=getattr(args, "qk_rope_head_dim", None),
|
||||||
|
v_head_dim=getattr(args, "v_head_dim", None),
|
||||||
|
sparse_mla_force_mqa=force_mqa,
|
||||||
|
sparse_mla_mha_mode=mha_mode,
|
||||||
|
sparse_mla_dense_mha_max_seq_len=dense_mha_max_seq_len,
|
||||||
|
sparse_mla_topk_pattern=sparse_mla_topk_pattern,
|
||||||
|
prefill_backend=prefill_backend,
|
||||||
|
)
|
||||||
|
|
||||||
|
# run_mla_benchmark needs the real backend name
|
||||||
|
from mla_runner import run_mla_benchmark as run_mla
|
||||||
|
|
||||||
|
run_label = f"{backend}_{variant_label} {spec}"
|
||||||
|
pbar.set_postfix_str(run_label)
|
||||||
|
|
||||||
|
try:
|
||||||
|
result = run_mla(
|
||||||
|
backend,
|
||||||
|
config,
|
||||||
|
prefill_backend=prefill_backend,
|
||||||
|
sparse_mla_force_mqa=force_mqa,
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
result = BenchmarkResult(
|
||||||
|
config=config,
|
||||||
|
mean_time=float("inf"),
|
||||||
|
median_time=float("inf"),
|
||||||
|
std_time=0,
|
||||||
|
min_time=float("inf"),
|
||||||
|
max_time=float("inf"),
|
||||||
|
error=str(e),
|
||||||
|
)
|
||||||
|
|
||||||
|
all_results.append(result)
|
||||||
|
if args.output_csv:
|
||||||
|
formatter.save_csv(all_results, args.output_csv)
|
||||||
|
if args.output_json:
|
||||||
|
formatter.save_json(all_results, args.output_json)
|
||||||
|
|
||||||
|
if not result.success:
|
||||||
|
console.print(
|
||||||
|
f"[red]Error {backend}_{variant_label} "
|
||||||
|
f"{spec}: {result.error}[/]"
|
||||||
|
)
|
||||||
|
|
||||||
|
pbar.update(1)
|
||||||
|
|
||||||
|
# Display results with variant labels as separate "backends"
|
||||||
|
console.print("\n[bold green]MHA vs MQA Results:[/]")
|
||||||
|
variant_backends = [f"{b}_{v}" for b in backends for v, _, _ in variants]
|
||||||
|
formatter.print_table(all_results, variant_backends)
|
||||||
|
|
||||||
# Handle model parameter sweep mode
|
# Handle model parameter sweep mode
|
||||||
elif hasattr(args, "model_parameter_sweep") and args.model_parameter_sweep:
|
elif hasattr(args, "model_parameter_sweep") and args.model_parameter_sweep:
|
||||||
# Model parameter sweep
|
# Model parameter sweep
|
||||||
@@ -1186,6 +1358,10 @@ def main():
|
|||||||
profile_memory=args.profile_memory,
|
profile_memory=args.profile_memory,
|
||||||
warmup_ms=args.warmup_ms,
|
warmup_ms=args.warmup_ms,
|
||||||
prefill_backend=pb,
|
prefill_backend=pb,
|
||||||
|
kv_lora_rank=args.kv_lora_rank,
|
||||||
|
qk_nope_head_dim=args.qk_nope_head_dim,
|
||||||
|
qk_rope_head_dim=args.qk_rope_head_dim,
|
||||||
|
v_head_dim=args.v_head_dim,
|
||||||
)
|
)
|
||||||
|
|
||||||
result = run_benchmark(config)
|
result = run_benchmark(config)
|
||||||
|
|||||||
@@ -4,8 +4,10 @@
|
|||||||
"""Common utilities for attention benchmarking."""
|
"""Common utilities for attention benchmarking."""
|
||||||
|
|
||||||
import csv
|
import csv
|
||||||
|
import gc
|
||||||
import json
|
import json
|
||||||
import math
|
import math
|
||||||
|
from collections.abc import Sequence
|
||||||
from dataclasses import asdict, dataclass
|
from dataclasses import asdict, dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
@@ -44,10 +46,13 @@ def run_do_bench(
|
|||||||
kwargs: dict[str, Any] = {"return_mode": "all"}
|
kwargs: dict[str, Any] = {"return_mode": "all"}
|
||||||
if use_cuda_graphs:
|
if use_cuda_graphs:
|
||||||
result = triton.testing.do_bench_cudagraph(benchmark_fn, **kwargs)
|
result = triton.testing.do_bench_cudagraph(benchmark_fn, **kwargs)
|
||||||
|
gc.collect()
|
||||||
|
torch.accelerator.empty_cache()
|
||||||
else:
|
else:
|
||||||
if warmup_ms is not None:
|
if warmup_ms is not None:
|
||||||
kwargs["warmup"] = warmup_ms
|
kwargs["warmup"] = warmup_ms
|
||||||
result = triton.testing.do_bench(benchmark_fn, **kwargs)
|
result = triton.testing.do_bench(benchmark_fn, **kwargs)
|
||||||
|
torch.accelerator.synchronize()
|
||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
@@ -91,42 +96,6 @@ except ImportError:
|
|||||||
AttentionLayerBase = object # Fallback
|
AttentionLayerBase = object # Fallback
|
||||||
|
|
||||||
|
|
||||||
class MockKVBProj:
|
|
||||||
"""Mock KV projection layer for MLA prefill mode.
|
|
||||||
|
|
||||||
Mimics ColumnParallelLinear behavior for kv_b_proj in MLA backends.
|
|
||||||
Projects kv_c_normed to [qk_nope_head_dim + v_head_dim] per head.
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, num_heads: int, qk_nope_head_dim: int, v_head_dim: int):
|
|
||||||
self.num_heads = num_heads
|
|
||||||
self.qk_nope_head_dim = qk_nope_head_dim
|
|
||||||
self.v_head_dim = v_head_dim
|
|
||||||
self.out_dim = qk_nope_head_dim + v_head_dim
|
|
||||||
self.weight = torch.empty(0, dtype=torch.bfloat16)
|
|
||||||
|
|
||||||
def __call__(self, x: torch.Tensor) -> tuple[torch.Tensor]:
|
|
||||||
"""
|
|
||||||
Project kv_c_normed to output space.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
x: Input tensor [num_tokens, kv_lora_rank]
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
Tuple containing output tensor
|
|
||||||
[num_tokens, num_heads, qk_nope_head_dim + v_head_dim]
|
|
||||||
"""
|
|
||||||
num_tokens = x.shape[0]
|
|
||||||
result = torch.randn(
|
|
||||||
num_tokens,
|
|
||||||
self.num_heads,
|
|
||||||
self.out_dim,
|
|
||||||
device=x.device,
|
|
||||||
dtype=x.dtype,
|
|
||||||
)
|
|
||||||
return (result,) # Return as tuple to match ColumnParallelLinear API
|
|
||||||
|
|
||||||
|
|
||||||
class MockIndexer:
|
class MockIndexer:
|
||||||
"""Mock Indexer for sparse MLA backends.
|
"""Mock Indexer for sparse MLA backends.
|
||||||
|
|
||||||
@@ -158,6 +127,60 @@ class MockIndexer:
|
|||||||
)
|
)
|
||||||
self.topk_indices_buffer[:num_tokens] = indices
|
self.topk_indices_buffer[:num_tokens] = indices
|
||||||
|
|
||||||
|
def fill_indices(
|
||||||
|
self,
|
||||||
|
num_tokens: int,
|
||||||
|
max_kv_len: int,
|
||||||
|
pattern: str = "random",
|
||||||
|
requests: Sequence[Any] | None = None,
|
||||||
|
):
|
||||||
|
if pattern == "random":
|
||||||
|
self.fill_random_indices(num_tokens, max_kv_len)
|
||||||
|
return
|
||||||
|
if pattern == "prefix":
|
||||||
|
indices = torch.arange(
|
||||||
|
self.topk_tokens,
|
||||||
|
dtype=torch.int32,
|
||||||
|
device=self.topk_indices_buffer.device,
|
||||||
|
)
|
||||||
|
indices = (indices % max_kv_len).expand(num_tokens, -1)
|
||||||
|
self.topk_indices_buffer[:num_tokens] = indices
|
||||||
|
return
|
||||||
|
if pattern == "sliding_window":
|
||||||
|
if requests is None:
|
||||||
|
start = max(max_kv_len - self.topk_tokens, 0)
|
||||||
|
indices = torch.arange(
|
||||||
|
start,
|
||||||
|
start + self.topk_tokens,
|
||||||
|
dtype=torch.int32,
|
||||||
|
device=self.topk_indices_buffer.device,
|
||||||
|
)
|
||||||
|
indices = indices.clamp(max=max_kv_len - 1).expand(num_tokens, -1)
|
||||||
|
self.topk_indices_buffer[:num_tokens] = indices
|
||||||
|
return
|
||||||
|
|
||||||
|
rows = []
|
||||||
|
offsets = torch.arange(
|
||||||
|
self.topk_tokens,
|
||||||
|
dtype=torch.int32,
|
||||||
|
device=self.topk_indices_buffer.device,
|
||||||
|
) - (self.topk_tokens - 1)
|
||||||
|
for request in requests:
|
||||||
|
q_len = request.q_len
|
||||||
|
kv_len = request.kv_len
|
||||||
|
context_len = kv_len - q_len
|
||||||
|
positions = torch.arange(
|
||||||
|
context_len,
|
||||||
|
kv_len,
|
||||||
|
dtype=torch.int32,
|
||||||
|
device=self.topk_indices_buffer.device,
|
||||||
|
)
|
||||||
|
row_indices = positions[:, None] + offsets[None, :]
|
||||||
|
rows.append(row_indices.clamp(min=0, max=kv_len - 1))
|
||||||
|
self.topk_indices_buffer[:num_tokens] = torch.cat(rows, dim=0)
|
||||||
|
return
|
||||||
|
raise ValueError(f"Unknown sparse MLA topk pattern: {pattern}")
|
||||||
|
|
||||||
|
|
||||||
class MockLayer(AttentionLayerBase):
|
class MockLayer(AttentionLayerBase):
|
||||||
"""Mock attention layer with scale parameters and impl.
|
"""Mock attention layer with scale parameters and impl.
|
||||||
@@ -252,10 +275,14 @@ class BenchmarkConfig:
|
|||||||
num_kv_heads: int
|
num_kv_heads: int
|
||||||
block_size: int
|
block_size: int
|
||||||
device: str
|
device: str
|
||||||
|
max_model_len: int | None = None
|
||||||
dtype: torch.dtype = torch.float16
|
dtype: torch.dtype = torch.float16
|
||||||
profile_memory: bool = False
|
profile_memory: bool = False
|
||||||
use_cuda_graphs: bool = False
|
use_cuda_graphs: bool = True
|
||||||
ncu_profile: bool = False
|
ncu_profile: bool = False
|
||||||
|
torch_profile: bool = False
|
||||||
|
torch_profile_dir: str | None = None
|
||||||
|
torch_profile_iters: int = 3
|
||||||
warmup_ms: int | None = None
|
warmup_ms: int | None = None
|
||||||
|
|
||||||
# "auto" or "fp8"
|
# "auto" or "fp8"
|
||||||
@@ -271,6 +298,10 @@ class BenchmarkConfig:
|
|||||||
# Backend-specific tuning
|
# Backend-specific tuning
|
||||||
num_kv_splits: int | None = None # CUTLASS MLA
|
num_kv_splits: int | None = None # CUTLASS MLA
|
||||||
reorder_batch_threshold: int | None = None # FlashAttn MLA, FlashMLA
|
reorder_batch_threshold: int | None = None # FlashAttn MLA, FlashMLA
|
||||||
|
sparse_mla_force_mqa: bool = False # Force MQA path for sparse MLA
|
||||||
|
sparse_mla_mha_mode: str = "auto" # "auto" or "dense"
|
||||||
|
sparse_mla_dense_mha_max_seq_len: int | None = None
|
||||||
|
sparse_mla_topk_pattern: str = "random" # "random", "prefix", "sliding_window"
|
||||||
num_splits: int | None = None # FlashAttention split-K (0=auto, 1=disabled)
|
num_splits: int | None = None # FlashAttention split-K (0=auto, 1=disabled)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,474 @@
|
|||||||
|
# Sparse MLA benchmark: forward_mha vs forward_mqa
|
||||||
|
#
|
||||||
|
# Usage:
|
||||||
|
# python benchmark.py --config configs/mla_sparse_mha_vs_mqa.yaml
|
||||||
|
#
|
||||||
|
# Heatmap grid:
|
||||||
|
# - batch_size: 1, 2, 4, 8, 16, 32
|
||||||
|
# - seq_len: 32, 64, 128, 256, 512, 1024, 2048
|
||||||
|
# - q_len: powers of two through seq_len
|
||||||
|
#
|
||||||
|
# Specs with q_len < seq_len include context; the q_len == seq_len diagonal
|
||||||
|
# covers pure prefill.
|
||||||
|
# The model shape below is the DP case. For the TP8 run, manually change
|
||||||
|
# model.num_q_heads from 128 to 16 before rerunning this benchmark.
|
||||||
|
|
||||||
|
mode: mha_vs_mqa
|
||||||
|
|
||||||
|
model:
|
||||||
|
name: "deepseek-v3"
|
||||||
|
num_layers: 60
|
||||||
|
num_q_heads: 128
|
||||||
|
num_kv_heads: 1
|
||||||
|
head_dim: 576
|
||||||
|
kv_lora_rank: 512
|
||||||
|
qk_nope_head_dim: 128
|
||||||
|
qk_rope_head_dim: 64
|
||||||
|
v_head_dim: 128
|
||||||
|
block_size: 128
|
||||||
|
max_model_len: 2048
|
||||||
|
|
||||||
|
batch_specs:
|
||||||
|
# Batch size 1
|
||||||
|
# seq_len = 32
|
||||||
|
- "1q1s32"
|
||||||
|
- "1q2s32"
|
||||||
|
- "1q4s32"
|
||||||
|
- "1q8s32"
|
||||||
|
- "1q16s32"
|
||||||
|
- "1q32"
|
||||||
|
# seq_len = 64
|
||||||
|
- "1q1s64"
|
||||||
|
- "1q2s64"
|
||||||
|
- "1q4s64"
|
||||||
|
- "1q8s64"
|
||||||
|
- "1q16s64"
|
||||||
|
- "1q32s64"
|
||||||
|
- "1q64"
|
||||||
|
# seq_len = 128
|
||||||
|
- "1q1s128"
|
||||||
|
- "1q2s128"
|
||||||
|
- "1q4s128"
|
||||||
|
- "1q8s128"
|
||||||
|
- "1q16s128"
|
||||||
|
- "1q32s128"
|
||||||
|
- "1q64s128"
|
||||||
|
- "1q128"
|
||||||
|
# seq_len = 256
|
||||||
|
- "1q1s256"
|
||||||
|
- "1q2s256"
|
||||||
|
- "1q4s256"
|
||||||
|
- "1q8s256"
|
||||||
|
- "1q16s256"
|
||||||
|
- "1q32s256"
|
||||||
|
- "1q64s256"
|
||||||
|
- "1q128s256"
|
||||||
|
- "1q256"
|
||||||
|
# seq_len = 512
|
||||||
|
- "1q1s512"
|
||||||
|
- "1q2s512"
|
||||||
|
- "1q4s512"
|
||||||
|
- "1q8s512"
|
||||||
|
- "1q16s512"
|
||||||
|
- "1q32s512"
|
||||||
|
- "1q64s512"
|
||||||
|
- "1q128s512"
|
||||||
|
- "1q256s512"
|
||||||
|
- "1q512"
|
||||||
|
# seq_len = 1024
|
||||||
|
- "1q1s1024"
|
||||||
|
- "1q2s1024"
|
||||||
|
- "1q4s1024"
|
||||||
|
- "1q8s1024"
|
||||||
|
- "1q16s1024"
|
||||||
|
- "1q32s1024"
|
||||||
|
- "1q64s1024"
|
||||||
|
- "1q128s1024"
|
||||||
|
- "1q256s1024"
|
||||||
|
- "1q512s1024"
|
||||||
|
- "1q1024"
|
||||||
|
# seq_len = 2048
|
||||||
|
- "1q1s2048"
|
||||||
|
- "1q2s2048"
|
||||||
|
- "1q4s2048"
|
||||||
|
- "1q8s2048"
|
||||||
|
- "1q16s2048"
|
||||||
|
- "1q32s2048"
|
||||||
|
- "1q64s2048"
|
||||||
|
- "1q128s2048"
|
||||||
|
- "1q256s2048"
|
||||||
|
- "1q512s2048"
|
||||||
|
- "1q1024s2048"
|
||||||
|
- "1q2048"
|
||||||
|
|
||||||
|
# Batch size 2
|
||||||
|
# seq_len = 32
|
||||||
|
- "2q1s32"
|
||||||
|
- "2q2s32"
|
||||||
|
- "2q4s32"
|
||||||
|
- "2q8s32"
|
||||||
|
- "2q16s32"
|
||||||
|
- "2q32"
|
||||||
|
# seq_len = 64
|
||||||
|
- "2q1s64"
|
||||||
|
- "2q2s64"
|
||||||
|
- "2q4s64"
|
||||||
|
- "2q8s64"
|
||||||
|
- "2q16s64"
|
||||||
|
- "2q32s64"
|
||||||
|
- "2q64"
|
||||||
|
# seq_len = 128
|
||||||
|
- "2q1s128"
|
||||||
|
- "2q2s128"
|
||||||
|
- "2q4s128"
|
||||||
|
- "2q8s128"
|
||||||
|
- "2q16s128"
|
||||||
|
- "2q32s128"
|
||||||
|
- "2q64s128"
|
||||||
|
- "2q128"
|
||||||
|
# seq_len = 256
|
||||||
|
- "2q1s256"
|
||||||
|
- "2q2s256"
|
||||||
|
- "2q4s256"
|
||||||
|
- "2q8s256"
|
||||||
|
- "2q16s256"
|
||||||
|
- "2q32s256"
|
||||||
|
- "2q64s256"
|
||||||
|
- "2q128s256"
|
||||||
|
- "2q256"
|
||||||
|
# seq_len = 512
|
||||||
|
- "2q1s512"
|
||||||
|
- "2q2s512"
|
||||||
|
- "2q4s512"
|
||||||
|
- "2q8s512"
|
||||||
|
- "2q16s512"
|
||||||
|
- "2q32s512"
|
||||||
|
- "2q64s512"
|
||||||
|
- "2q128s512"
|
||||||
|
- "2q256s512"
|
||||||
|
- "2q512"
|
||||||
|
# seq_len = 1024
|
||||||
|
- "2q1s1024"
|
||||||
|
- "2q2s1024"
|
||||||
|
- "2q4s1024"
|
||||||
|
- "2q8s1024"
|
||||||
|
- "2q16s1024"
|
||||||
|
- "2q32s1024"
|
||||||
|
- "2q64s1024"
|
||||||
|
- "2q128s1024"
|
||||||
|
- "2q256s1024"
|
||||||
|
- "2q512s1024"
|
||||||
|
- "2q1024"
|
||||||
|
# seq_len = 2048
|
||||||
|
- "2q1s2048"
|
||||||
|
- "2q2s2048"
|
||||||
|
- "2q4s2048"
|
||||||
|
- "2q8s2048"
|
||||||
|
- "2q16s2048"
|
||||||
|
- "2q32s2048"
|
||||||
|
- "2q64s2048"
|
||||||
|
- "2q128s2048"
|
||||||
|
- "2q256s2048"
|
||||||
|
- "2q512s2048"
|
||||||
|
- "2q1024s2048"
|
||||||
|
- "2q2048"
|
||||||
|
|
||||||
|
# Batch size 4
|
||||||
|
# seq_len = 32
|
||||||
|
- "4q1s32"
|
||||||
|
- "4q2s32"
|
||||||
|
- "4q4s32"
|
||||||
|
- "4q8s32"
|
||||||
|
- "4q16s32"
|
||||||
|
- "4q32"
|
||||||
|
# seq_len = 64
|
||||||
|
- "4q1s64"
|
||||||
|
- "4q2s64"
|
||||||
|
- "4q4s64"
|
||||||
|
- "4q8s64"
|
||||||
|
- "4q16s64"
|
||||||
|
- "4q32s64"
|
||||||
|
- "4q64"
|
||||||
|
# seq_len = 128
|
||||||
|
- "4q1s128"
|
||||||
|
- "4q2s128"
|
||||||
|
- "4q4s128"
|
||||||
|
- "4q8s128"
|
||||||
|
- "4q16s128"
|
||||||
|
- "4q32s128"
|
||||||
|
- "4q64s128"
|
||||||
|
- "4q128"
|
||||||
|
# seq_len = 256
|
||||||
|
- "4q1s256"
|
||||||
|
- "4q2s256"
|
||||||
|
- "4q4s256"
|
||||||
|
- "4q8s256"
|
||||||
|
- "4q16s256"
|
||||||
|
- "4q32s256"
|
||||||
|
- "4q64s256"
|
||||||
|
- "4q128s256"
|
||||||
|
- "4q256"
|
||||||
|
# seq_len = 512
|
||||||
|
- "4q1s512"
|
||||||
|
- "4q2s512"
|
||||||
|
- "4q4s512"
|
||||||
|
- "4q8s512"
|
||||||
|
- "4q16s512"
|
||||||
|
- "4q32s512"
|
||||||
|
- "4q64s512"
|
||||||
|
- "4q128s512"
|
||||||
|
- "4q256s512"
|
||||||
|
- "4q512"
|
||||||
|
# seq_len = 1024
|
||||||
|
- "4q1s1024"
|
||||||
|
- "4q2s1024"
|
||||||
|
- "4q4s1024"
|
||||||
|
- "4q8s1024"
|
||||||
|
- "4q16s1024"
|
||||||
|
- "4q32s1024"
|
||||||
|
- "4q64s1024"
|
||||||
|
- "4q128s1024"
|
||||||
|
- "4q256s1024"
|
||||||
|
- "4q512s1024"
|
||||||
|
- "4q1024"
|
||||||
|
# seq_len = 2048
|
||||||
|
- "4q1s2048"
|
||||||
|
- "4q2s2048"
|
||||||
|
- "4q4s2048"
|
||||||
|
- "4q8s2048"
|
||||||
|
- "4q16s2048"
|
||||||
|
- "4q32s2048"
|
||||||
|
- "4q64s2048"
|
||||||
|
- "4q128s2048"
|
||||||
|
- "4q256s2048"
|
||||||
|
- "4q512s2048"
|
||||||
|
- "4q1024s2048"
|
||||||
|
- "4q2048"
|
||||||
|
|
||||||
|
# Batch size 8
|
||||||
|
# seq_len = 32
|
||||||
|
- "8q1s32"
|
||||||
|
- "8q2s32"
|
||||||
|
- "8q4s32"
|
||||||
|
- "8q8s32"
|
||||||
|
- "8q16s32"
|
||||||
|
- "8q32"
|
||||||
|
# seq_len = 64
|
||||||
|
- "8q1s64"
|
||||||
|
- "8q2s64"
|
||||||
|
- "8q4s64"
|
||||||
|
- "8q8s64"
|
||||||
|
- "8q16s64"
|
||||||
|
- "8q32s64"
|
||||||
|
- "8q64"
|
||||||
|
# seq_len = 128
|
||||||
|
- "8q1s128"
|
||||||
|
- "8q2s128"
|
||||||
|
- "8q4s128"
|
||||||
|
- "8q8s128"
|
||||||
|
- "8q16s128"
|
||||||
|
- "8q32s128"
|
||||||
|
- "8q64s128"
|
||||||
|
- "8q128"
|
||||||
|
# seq_len = 256
|
||||||
|
- "8q1s256"
|
||||||
|
- "8q2s256"
|
||||||
|
- "8q4s256"
|
||||||
|
- "8q8s256"
|
||||||
|
- "8q16s256"
|
||||||
|
- "8q32s256"
|
||||||
|
- "8q64s256"
|
||||||
|
- "8q128s256"
|
||||||
|
- "8q256"
|
||||||
|
# seq_len = 512
|
||||||
|
- "8q1s512"
|
||||||
|
- "8q2s512"
|
||||||
|
- "8q4s512"
|
||||||
|
- "8q8s512"
|
||||||
|
- "8q16s512"
|
||||||
|
- "8q32s512"
|
||||||
|
- "8q64s512"
|
||||||
|
- "8q128s512"
|
||||||
|
- "8q256s512"
|
||||||
|
- "8q512"
|
||||||
|
# seq_len = 1024
|
||||||
|
- "8q1s1024"
|
||||||
|
- "8q2s1024"
|
||||||
|
- "8q4s1024"
|
||||||
|
- "8q8s1024"
|
||||||
|
- "8q16s1024"
|
||||||
|
- "8q32s1024"
|
||||||
|
- "8q64s1024"
|
||||||
|
- "8q128s1024"
|
||||||
|
- "8q256s1024"
|
||||||
|
- "8q512s1024"
|
||||||
|
- "8q1024"
|
||||||
|
# seq_len = 2048
|
||||||
|
- "8q1s2048"
|
||||||
|
- "8q2s2048"
|
||||||
|
- "8q4s2048"
|
||||||
|
- "8q8s2048"
|
||||||
|
- "8q16s2048"
|
||||||
|
- "8q32s2048"
|
||||||
|
- "8q64s2048"
|
||||||
|
- "8q128s2048"
|
||||||
|
- "8q256s2048"
|
||||||
|
- "8q512s2048"
|
||||||
|
- "8q1024s2048"
|
||||||
|
- "8q2048"
|
||||||
|
|
||||||
|
# Batch size 16
|
||||||
|
# seq_len = 32
|
||||||
|
- "16q1s32"
|
||||||
|
- "16q2s32"
|
||||||
|
- "16q4s32"
|
||||||
|
- "16q8s32"
|
||||||
|
- "16q16s32"
|
||||||
|
- "16q32"
|
||||||
|
# seq_len = 64
|
||||||
|
- "16q1s64"
|
||||||
|
- "16q2s64"
|
||||||
|
- "16q4s64"
|
||||||
|
- "16q8s64"
|
||||||
|
- "16q16s64"
|
||||||
|
- "16q32s64"
|
||||||
|
- "16q64"
|
||||||
|
# seq_len = 128
|
||||||
|
- "16q1s128"
|
||||||
|
- "16q2s128"
|
||||||
|
- "16q4s128"
|
||||||
|
- "16q8s128"
|
||||||
|
- "16q16s128"
|
||||||
|
- "16q32s128"
|
||||||
|
- "16q64s128"
|
||||||
|
- "16q128"
|
||||||
|
# seq_len = 256
|
||||||
|
- "16q1s256"
|
||||||
|
- "16q2s256"
|
||||||
|
- "16q4s256"
|
||||||
|
- "16q8s256"
|
||||||
|
- "16q16s256"
|
||||||
|
- "16q32s256"
|
||||||
|
- "16q64s256"
|
||||||
|
- "16q128s256"
|
||||||
|
- "16q256"
|
||||||
|
# seq_len = 512
|
||||||
|
- "16q1s512"
|
||||||
|
- "16q2s512"
|
||||||
|
- "16q4s512"
|
||||||
|
- "16q8s512"
|
||||||
|
- "16q16s512"
|
||||||
|
- "16q32s512"
|
||||||
|
- "16q64s512"
|
||||||
|
- "16q128s512"
|
||||||
|
- "16q256s512"
|
||||||
|
- "16q512"
|
||||||
|
# seq_len = 1024
|
||||||
|
- "16q1s1024"
|
||||||
|
- "16q2s1024"
|
||||||
|
- "16q4s1024"
|
||||||
|
- "16q8s1024"
|
||||||
|
- "16q16s1024"
|
||||||
|
- "16q32s1024"
|
||||||
|
- "16q64s1024"
|
||||||
|
- "16q128s1024"
|
||||||
|
- "16q256s1024"
|
||||||
|
- "16q512s1024"
|
||||||
|
- "16q1024"
|
||||||
|
# seq_len = 2048
|
||||||
|
- "16q1s2048"
|
||||||
|
- "16q2s2048"
|
||||||
|
- "16q4s2048"
|
||||||
|
- "16q8s2048"
|
||||||
|
- "16q16s2048"
|
||||||
|
- "16q32s2048"
|
||||||
|
- "16q64s2048"
|
||||||
|
- "16q128s2048"
|
||||||
|
- "16q256s2048"
|
||||||
|
- "16q512s2048"
|
||||||
|
- "16q1024s2048"
|
||||||
|
- "16q2048"
|
||||||
|
|
||||||
|
# Batch size 32
|
||||||
|
# seq_len = 32
|
||||||
|
- "32q1s32"
|
||||||
|
- "32q2s32"
|
||||||
|
- "32q4s32"
|
||||||
|
- "32q8s32"
|
||||||
|
- "32q16s32"
|
||||||
|
- "32q32"
|
||||||
|
# seq_len = 64
|
||||||
|
- "32q1s64"
|
||||||
|
- "32q2s64"
|
||||||
|
- "32q4s64"
|
||||||
|
- "32q8s64"
|
||||||
|
- "32q16s64"
|
||||||
|
- "32q32s64"
|
||||||
|
- "32q64"
|
||||||
|
# seq_len = 128
|
||||||
|
- "32q1s128"
|
||||||
|
- "32q2s128"
|
||||||
|
- "32q4s128"
|
||||||
|
- "32q8s128"
|
||||||
|
- "32q16s128"
|
||||||
|
- "32q32s128"
|
||||||
|
- "32q64s128"
|
||||||
|
- "32q128"
|
||||||
|
# seq_len = 256
|
||||||
|
- "32q1s256"
|
||||||
|
- "32q2s256"
|
||||||
|
- "32q4s256"
|
||||||
|
- "32q8s256"
|
||||||
|
- "32q16s256"
|
||||||
|
- "32q32s256"
|
||||||
|
- "32q64s256"
|
||||||
|
- "32q128s256"
|
||||||
|
- "32q256"
|
||||||
|
# seq_len = 512
|
||||||
|
- "32q1s512"
|
||||||
|
- "32q2s512"
|
||||||
|
- "32q4s512"
|
||||||
|
- "32q8s512"
|
||||||
|
- "32q16s512"
|
||||||
|
- "32q32s512"
|
||||||
|
- "32q64s512"
|
||||||
|
- "32q128s512"
|
||||||
|
- "32q256s512"
|
||||||
|
- "32q512"
|
||||||
|
# seq_len = 1024
|
||||||
|
- "32q1s1024"
|
||||||
|
- "32q2s1024"
|
||||||
|
- "32q4s1024"
|
||||||
|
- "32q8s1024"
|
||||||
|
- "32q16s1024"
|
||||||
|
- "32q32s1024"
|
||||||
|
- "32q64s1024"
|
||||||
|
- "32q128s1024"
|
||||||
|
- "32q256s1024"
|
||||||
|
- "32q512s1024"
|
||||||
|
- "32q1024"
|
||||||
|
# seq_len = 2048
|
||||||
|
- "32q1s2048"
|
||||||
|
- "32q2s2048"
|
||||||
|
- "32q4s2048"
|
||||||
|
- "32q8s2048"
|
||||||
|
- "32q16s2048"
|
||||||
|
- "32q32s2048"
|
||||||
|
- "32q64s2048"
|
||||||
|
- "32q128s2048"
|
||||||
|
- "32q256s2048"
|
||||||
|
- "32q512s2048"
|
||||||
|
- "32q1024s2048"
|
||||||
|
- "32q2048"
|
||||||
|
|
||||||
|
backends:
|
||||||
|
- FLASHMLA_SPARSE
|
||||||
|
|
||||||
|
device: "cuda:0"
|
||||||
|
profile_memory: false
|
||||||
|
sparse_mla_dense_mha_max_seq_len: 2048
|
||||||
|
sparse_mla_topk_pattern: "random"
|
||||||
|
|
||||||
|
output:
|
||||||
|
csv: "benchmark_output/mla_sparse_mha_vs_mqa.csv"
|
||||||
|
json: "benchmark_output/mla_sparse_mha_vs_mqa.json"
|
||||||
@@ -9,6 +9,8 @@ needing full VllmConfig integration.
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
import statistics
|
import statistics
|
||||||
|
import tempfile
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import torch
|
import torch
|
||||||
@@ -17,7 +19,6 @@ from common import (
|
|||||||
BenchmarkResult,
|
BenchmarkResult,
|
||||||
MockHfConfig,
|
MockHfConfig,
|
||||||
MockIndexer,
|
MockIndexer,
|
||||||
MockKVBProj,
|
|
||||||
MockLayer,
|
MockLayer,
|
||||||
run_do_bench,
|
run_do_bench,
|
||||||
run_ncu_profile,
|
run_ncu_profile,
|
||||||
@@ -33,8 +34,59 @@ from vllm.config import (
|
|||||||
VllmConfig,
|
VllmConfig,
|
||||||
set_current_vllm_config,
|
set_current_vllm_config,
|
||||||
)
|
)
|
||||||
|
from vllm.model_executor.layers.linear import ColumnParallelLinear
|
||||||
from vllm.v1.attention.backends.mla.prefill.registry import MLAPrefillBackendEnum
|
from vllm.v1.attention.backends.mla.prefill.registry import MLAPrefillBackendEnum
|
||||||
|
|
||||||
|
|
||||||
|
def _safe_profile_name(value: str) -> str:
|
||||||
|
return "".join(c if c.isalnum() or c in "._-" else "_" for c in value)
|
||||||
|
|
||||||
|
|
||||||
|
def _create_kv_b_proj(
|
||||||
|
mla_dims: dict,
|
||||||
|
device: torch.device,
|
||||||
|
):
|
||||||
|
kv_b_proj = ColumnParallelLinear(
|
||||||
|
mla_dims["kv_lora_rank"],
|
||||||
|
mla_dims["num_q_heads"]
|
||||||
|
* (mla_dims["qk_nope_head_dim"] + mla_dims["v_head_dim"]),
|
||||||
|
bias=False,
|
||||||
|
params_dtype=torch.bfloat16,
|
||||||
|
quant_config=None,
|
||||||
|
prefix="benchmark.kv_b_proj",
|
||||||
|
).to(device)
|
||||||
|
with torch.no_grad():
|
||||||
|
kv_b_proj.weight.copy_(torch.randn_like(kv_b_proj.weight))
|
||||||
|
return kv_b_proj
|
||||||
|
|
||||||
|
|
||||||
|
def _ensure_single_rank_model_parallel() -> None:
|
||||||
|
import torch.distributed as dist
|
||||||
|
|
||||||
|
from vllm.distributed import (
|
||||||
|
ensure_model_parallel_initialized,
|
||||||
|
init_distributed_environment,
|
||||||
|
model_parallel_is_initialized,
|
||||||
|
)
|
||||||
|
|
||||||
|
if not dist.is_available():
|
||||||
|
return
|
||||||
|
if not dist.is_initialized():
|
||||||
|
with tempfile.NamedTemporaryFile(
|
||||||
|
prefix="vllm_bench_dist_", delete=False
|
||||||
|
) as init_file:
|
||||||
|
distributed_init_method = f"file://{init_file.name}"
|
||||||
|
init_distributed_environment(
|
||||||
|
world_size=1,
|
||||||
|
rank=0,
|
||||||
|
distributed_init_method=distributed_init_method,
|
||||||
|
local_rank=0,
|
||||||
|
backend="nccl",
|
||||||
|
)
|
||||||
|
if not model_parallel_is_initialized():
|
||||||
|
ensure_model_parallel_initialized(1, 1)
|
||||||
|
|
||||||
|
|
||||||
# ============================================================================
|
# ============================================================================
|
||||||
# VllmConfig Creation
|
# VllmConfig Creation
|
||||||
# ============================================================================
|
# ============================================================================
|
||||||
@@ -66,10 +118,12 @@ def create_minimal_vllm_config(
|
|||||||
block_size: int = 128,
|
block_size: int = 128,
|
||||||
max_num_seqs: int = 256,
|
max_num_seqs: int = 256,
|
||||||
max_num_batched_tokens: int = 8192,
|
max_num_batched_tokens: int = 8192,
|
||||||
|
max_model_len: int = 32768,
|
||||||
mla_dims: dict | None = None,
|
mla_dims: dict | None = None,
|
||||||
index_topk: int | None = None,
|
index_topk: int | None = None,
|
||||||
prefill_backend: str | None = None,
|
prefill_backend: str | None = None,
|
||||||
kv_cache_dtype: str = "auto",
|
kv_cache_dtype: str = "auto",
|
||||||
|
sparse_mla_force_mqa: bool = False,
|
||||||
) -> VllmConfig:
|
) -> VllmConfig:
|
||||||
"""
|
"""
|
||||||
Create minimal VllmConfig for MLA benchmarks.
|
Create minimal VllmConfig for MLA benchmarks.
|
||||||
@@ -86,6 +140,8 @@ def create_minimal_vllm_config(
|
|||||||
prefill_backend: Prefill backend name (e.g., "fa3", "fa4", "flashinfer",
|
prefill_backend: Prefill backend name (e.g., "fa3", "fa4", "flashinfer",
|
||||||
"trtllm"). Configures the attention config to force
|
"trtllm"). Configures the attention config to force
|
||||||
the specified prefill backend.
|
the specified prefill backend.
|
||||||
|
sparse_mla_force_mqa: If True, forces all sparse MLA tokens through
|
||||||
|
forward_mqa (even prefill tokens).
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
VllmConfig for benchmarking
|
VllmConfig for benchmarking
|
||||||
@@ -131,7 +187,7 @@ def create_minimal_vllm_config(
|
|||||||
trust_remote_code=True,
|
trust_remote_code=True,
|
||||||
dtype="bfloat16",
|
dtype="bfloat16",
|
||||||
seed=0,
|
seed=0,
|
||||||
max_model_len=32768,
|
max_model_len=max_model_len,
|
||||||
quantization=None,
|
quantization=None,
|
||||||
enforce_eager=False,
|
enforce_eager=False,
|
||||||
max_logprobs=20,
|
max_logprobs=20,
|
||||||
@@ -163,7 +219,7 @@ def create_minimal_vllm_config(
|
|||||||
scheduler_config = SchedulerConfig(
|
scheduler_config = SchedulerConfig(
|
||||||
max_num_seqs=max_num_seqs,
|
max_num_seqs=max_num_seqs,
|
||||||
max_num_batched_tokens=max(max_num_batched_tokens, max_num_seqs),
|
max_num_batched_tokens=max(max_num_batched_tokens, max_num_seqs),
|
||||||
max_model_len=32768,
|
max_model_len=max_model_len,
|
||||||
is_encoder_decoder=False,
|
is_encoder_decoder=False,
|
||||||
enable_chunked_prefill=True,
|
enable_chunked_prefill=True,
|
||||||
)
|
)
|
||||||
@@ -192,6 +248,9 @@ def create_minimal_vllm_config(
|
|||||||
"flash_attn_version"
|
"flash_attn_version"
|
||||||
]
|
]
|
||||||
|
|
||||||
|
if sparse_mla_force_mqa:
|
||||||
|
vllm_config.attention_config.sparse_mla_force_mqa = True
|
||||||
|
|
||||||
return vllm_config
|
return vllm_config
|
||||||
|
|
||||||
|
|
||||||
@@ -548,12 +607,7 @@ def _create_backend_impl(
|
|||||||
# Calculate scale
|
# Calculate scale
|
||||||
scale = 1.0 / np.sqrt(mla_dims["qk_nope_head_dim"] + mla_dims["qk_rope_head_dim"])
|
scale = 1.0 / np.sqrt(mla_dims["qk_nope_head_dim"] + mla_dims["qk_rope_head_dim"])
|
||||||
|
|
||||||
# Create mock kv_b_proj layer for prefill mode
|
kv_b_proj = _create_kv_b_proj(mla_dims, device)
|
||||||
mock_kv_b_proj = MockKVBProj(
|
|
||||||
num_heads=mla_dims["num_q_heads"],
|
|
||||||
qk_nope_head_dim=mla_dims["qk_nope_head_dim"],
|
|
||||||
v_head_dim=mla_dims["v_head_dim"],
|
|
||||||
)
|
|
||||||
|
|
||||||
# Create indexer for sparse backends
|
# Create indexer for sparse backends
|
||||||
indexer = None
|
indexer = None
|
||||||
@@ -584,7 +638,7 @@ def _create_backend_impl(
|
|||||||
"qk_rope_head_dim": mla_dims["qk_rope_head_dim"],
|
"qk_rope_head_dim": mla_dims["qk_rope_head_dim"],
|
||||||
"qk_head_dim": mla_dims["qk_nope_head_dim"] + mla_dims["qk_rope_head_dim"],
|
"qk_head_dim": mla_dims["qk_nope_head_dim"] + mla_dims["qk_rope_head_dim"],
|
||||||
"v_head_dim": mla_dims["v_head_dim"],
|
"v_head_dim": mla_dims["v_head_dim"],
|
||||||
"kv_b_proj": mock_kv_b_proj,
|
"kv_b_proj": kv_b_proj,
|
||||||
}
|
}
|
||||||
|
|
||||||
# Add indexer for sparse backends
|
# Add indexer for sparse backends
|
||||||
@@ -785,14 +839,35 @@ def _run_single_benchmark(
|
|||||||
# Fill indexer with random indices for sparse backends
|
# Fill indexer with random indices for sparse backends
|
||||||
is_sparse = backend_cfg.get("is_sparse", False)
|
is_sparse = backend_cfg.get("is_sparse", False)
|
||||||
if is_sparse and indexer is not None:
|
if is_sparse and indexer is not None:
|
||||||
indexer.fill_random_indices(total_q, max_kv_len)
|
indexer.fill_indices(
|
||||||
|
total_q,
|
||||||
|
max_kv_len,
|
||||||
|
getattr(config, "sparse_mla_topk_pattern", "random"),
|
||||||
|
)
|
||||||
|
|
||||||
# Determine which forward methods to use based on metadata.
|
# Determine which forward methods to use based on metadata.
|
||||||
# Sparse MLA backends always use forward_mqa
|
# Non-sparse backends use .decode/.prefill sub-objects.
|
||||||
has_decode = is_sparse or getattr(metadata, "decode", None) is not None
|
# Sparse backends use num_decode_tokens/num_prefills directly.
|
||||||
has_prefill = not is_sparse and getattr(metadata, "prefill", None) is not None
|
#
|
||||||
|
# sparse_mla_force_mqa overrides: even for prefill metadata, use MQA.
|
||||||
|
force_mqa = getattr(config, "sparse_mla_force_mqa", False)
|
||||||
|
force_dense_mha = getattr(config, "sparse_mla_mha_mode", "auto") == "dense"
|
||||||
|
if force_mqa:
|
||||||
|
has_decode = True
|
||||||
|
has_prefill = False
|
||||||
|
elif is_sparse:
|
||||||
|
has_decode = metadata.num_decode_tokens > 0
|
||||||
|
has_prefill = metadata.num_prefills > 0
|
||||||
|
else:
|
||||||
|
has_decode = metadata.decode is not None
|
||||||
|
has_prefill = metadata.prefill is not None
|
||||||
if not has_decode and not has_prefill:
|
if not has_decode and not has_prefill:
|
||||||
raise RuntimeError("Metadata has neither decode nor prefill metadata")
|
raise RuntimeError("Metadata has neither decode nor prefill metadata")
|
||||||
|
if is_sparse and force_dense_mha and not has_prefill:
|
||||||
|
raise RuntimeError(
|
||||||
|
"Sparse MLA dense_mha benchmark did not produce prefill metadata. "
|
||||||
|
"Check reorder_batch_threshold/path forcing."
|
||||||
|
)
|
||||||
|
|
||||||
num_decode = (
|
num_decode = (
|
||||||
metadata.num_decode_tokens
|
metadata.num_decode_tokens
|
||||||
@@ -871,7 +946,6 @@ def _run_single_benchmark(
|
|||||||
metadata,
|
metadata,
|
||||||
prefill_inputs["k_scale"],
|
prefill_inputs["k_scale"],
|
||||||
prefill_fp8_output if fused_output else prefill_inputs["output"],
|
prefill_fp8_output if fused_output else prefill_inputs["output"],
|
||||||
prefill_output_scale if fused_output else None,
|
|
||||||
)
|
)
|
||||||
if fused_output:
|
if fused_output:
|
||||||
out = prefill_fp8_output
|
out = prefill_fp8_output
|
||||||
@@ -898,6 +972,48 @@ def _run_single_benchmark(
|
|||||||
throughput_tokens_per_sec=0.0,
|
throughput_tokens_per_sec=0.0,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if config.torch_profile:
|
||||||
|
profile_dir = Path(
|
||||||
|
config.torch_profile_dir or "benchmark_outputs/torch_profiles"
|
||||||
|
)
|
||||||
|
profile_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
trace_name = _safe_profile_name(f"{config.backend}_{config.batch_spec}")
|
||||||
|
trace_path = profile_dir / f"{trace_name}.json"
|
||||||
|
iters = max(config.torch_profile_iters, 1)
|
||||||
|
|
||||||
|
forward_fn()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
with torch.profiler.profile(
|
||||||
|
activities=[
|
||||||
|
torch.profiler.ProfilerActivity.CPU,
|
||||||
|
torch.profiler.ProfilerActivity.CUDA,
|
||||||
|
],
|
||||||
|
record_shapes=True,
|
||||||
|
profile_memory=True,
|
||||||
|
with_stack=False,
|
||||||
|
) as prof:
|
||||||
|
for _ in range(iters):
|
||||||
|
forward_fn()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
prof.step()
|
||||||
|
prof.export_chrome_trace(str(trace_path))
|
||||||
|
print(f"Saved PyTorch profiler trace to {trace_path}")
|
||||||
|
print(
|
||||||
|
prof.key_averages().table(
|
||||||
|
sort_by="cuda_time_total",
|
||||||
|
row_limit=25,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return BenchmarkResult(
|
||||||
|
config=config,
|
||||||
|
mean_time=0.0,
|
||||||
|
median_time=0.0,
|
||||||
|
std_time=0.0,
|
||||||
|
min_time=0.0,
|
||||||
|
max_time=0.0,
|
||||||
|
throughput_tokens_per_sec=0.0,
|
||||||
|
)
|
||||||
|
|
||||||
all_ms = run_do_bench(benchmark_fn, config.use_cuda_graphs, config.warmup_ms)
|
all_ms = run_do_bench(benchmark_fn, config.use_cuda_graphs, config.warmup_ms)
|
||||||
|
|
||||||
# Convert ms to seconds per layer
|
# Convert ms to seconds per layer
|
||||||
@@ -920,6 +1036,7 @@ def _run_mla_benchmark_batched(
|
|||||||
configs_with_params: list[tuple], # [(config, threshold, num_splits), ...]
|
configs_with_params: list[tuple], # [(config, threshold, num_splits), ...]
|
||||||
index_topk: int = 2048,
|
index_topk: int = 2048,
|
||||||
prefill_backend: str | None = None,
|
prefill_backend: str | None = None,
|
||||||
|
sparse_mla_force_mqa: bool = False,
|
||||||
output_scale: float | None = None,
|
output_scale: float | None = None,
|
||||||
fuse_quant_op: bool = False,
|
fuse_quant_op: bool = False,
|
||||||
) -> list[BenchmarkResult]:
|
) -> list[BenchmarkResult]:
|
||||||
@@ -940,6 +1057,8 @@ def _run_mla_benchmark_batched(
|
|||||||
index_topk: Topk value for sparse MLA backends (default 2048)
|
index_topk: Topk value for sparse MLA backends (default 2048)
|
||||||
prefill_backend: Prefill backend name (e.g., "fa3", "fa4").
|
prefill_backend: Prefill backend name (e.g., "fa3", "fa4").
|
||||||
When set, forces the specified FlashAttention version for prefill.
|
When set, forces the specified FlashAttention version for prefill.
|
||||||
|
sparse_mla_force_mqa: If True, forces all sparse MLA tokens through
|
||||||
|
forward_mqa (even prefill tokens).
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
List of BenchmarkResult objects
|
List of BenchmarkResult objects
|
||||||
@@ -980,21 +1099,41 @@ def _run_mla_benchmark_batched(
|
|||||||
sum(r.q_len for r in parse_batch_spec(cfg.batch_spec))
|
sum(r.q_len for r in parse_batch_spec(cfg.batch_spec))
|
||||||
for cfg, *_ in configs_with_params
|
for cfg, *_ in configs_with_params
|
||||||
)
|
)
|
||||||
|
max_model_len = max(
|
||||||
|
max_total_q,
|
||||||
|
max(
|
||||||
|
getattr(cfg, "max_model_len", None) or 32768
|
||||||
|
for cfg, *_ in configs_with_params
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
# Create and set vLLM config for MLA (reused across all benchmarks)
|
# Create and set vLLM config for MLA (reused across all benchmarks)
|
||||||
vllm_config = create_minimal_vllm_config(
|
vllm_config = create_minimal_vllm_config(
|
||||||
model_name="deepseek-v3", # Used only for model path
|
model_name="deepseek-v3", # Used only for model path
|
||||||
block_size=block_size,
|
block_size=block_size,
|
||||||
max_num_batched_tokens=max_total_q,
|
max_num_batched_tokens=max_total_q,
|
||||||
|
max_model_len=max_model_len,
|
||||||
mla_dims=mla_dims, # Use custom dims from config or default
|
mla_dims=mla_dims, # Use custom dims from config or default
|
||||||
index_topk=index_topk if is_sparse else None,
|
index_topk=index_topk if is_sparse else None,
|
||||||
prefill_backend=prefill_backend,
|
prefill_backend=prefill_backend,
|
||||||
kv_cache_dtype=kv_cache_dtype,
|
kv_cache_dtype=kv_cache_dtype,
|
||||||
|
sparse_mla_force_mqa=sparse_mla_force_mqa,
|
||||||
)
|
)
|
||||||
|
|
||||||
results = []
|
results = []
|
||||||
|
|
||||||
|
# Initialize workspace manager (needed by metadata builders)
|
||||||
|
from vllm.v1.worker.workspace import (
|
||||||
|
init_workspace_manager,
|
||||||
|
is_workspace_manager_initialized,
|
||||||
|
)
|
||||||
|
|
||||||
|
if not is_workspace_manager_initialized():
|
||||||
|
init_workspace_manager(device)
|
||||||
|
|
||||||
with set_current_vllm_config(vllm_config):
|
with set_current_vllm_config(vllm_config):
|
||||||
|
_ensure_single_rank_model_parallel()
|
||||||
|
|
||||||
# Create backend impl, layer, builder, and indexer (reused across benchmarks)
|
# Create backend impl, layer, builder, and indexer (reused across benchmarks)
|
||||||
impl, layer, builder_instance, indexer = _create_backend_impl(
|
impl, layer, builder_instance, indexer = _create_backend_impl(
|
||||||
backend_cfg,
|
backend_cfg,
|
||||||
@@ -1040,9 +1179,20 @@ def _run_mla_benchmark_batched(
|
|||||||
for config, threshold, num_splits in configs_with_params:
|
for config, threshold, num_splits in configs_with_params:
|
||||||
# Set threshold for this benchmark (FlashAttn/FlashMLA only)
|
# Set threshold for this benchmark (FlashAttn/FlashMLA only)
|
||||||
original_threshold = None
|
original_threshold = None
|
||||||
if threshold is not None and builder_instance:
|
effective_threshold = threshold
|
||||||
|
force_dense_mha = (
|
||||||
|
is_sparse
|
||||||
|
and getattr(config, "sparse_mla_mha_mode", "auto") == "dense"
|
||||||
|
and not getattr(config, "sparse_mla_force_mqa", False)
|
||||||
|
)
|
||||||
|
if force_dense_mha:
|
||||||
|
# Sparse MLA normally treats q_len <= 1 as decode. Use an
|
||||||
|
# impossible threshold so dense_mha benchmarks actually run
|
||||||
|
# the prefill/MHA path, including q_len=1 short extends.
|
||||||
|
effective_threshold = -1
|
||||||
|
if effective_threshold is not None and builder_instance:
|
||||||
original_threshold = builder_instance.reorder_batch_threshold
|
original_threshold = builder_instance.reorder_batch_threshold
|
||||||
builder_instance.reorder_batch_threshold = threshold
|
builder_instance.reorder_batch_threshold = effective_threshold
|
||||||
|
|
||||||
# Set num_splits for CUTLASS
|
# Set num_splits for CUTLASS
|
||||||
original_num_splits = None
|
original_num_splits = None
|
||||||
@@ -1090,6 +1240,7 @@ def run_mla_benchmark(
|
|||||||
num_kv_splits: int | None = None,
|
num_kv_splits: int | None = None,
|
||||||
index_topk: int = 2048,
|
index_topk: int = 2048,
|
||||||
prefill_backend: str | None = None,
|
prefill_backend: str | None = None,
|
||||||
|
sparse_mla_force_mqa: bool = False,
|
||||||
output_scale: float | None = None,
|
output_scale: float | None = None,
|
||||||
fuse_quant_op: bool = False,
|
fuse_quant_op: bool = False,
|
||||||
) -> BenchmarkResult | list[BenchmarkResult]:
|
) -> BenchmarkResult | list[BenchmarkResult]:
|
||||||
@@ -1111,6 +1262,8 @@ def run_mla_benchmark(
|
|||||||
index_topk: Topk value for sparse MLA backends (default 2048)
|
index_topk: Topk value for sparse MLA backends (default 2048)
|
||||||
prefill_backend: Prefill backend name (e.g., "fa3", "fa4").
|
prefill_backend: Prefill backend name (e.g., "fa3", "fa4").
|
||||||
When set, forces the specified FlashAttention version for prefill.
|
When set, forces the specified FlashAttention version for prefill.
|
||||||
|
sparse_mla_force_mqa: If True, forces all sparse MLA tokens through
|
||||||
|
forward_mqa (even prefill tokens).
|
||||||
output_scale: Static per-tensor FP8 scale for prefill output (None = bf16).
|
output_scale: Static per-tensor FP8 scale for prefill output (None = bf16).
|
||||||
fuse_quant_op: With output_scale set, fuse the FP8 write into the prefill
|
fuse_quant_op: With output_scale set, fuse the FP8 write into the prefill
|
||||||
kernel vs a standalone post-quant kernel. See _run_single_benchmark.
|
kernel vs a standalone post-quant kernel. See _run_single_benchmark.
|
||||||
@@ -1142,6 +1295,7 @@ def run_mla_benchmark(
|
|||||||
configs_with_params,
|
configs_with_params,
|
||||||
index_topk,
|
index_topk,
|
||||||
prefill_backend=prefill_backend,
|
prefill_backend=prefill_backend,
|
||||||
|
sparse_mla_force_mqa=sparse_mla_force_mqa,
|
||||||
output_scale=output_scale,
|
output_scale=output_scale,
|
||||||
fuse_quant_op=fuse_quant_op,
|
fuse_quant_op=fuse_quant_op,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -69,12 +69,11 @@ def make_inputs(total_tokens, num_reqs, block_size):
|
|||||||
# Output workspace
|
# Output workspace
|
||||||
dst = torch.zeros(total_tokens, HEAD_DIM, dtype=torch.bfloat16, device="cuda")
|
dst = torch.zeros(total_tokens, HEAD_DIM, dtype=torch.bfloat16, device="cuda")
|
||||||
|
|
||||||
seq_lens_t = torch.tensor(seq_lens, dtype=torch.int32, device="cuda")
|
|
||||||
workspace_starts_t = torch.tensor(
|
workspace_starts_t = torch.tensor(
|
||||||
workspace_starts, dtype=torch.int32, device="cuda"
|
workspace_starts, dtype=torch.int32, device="cuda"
|
||||||
)
|
)
|
||||||
|
|
||||||
return cache, dst, block_table, seq_lens_t, workspace_starts_t
|
return cache, dst, block_table, workspace_starts_t
|
||||||
|
|
||||||
|
|
||||||
def bench_scenario(label, num_reqs, total_tokens_list, save_path):
|
def bench_scenario(label, num_reqs, total_tokens_list, save_path):
|
||||||
@@ -94,7 +93,7 @@ def bench_scenario(label, num_reqs, total_tokens_list, save_path):
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
def bench_fn(total_tokens, provider, num_reqs):
|
def bench_fn(total_tokens, provider, num_reqs):
|
||||||
cache, dst, block_table, seq_lens_t, ws_starts = make_inputs(
|
cache, dst, block_table, ws_starts = make_inputs(
|
||||||
total_tokens, num_reqs, BLOCK_SIZE
|
total_tokens, num_reqs, BLOCK_SIZE
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -102,7 +101,7 @@ def bench_scenario(label, num_reqs, total_tokens_list, save_path):
|
|||||||
|
|
||||||
ms, min_ms, max_ms = triton.testing.do_bench_cudagraph(
|
ms, min_ms, max_ms = triton.testing.do_bench_cudagraph(
|
||||||
lambda: ops.cp_gather_and_upconvert_fp8_kv_cache(
|
lambda: ops.cp_gather_and_upconvert_fp8_kv_cache(
|
||||||
cache, dst, block_table, seq_lens_t, ws_starts, num_reqs
|
cache, dst, block_table, ws_starts, num_reqs
|
||||||
),
|
),
|
||||||
quantiles=quantiles,
|
quantiles=quantiles,
|
||||||
rep=500,
|
rep=500,
|
||||||
|
|||||||
@@ -0,0 +1,176 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
|
||||||
|
import statistics
|
||||||
|
|
||||||
|
import torch
|
||||||
|
from tabulate import tabulate
|
||||||
|
|
||||||
|
from vllm.models.inkling.nvidia.ops import qkvr_prep
|
||||||
|
from vllm.utils.argparse_utils import FlexibleArgumentParser
|
||||||
|
|
||||||
|
|
||||||
|
def make_inputs(tokens: int, tp_size: int, is_local: bool):
|
||||||
|
torch.manual_seed(0)
|
||||||
|
num_q_heads = 64 // tp_size
|
||||||
|
num_kv_heads = (16 if is_local else 8) // tp_size
|
||||||
|
head_dim = 128
|
||||||
|
d_rel = 16
|
||||||
|
rel_extent = 512 if is_local else 1024
|
||||||
|
page_size = 16
|
||||||
|
num_blocks = (tokens + page_size - 1) // page_size
|
||||||
|
q_width = num_q_heads * head_dim
|
||||||
|
kv_width = num_kv_heads * head_dim
|
||||||
|
r_width = num_q_heads * d_rel
|
||||||
|
device = "cuda"
|
||||||
|
|
||||||
|
qkvr = torch.randn(
|
||||||
|
tokens,
|
||||||
|
q_width + 2 * kv_width + r_width,
|
||||||
|
device=device,
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
)
|
||||||
|
k_weight = torch.randn(kv_width, 4, device=device, dtype=torch.bfloat16)
|
||||||
|
v_weight = torch.randn_like(k_weight)
|
||||||
|
q_norm_weight = torch.randn(head_dim, device=device, dtype=torch.bfloat16)
|
||||||
|
k_norm_weight = torch.randn_like(q_norm_weight)
|
||||||
|
rel_proj = torch.randn(d_rel, rel_extent, device=device, dtype=torch.bfloat16)
|
||||||
|
conv_cache = torch.zeros(
|
||||||
|
num_blocks,
|
||||||
|
num_kv_heads,
|
||||||
|
page_size,
|
||||||
|
2 * head_dim,
|
||||||
|
device=device,
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
)
|
||||||
|
key_cache = torch.empty(
|
||||||
|
num_blocks,
|
||||||
|
page_size,
|
||||||
|
num_kv_heads,
|
||||||
|
head_dim,
|
||||||
|
device=device,
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
)
|
||||||
|
value_cache = torch.empty_like(key_cache)
|
||||||
|
positions = torch.arange(tokens, device=device, dtype=torch.int64)
|
||||||
|
block_table = torch.arange(num_blocks, device=device, dtype=torch.int32)[None]
|
||||||
|
seq_idx = torch.zeros(tokens, device=device, dtype=torch.int32)
|
||||||
|
slots = torch.arange(tokens, device=device, dtype=torch.int64)
|
||||||
|
query_start = torch.zeros(tokens, device=device, dtype=torch.int32)
|
||||||
|
log_scaling = None
|
||||||
|
if not is_local:
|
||||||
|
effective_n = (positions + 1).to(torch.float32)
|
||||||
|
log_scaling = 1.0 + 0.1 * torch.log(torch.clamp(effective_n / 128000, min=1.0))
|
||||||
|
return (
|
||||||
|
qkvr,
|
||||||
|
k_weight,
|
||||||
|
v_weight,
|
||||||
|
q_norm_weight,
|
||||||
|
k_norm_weight,
|
||||||
|
rel_proj,
|
||||||
|
1e-6,
|
||||||
|
num_q_heads,
|
||||||
|
num_kv_heads,
|
||||||
|
head_dim,
|
||||||
|
d_rel,
|
||||||
|
conv_cache,
|
||||||
|
key_cache,
|
||||||
|
value_cache,
|
||||||
|
positions,
|
||||||
|
block_table,
|
||||||
|
seq_idx,
|
||||||
|
slots,
|
||||||
|
query_start,
|
||||||
|
slots,
|
||||||
|
0,
|
||||||
|
head_dim,
|
||||||
|
page_size,
|
||||||
|
log_scaling,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def capture(implementation, inputs):
|
||||||
|
outputs = []
|
||||||
|
|
||||||
|
def run():
|
||||||
|
outputs[:] = implementation.fused_qkvr_prep(*inputs)
|
||||||
|
|
||||||
|
stream = torch.cuda.Stream()
|
||||||
|
stream.wait_stream(torch.cuda.current_stream())
|
||||||
|
with torch.cuda.stream(stream):
|
||||||
|
for _ in range(3):
|
||||||
|
run()
|
||||||
|
torch.cuda.current_stream().wait_stream(stream)
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
graph = torch.cuda.CUDAGraph()
|
||||||
|
with torch.cuda.graph(graph):
|
||||||
|
run()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
return graph, outputs
|
||||||
|
|
||||||
|
|
||||||
|
def time_graph(graph: torch.cuda.CUDAGraph, warmup: int, repeats: int) -> float:
|
||||||
|
for _ in range(warmup):
|
||||||
|
graph.replay()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
start = torch.cuda.Event(enable_timing=True)
|
||||||
|
end = torch.cuda.Event(enable_timing=True)
|
||||||
|
start.record()
|
||||||
|
for _ in range(repeats):
|
||||||
|
graph.replay()
|
||||||
|
end.record()
|
||||||
|
end.synchronize()
|
||||||
|
return start.elapsed_time(end) * 1000 / repeats
|
||||||
|
|
||||||
|
|
||||||
|
def benchmark(inputs, args) -> float:
|
||||||
|
graph, _ = capture(qkvr_prep, inputs)
|
||||||
|
return statistics.median(
|
||||||
|
time_graph(graph, args.warmup, args.repeats) for _ in range(args.trials)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@torch.inference_mode()
|
||||||
|
def main(args):
|
||||||
|
rows = []
|
||||||
|
for tp_size in args.tp_sizes:
|
||||||
|
for tokens in args.tokens:
|
||||||
|
for is_local in (True, False):
|
||||||
|
triton_us = benchmark(make_inputs(tokens, tp_size, is_local), args)
|
||||||
|
rows.append(
|
||||||
|
[
|
||||||
|
tp_size,
|
||||||
|
tokens,
|
||||||
|
"local" if is_local else "global",
|
||||||
|
triton_us,
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
print("Inkling QKVR prep (CUDA graph, median latency)")
|
||||||
|
print(
|
||||||
|
tabulate(
|
||||||
|
rows,
|
||||||
|
headers=[
|
||||||
|
"TP",
|
||||||
|
"tokens",
|
||||||
|
"scope",
|
||||||
|
"Triton (us)",
|
||||||
|
],
|
||||||
|
floatfmt=("d", "d", "", ".2f"),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
parser = FlexibleArgumentParser()
|
||||||
|
parser.add_argument(
|
||||||
|
"--tokens",
|
||||||
|
type=int,
|
||||||
|
nargs="+",
|
||||||
|
default=[1 << power for power in range(15)],
|
||||||
|
)
|
||||||
|
parser.add_argument("--tp-sizes", type=int, nargs="+", default=[4, 8])
|
||||||
|
parser.add_argument("--warmup", type=int, default=20)
|
||||||
|
parser.add_argument("--repeats", type=int, default=200)
|
||||||
|
parser.add_argument("--trials", type=int, default=5)
|
||||||
|
main(parser.parse_args())
|
||||||
@@ -0,0 +1,367 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
"""Benchmark the Kimi-K3 latent MoE addmm against CuTe residual GEMM.
|
||||||
|
|
||||||
|
The benchmark covers ``BF16[M, 3584] @ BF16[7168, 3584].T + BF16[M, 7168]``
|
||||||
|
with FP32 accumulation and BF16 output. Both backends execute through CUDA
|
||||||
|
Graph replay. Weights and residuals rotate across buffers exceeding L2 so the
|
||||||
|
comparison models the full latent MoE projection-and-add path.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import dataclasses
|
||||||
|
import importlib.util
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
import statistics
|
||||||
|
from collections.abc import Callable, Sequence
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import cutlass
|
||||||
|
import cutlass.cute as cute
|
||||||
|
import torch
|
||||||
|
from cuda.bindings import driver as cuda
|
||||||
|
from cuda.bindings.driver import CUstream
|
||||||
|
from quack.compile_utils import make_fake_tensor
|
||||||
|
|
||||||
|
N = 7168
|
||||||
|
K = 3584
|
||||||
|
|
||||||
|
|
||||||
|
@dataclasses.dataclass(frozen=True, slots=True)
|
||||||
|
class Config:
|
||||||
|
block_size: int
|
||||||
|
outputs_per_block: int
|
||||||
|
k_unroll: int
|
||||||
|
vector_width: int = 8
|
||||||
|
|
||||||
|
|
||||||
|
def parse_config(value: str) -> Config:
|
||||||
|
try:
|
||||||
|
parts = [int(part) for part in value.split(",")]
|
||||||
|
except ValueError as error:
|
||||||
|
raise argparse.ArgumentTypeError(
|
||||||
|
"config must be BLOCK,OUTPUTS,K_UNROLL[,VECTOR_WIDTH]"
|
||||||
|
) from error
|
||||||
|
if len(parts) == 3:
|
||||||
|
return Config(*parts)
|
||||||
|
if len(parts) == 4:
|
||||||
|
return Config(*parts)
|
||||||
|
raise argparse.ArgumentTypeError(
|
||||||
|
"config must be BLOCK,OUTPUTS,K_UNROLL[,VECTOR_WIDTH]"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def production_residual_config(m: int) -> Config | None:
|
||||||
|
"""The measured Latent-MoE residual config for M, from the K3 table."""
|
||||||
|
from vllm.models.kimi_k3.nvidia.low_latency_gemm import KIMI_K3_PROJECTIONS
|
||||||
|
|
||||||
|
spec = KIMI_K3_PROJECTIONS.get((N, K))
|
||||||
|
config = spec.residual_config(m) if spec is not None else None
|
||||||
|
if config is None:
|
||||||
|
return None
|
||||||
|
return Config(
|
||||||
|
config.block_size,
|
||||||
|
config.outputs_per_block,
|
||||||
|
config.k_unroll,
|
||||||
|
config.vector_width,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def candidate_configs(mode: str, selected: Config | None, m: int) -> list[Config]:
|
||||||
|
if mode == "selected":
|
||||||
|
if selected is not None:
|
||||||
|
return [selected]
|
||||||
|
# No explicit --config: fall back to the production table for this M.
|
||||||
|
config = production_residual_config(m)
|
||||||
|
return [config] if config is not None else []
|
||||||
|
if mode == "baseline":
|
||||||
|
return [Config(224, 4, 2)]
|
||||||
|
return [
|
||||||
|
Config(block_size, outputs_per_block, k_unroll, vector_width)
|
||||||
|
for vector_width in (4, 8)
|
||||||
|
for block_size in (32, 64, 128, 224, 448)
|
||||||
|
if block_size % 32 == 0 and K % (block_size * vector_width) == 0
|
||||||
|
for outputs_per_block in (1, 2, 4, 7, 8)
|
||||||
|
if N % outputs_per_block == 0
|
||||||
|
for k_unroll in (1, 2, 4)
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def load_kernel_class(path: Path):
|
||||||
|
spec = importlib.util.spec_from_file_location("cute_skinny_device", path)
|
||||||
|
if spec is None or spec.loader is None:
|
||||||
|
raise RuntimeError(f"cannot load CuTe kernel from {path}")
|
||||||
|
module = importlib.util.module_from_spec(spec)
|
||||||
|
spec.loader.exec_module(module)
|
||||||
|
return module.CuteSkinnyGemm
|
||||||
|
|
||||||
|
|
||||||
|
def stream() -> CUstream:
|
||||||
|
return CUstream(torch.cuda.current_stream().cuda_stream)
|
||||||
|
|
||||||
|
|
||||||
|
def compile_kernel(kernel_class, m: int, config: Config, max_registers: int):
|
||||||
|
element_type = cutlass.BFloat16
|
||||||
|
n = cute.sym_int(divisibility=config.outputs_per_block)
|
||||||
|
k = cute.sym_int(divisibility=config.block_size * config.vector_width)
|
||||||
|
a = make_fake_tensor(element_type, (m, k), divisibility=config.vector_width)
|
||||||
|
b = make_fake_tensor(element_type, (n, k), divisibility=config.vector_width)
|
||||||
|
residual = make_fake_tensor(element_type, (m, n), divisibility=1)
|
||||||
|
c = make_fake_tensor(element_type, (m, n), divisibility=1)
|
||||||
|
kernel = kernel_class(
|
||||||
|
element_type=element_type,
|
||||||
|
num_rows=m,
|
||||||
|
block_size=config.block_size,
|
||||||
|
outputs_per_block=config.outputs_per_block,
|
||||||
|
vector_width=config.vector_width,
|
||||||
|
k_unroll=config.k_unroll,
|
||||||
|
has_residual=True,
|
||||||
|
use_pdl=True,
|
||||||
|
)
|
||||||
|
return cute.compile(
|
||||||
|
kernel,
|
||||||
|
a,
|
||||||
|
b,
|
||||||
|
residual,
|
||||||
|
c,
|
||||||
|
stream(),
|
||||||
|
options=(
|
||||||
|
"--enable-tvm-ffi --keep-cubin "
|
||||||
|
f"--ptxas-options -maxrregcount={max_registers} "
|
||||||
|
"--ptxas-options -lineinfo"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def resource_usage(compiled) -> dict[str, Any]:
|
||||||
|
executor = getattr(compiled, "_default_executor", None)
|
||||||
|
context = getattr(executor, "exec_context", None)
|
||||||
|
functions = getattr(context, "kernel_functions", None)
|
||||||
|
if not functions:
|
||||||
|
return {"resource_metrics_available": False}
|
||||||
|
|
||||||
|
def attribute(name, function) -> int:
|
||||||
|
error, value = cuda.cuFuncGetAttribute(name, function)
|
||||||
|
if error != cuda.CUresult.CUDA_SUCCESS:
|
||||||
|
raise RuntimeError(f"cuFuncGetAttribute failed with {error}")
|
||||||
|
return int(value)
|
||||||
|
|
||||||
|
registers = [
|
||||||
|
attribute(cuda.CUfunction_attribute.CU_FUNC_ATTRIBUTE_NUM_REGS, function)
|
||||||
|
for function in functions
|
||||||
|
]
|
||||||
|
local_bytes = [
|
||||||
|
attribute(
|
||||||
|
cuda.CUfunction_attribute.CU_FUNC_ATTRIBUTE_LOCAL_SIZE_BYTES,
|
||||||
|
function,
|
||||||
|
)
|
||||||
|
for function in functions
|
||||||
|
]
|
||||||
|
return {
|
||||||
|
"resource_metrics_available": True,
|
||||||
|
"registers_per_thread": max(registers, default=0),
|
||||||
|
"spill_bytes": max(local_bytes, default=0),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def rotating_buffer_count(m: int, multiplier: float, limit: int) -> int:
|
||||||
|
properties = torch.cuda.get_device_properties(0)
|
||||||
|
bytes_per_pair = (N * K + m * N) * 2
|
||||||
|
target = math.ceil(multiplier * properties.L2_cache_size)
|
||||||
|
return max(2, min(limit, math.ceil(target / bytes_per_pair)))
|
||||||
|
|
||||||
|
|
||||||
|
def graph_samples(
|
||||||
|
launch: Callable[[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor], None],
|
||||||
|
activation: torch.Tensor,
|
||||||
|
weights: Sequence[torch.Tensor],
|
||||||
|
residuals: Sequence[torch.Tensor],
|
||||||
|
repeats: int,
|
||||||
|
replays: int,
|
||||||
|
) -> tuple[list[float], list[torch.Tensor]]:
|
||||||
|
outputs = [torch.empty_like(residual) for residual in residuals]
|
||||||
|
for weight, residual, output in zip(weights, residuals, outputs):
|
||||||
|
launch(activation, weight, residual, output)
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
|
||||||
|
graph = torch.cuda.CUDAGraph()
|
||||||
|
with torch.cuda.graph(graph):
|
||||||
|
for weight, residual, output in zip(weights, residuals, outputs):
|
||||||
|
launch(activation, weight, residual, output)
|
||||||
|
for _ in range(20):
|
||||||
|
graph.replay()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
|
||||||
|
samples = []
|
||||||
|
for _ in range(repeats):
|
||||||
|
start = torch.cuda.Event(enable_timing=True)
|
||||||
|
end = torch.cuda.Event(enable_timing=True)
|
||||||
|
start.record()
|
||||||
|
for _ in range(replays):
|
||||||
|
graph.replay()
|
||||||
|
end.record()
|
||||||
|
end.synchronize()
|
||||||
|
samples.append(start.elapsed_time(end) * 1000.0 / (replays * len(weights)))
|
||||||
|
return samples, outputs
|
||||||
|
|
||||||
|
|
||||||
|
def summarize(samples: Sequence[float]) -> dict[str, Any]:
|
||||||
|
ordered = sorted(samples)
|
||||||
|
|
||||||
|
def percentile(fraction: float) -> float:
|
||||||
|
position = fraction * (len(ordered) - 1)
|
||||||
|
lower = math.floor(position)
|
||||||
|
upper = math.ceil(position)
|
||||||
|
if lower == upper:
|
||||||
|
return ordered[lower]
|
||||||
|
weight = position - lower
|
||||||
|
return ordered[lower] * (1.0 - weight) + ordered[upper] * weight
|
||||||
|
|
||||||
|
mean = statistics.mean(samples)
|
||||||
|
return {
|
||||||
|
"median_us": statistics.median(samples),
|
||||||
|
"p10_us": percentile(0.1),
|
||||||
|
"p90_us": percentile(0.9),
|
||||||
|
"mean_us": mean,
|
||||||
|
"cv_pct": statistics.pstdev(samples) / mean * 100.0,
|
||||||
|
"samples_us": list(samples),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def correctness(
|
||||||
|
output: torch.Tensor,
|
||||||
|
activation: torch.Tensor,
|
||||||
|
weight: torch.Tensor,
|
||||||
|
residual: torch.Tensor,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
actual = output.float()
|
||||||
|
reference = activation.float() @ weight.float().t() + residual.float()
|
||||||
|
error = (actual - reference).abs()
|
||||||
|
scaled_error = error / (reference.abs() + 1.0)
|
||||||
|
cosine = torch.nn.functional.cosine_similarity(
|
||||||
|
actual.flatten(), reference.flatten(), dim=0
|
||||||
|
).item()
|
||||||
|
return {
|
||||||
|
"valid": cosine > 0.999,
|
||||||
|
"cosine": cosine,
|
||||||
|
"max_abs_error": error.max().item(),
|
||||||
|
"max_scaled_error": scaled_error.max().item(),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--kernel", type=Path, required=True)
|
||||||
|
parser.add_argument("--output", type=Path, required=True)
|
||||||
|
parser.add_argument(
|
||||||
|
"--mode", choices=("baseline", "sweep", "selected"), default="baseline"
|
||||||
|
)
|
||||||
|
parser.add_argument("--config", type=parse_config)
|
||||||
|
parser.add_argument("--m", type=int, action="append")
|
||||||
|
parser.add_argument("--config-shard", type=int, default=0)
|
||||||
|
parser.add_argument("--num-config-shards", type=int, default=1)
|
||||||
|
parser.add_argument("--repeats", type=int, default=21)
|
||||||
|
parser.add_argument("--replays", type=int, default=200)
|
||||||
|
parser.add_argument("--cache-multiplier", type=float, default=3.0)
|
||||||
|
parser.add_argument("--max-buffers", type=int, default=32)
|
||||||
|
parser.add_argument("--max-registers", type=int, default=64)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
token_counts = args.m or list(range(1, 17))
|
||||||
|
if any(not 1 <= m <= 16 for m in token_counts):
|
||||||
|
raise ValueError("expected 1 <= M <= 16")
|
||||||
|
if not 0 <= args.config_shard < args.num_config_shards:
|
||||||
|
raise ValueError("config shard must be in [0, num_config_shards)")
|
||||||
|
torch.accelerator.set_device_index(0)
|
||||||
|
if torch.cuda.get_device_capability() != (10, 3):
|
||||||
|
raise RuntimeError("this benchmark requires SM103")
|
||||||
|
|
||||||
|
kernel_class = load_kernel_class(args.kernel)
|
||||||
|
properties = torch.cuda.get_device_properties(0)
|
||||||
|
metadata = {
|
||||||
|
"device": properties.name,
|
||||||
|
"compute_capability": list(torch.cuda.get_device_capability()),
|
||||||
|
"torch_version": torch.__version__,
|
||||||
|
"cuda_version": torch.version.cuda,
|
||||||
|
}
|
||||||
|
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
with args.output.open("w", encoding="utf-8") as output_file:
|
||||||
|
for m in token_counts:
|
||||||
|
configs = candidate_configs(args.mode, args.config, m)
|
||||||
|
torch.manual_seed(20260722 + m)
|
||||||
|
count = rotating_buffer_count(m, args.cache_multiplier, args.max_buffers)
|
||||||
|
activation = torch.randn((m, K), device="cuda", dtype=torch.bfloat16)
|
||||||
|
weights = [
|
||||||
|
torch.randn((N, K), device="cuda", dtype=torch.bfloat16)
|
||||||
|
for _ in range(count)
|
||||||
|
]
|
||||||
|
residuals = [
|
||||||
|
torch.randn((m, N), device="cuda", dtype=torch.bfloat16)
|
||||||
|
for _ in range(count)
|
||||||
|
]
|
||||||
|
candidates: list[tuple[str, Config | None]] = [("cublas_addmm", None)]
|
||||||
|
candidates.extend(
|
||||||
|
("cute_residual", config)
|
||||||
|
for index, config in enumerate(configs)
|
||||||
|
if index % args.num_config_shards == args.config_shard
|
||||||
|
)
|
||||||
|
for backend, config in candidates:
|
||||||
|
row: dict[str, Any] = {
|
||||||
|
"m": m,
|
||||||
|
"n": N,
|
||||||
|
"k": K,
|
||||||
|
"backend": backend,
|
||||||
|
"mode": args.mode,
|
||||||
|
"config": dataclasses.asdict(config) if config else {},
|
||||||
|
"num_buffers": count,
|
||||||
|
"cache_multiplier": args.cache_multiplier,
|
||||||
|
**metadata,
|
||||||
|
}
|
||||||
|
try:
|
||||||
|
if backend == "cublas_addmm":
|
||||||
|
launch = lambda a, b, residual, c: torch.addmm(
|
||||||
|
residual, a, b.t(), out=c
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
if config is None:
|
||||||
|
raise AssertionError("missing CuTe config")
|
||||||
|
compiled = compile_kernel(
|
||||||
|
kernel_class, m, config, args.max_registers
|
||||||
|
)
|
||||||
|
launch = lambda a, b, residual, c, fn=compiled: fn(
|
||||||
|
a, b, residual, c, stream()
|
||||||
|
)
|
||||||
|
row.update(resource_usage(compiled))
|
||||||
|
samples, outputs = graph_samples(
|
||||||
|
launch,
|
||||||
|
activation,
|
||||||
|
weights,
|
||||||
|
residuals,
|
||||||
|
args.repeats,
|
||||||
|
args.replays,
|
||||||
|
)
|
||||||
|
row.update(
|
||||||
|
correctness(outputs[0], activation, weights[0], residuals[0])
|
||||||
|
)
|
||||||
|
row.update(summarize(samples))
|
||||||
|
except Exception as error: # noqa: BLE001
|
||||||
|
row.update(
|
||||||
|
{
|
||||||
|
"valid": False,
|
||||||
|
"error": f"{type(error).__name__}: {error}",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
output_file.write(json.dumps(row, sort_keys=True) + "\n")
|
||||||
|
output_file.flush()
|
||||||
|
print(json.dumps(row, sort_keys=True), flush=True)
|
||||||
|
|
||||||
|
del activation, weights, residuals
|
||||||
|
torch.accelerator.empty_cache()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,806 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
"""Benchmark the Kimi K3 latent-MoE tail and its up-projection kernels.
|
||||||
|
|
||||||
|
The ``up-projection`` subcommand isolates the TP-local dynamic and static-M
|
||||||
|
skinny GEMMs. It rotates weights through a working set larger than L2 to model
|
||||||
|
successive model layers.
|
||||||
|
|
||||||
|
The ``whole-tail`` subcommand measures the distributed operator. Its reference
|
||||||
|
path includes two AllReduces, RMSNorm, the replicated up-projection, and the
|
||||||
|
final add. CUDA-event samples report the slowest rank so cross-rank skew is
|
||||||
|
included.
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
|
||||||
|
.. code-block:: console
|
||||||
|
|
||||||
|
.venv/bin/python \
|
||||||
|
benchmarks/kernels/benchmark_kimi_k3_latent_moe_tail.py up-projection
|
||||||
|
|
||||||
|
torchrun --nproc-per-node=8 \
|
||||||
|
benchmarks/kernels/benchmark_kimi_k3_latent_moe_tail.py whole-tail
|
||||||
|
|
||||||
|
For multi-node runs, launch one ``torchrun`` agent per node and use a shared
|
||||||
|
rendezvous endpoint.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
import os
|
||||||
|
import statistics
|
||||||
|
from collections.abc import Callable, Sequence
|
||||||
|
from dataclasses import asdict
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import cutlass
|
||||||
|
import cutlass.utils as utils
|
||||||
|
import torch
|
||||||
|
import torch.distributed as dist
|
||||||
|
import torch.nn.functional as F
|
||||||
|
from cuda.bindings import driver as cuda
|
||||||
|
|
||||||
|
from vllm.distributed import get_tp_group
|
||||||
|
from vllm.distributed.parallel_state import (
|
||||||
|
init_distributed_environment,
|
||||||
|
initialize_model_parallel,
|
||||||
|
set_custom_all_reduce,
|
||||||
|
)
|
||||||
|
from vllm.model_executor.warmup.cutedsl_warmup import cutedsl_warmup
|
||||||
|
from vllm.models.kimi_k3.nvidia.ops import latent_moe_tail
|
||||||
|
from vllm.models.kimi_k3.nvidia.ops.cute_dsl.latent_moe_tail import (
|
||||||
|
fused_add_multicast_gemm,
|
||||||
|
fused_add_multicast_skinny_gemm,
|
||||||
|
)
|
||||||
|
|
||||||
|
HIDDEN_SIZE = 7168
|
||||||
|
LATENT_SIZE = 3584
|
||||||
|
RMS_EPS = 0.1
|
||||||
|
MAX_NUM_TOKENS = 16
|
||||||
|
MMA_TILER_MN = (64, 32)
|
||||||
|
CLUSTER_SHAPE_MN = (1, 8)
|
||||||
|
B_PRIME_STAGES = 2
|
||||||
|
|
||||||
|
|
||||||
|
def parse_up_projection_config(
|
||||||
|
value: str,
|
||||||
|
) -> fused_add_multicast_skinny_gemm.SkinnyConfig:
|
||||||
|
try:
|
||||||
|
values = [int(part) for part in value.split(",")]
|
||||||
|
except ValueError as error:
|
||||||
|
raise argparse.ArgumentTypeError(
|
||||||
|
"config must be BLOCK,OUTPUTS,K_UNROLL[,VECTOR_WIDTH[,PREFETCH_B]]"
|
||||||
|
) from error
|
||||||
|
if len(values) in (3, 4):
|
||||||
|
return fused_add_multicast_skinny_gemm.SkinnyConfig(*values)
|
||||||
|
if len(values) == 5 and values[4] in (0, 1):
|
||||||
|
return fused_add_multicast_skinny_gemm.SkinnyConfig(
|
||||||
|
*values[:4],
|
||||||
|
prefetch_b_before_pdl=bool(values[4]),
|
||||||
|
)
|
||||||
|
raise argparse.ArgumentTypeError(
|
||||||
|
"config must be BLOCK,OUTPUTS,K_UNROLL"
|
||||||
|
"[,VECTOR_WIDTH[,PREFETCH_B]], where PREFETCH_B is 0 or 1"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_tail_skinny_config(
|
||||||
|
value: str,
|
||||||
|
) -> tuple[int, fused_add_multicast_skinny_gemm.SkinnyConfig]:
|
||||||
|
try:
|
||||||
|
values = [int(part) for part in value.split(",")]
|
||||||
|
except ValueError as error:
|
||||||
|
raise argparse.ArgumentTypeError(
|
||||||
|
"config must be M,BLOCK,OUTPUTS,K_UNROLL[,VECTOR_WIDTH[,PREFETCH_B]]"
|
||||||
|
) from error
|
||||||
|
if len(values) == 4:
|
||||||
|
num_tokens, *config = values
|
||||||
|
return num_tokens, fused_add_multicast_skinny_gemm.SkinnyConfig(*config)
|
||||||
|
if len(values) == 5:
|
||||||
|
num_tokens, *config = values
|
||||||
|
return num_tokens, fused_add_multicast_skinny_gemm.SkinnyConfig(*config)
|
||||||
|
if len(values) == 6 and values[5] in (0, 1):
|
||||||
|
num_tokens, block, outputs, unroll, vector_width, prefetch = values
|
||||||
|
return num_tokens, fused_add_multicast_skinny_gemm.SkinnyConfig(
|
||||||
|
block,
|
||||||
|
outputs,
|
||||||
|
unroll,
|
||||||
|
vector_width,
|
||||||
|
bool(prefetch),
|
||||||
|
)
|
||||||
|
raise argparse.ArgumentTypeError(
|
||||||
|
"config must be M,BLOCK,OUTPUTS,K_UNROLL"
|
||||||
|
"[,VECTOR_WIDTH[,PREFETCH_B]], where PREFETCH_B is 0 or 1"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_args() -> argparse.Namespace:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
subparsers = parser.add_subparsers(dest="scope", required=True)
|
||||||
|
|
||||||
|
up_projection = subparsers.add_parser(
|
||||||
|
"up-projection",
|
||||||
|
help="Benchmark the isolated TP-local up-projection kernels.",
|
||||||
|
)
|
||||||
|
up_projection.add_argument(
|
||||||
|
"--backend",
|
||||||
|
choices=("dynamic", "skinny", "both"),
|
||||||
|
default="both",
|
||||||
|
)
|
||||||
|
up_projection.add_argument("--tp-size", type=int, default=16)
|
||||||
|
up_projection.add_argument(
|
||||||
|
"--num-tokens",
|
||||||
|
type=int,
|
||||||
|
nargs="+",
|
||||||
|
default=[*range(1, 9), 16],
|
||||||
|
)
|
||||||
|
up_projection.add_argument(
|
||||||
|
"--skinny-config",
|
||||||
|
type=parse_up_projection_config,
|
||||||
|
action="append",
|
||||||
|
help="Benchmark a static-M config for every selected token count.",
|
||||||
|
)
|
||||||
|
up_projection.add_argument("--cache-multiplier", type=float, default=2.0)
|
||||||
|
up_projection.add_argument("--max-weights", type=int, default=64)
|
||||||
|
up_projection.add_argument("--warmup-replays", type=int, default=10)
|
||||||
|
up_projection.add_argument("--samples", type=int, default=31)
|
||||||
|
up_projection.add_argument("--output", type=Path)
|
||||||
|
|
||||||
|
whole_tail = subparsers.add_parser(
|
||||||
|
"whole-tail",
|
||||||
|
help="Benchmark the distributed latent-MoE tail operator.",
|
||||||
|
)
|
||||||
|
whole_tail.add_argument(
|
||||||
|
"--backend",
|
||||||
|
choices=("reference", "fused", "both"),
|
||||||
|
default="both",
|
||||||
|
)
|
||||||
|
whole_tail.add_argument(
|
||||||
|
"--num-tokens",
|
||||||
|
type=int,
|
||||||
|
nargs="+",
|
||||||
|
default=[1, 5, 8, 16],
|
||||||
|
)
|
||||||
|
whole_tail.add_argument("--warmup-replays", type=int, default=20)
|
||||||
|
whole_tail.add_argument("--samples", type=int, default=51)
|
||||||
|
whole_tail.add_argument(
|
||||||
|
"--skinny-max-num-tokens",
|
||||||
|
type=int,
|
||||||
|
nargs="+",
|
||||||
|
help="Override the fused operator's static-M cutoff; use 0 for dynamic-only.",
|
||||||
|
)
|
||||||
|
whole_tail.add_argument(
|
||||||
|
"--skinny-config",
|
||||||
|
type=parse_tail_skinny_config,
|
||||||
|
action="append",
|
||||||
|
help="Override one static-M config for tuning.",
|
||||||
|
)
|
||||||
|
whole_tail.add_argument("--output", type=Path)
|
||||||
|
return parser.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def percentile(samples: Sequence[float], fraction: float) -> float:
|
||||||
|
ordered = sorted(samples)
|
||||||
|
position = fraction * (len(ordered) - 1)
|
||||||
|
lower = math.floor(position)
|
||||||
|
upper = math.ceil(position)
|
||||||
|
if lower == upper:
|
||||||
|
return ordered[lower]
|
||||||
|
upper_weight = position - lower
|
||||||
|
return ordered[lower] * (1.0 - upper_weight) + ordered[upper] * upper_weight
|
||||||
|
|
||||||
|
|
||||||
|
def summarize(samples_us: Sequence[float]) -> dict[str, Any]:
|
||||||
|
mean_us = statistics.mean(samples_us)
|
||||||
|
return {
|
||||||
|
"median_us": statistics.median(samples_us),
|
||||||
|
"p10_us": percentile(samples_us, 0.1),
|
||||||
|
"p90_us": percentile(samples_us, 0.9),
|
||||||
|
"mean_us": mean_us,
|
||||||
|
"cv_pct": statistics.pstdev(samples_us) / mean_us * 100.0,
|
||||||
|
"samples_us": list(samples_us),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def rotating_weight_count(
|
||||||
|
shard_size: int,
|
||||||
|
cache_multiplier: float,
|
||||||
|
limit: int,
|
||||||
|
) -> int:
|
||||||
|
properties = torch.cuda.get_device_properties(
|
||||||
|
torch.accelerator.current_device_index()
|
||||||
|
)
|
||||||
|
weight_bytes = shard_size * LATENT_SIZE * 2
|
||||||
|
target_bytes = math.ceil(properties.L2_cache_size * cache_multiplier)
|
||||||
|
return max(2, min(limit, math.ceil(target_bytes / weight_bytes)))
|
||||||
|
|
||||||
|
|
||||||
|
def capture_up_projection_graph(
|
||||||
|
launches: Sequence[Callable[[], None]],
|
||||||
|
) -> torch.cuda.CUDAGraph:
|
||||||
|
for launch in launches:
|
||||||
|
launch()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
|
||||||
|
graph = torch.cuda.CUDAGraph()
|
||||||
|
with torch.cuda.graph(graph):
|
||||||
|
for launch in launches:
|
||||||
|
launch()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
return graph
|
||||||
|
|
||||||
|
|
||||||
|
def benchmark_up_projection_graph(
|
||||||
|
graph: torch.cuda.CUDAGraph,
|
||||||
|
*,
|
||||||
|
operations_per_replay: int,
|
||||||
|
warmup_replays: int,
|
||||||
|
samples: int,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
for _ in range(warmup_replays):
|
||||||
|
graph.replay()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
|
||||||
|
samples_us = []
|
||||||
|
start = torch.cuda.Event(enable_timing=True)
|
||||||
|
end = torch.cuda.Event(enable_timing=True)
|
||||||
|
for _ in range(samples):
|
||||||
|
start.record()
|
||||||
|
graph.replay()
|
||||||
|
end.record()
|
||||||
|
end.synchronize()
|
||||||
|
samples_us.append(start.elapsed_time(end) * 1000.0 / operations_per_replay)
|
||||||
|
return summarize(samples_us)
|
||||||
|
|
||||||
|
|
||||||
|
class DynamicKernel:
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
shard_size: int,
|
||||||
|
mailbox: torch.Tensor,
|
||||||
|
shared_shard: torch.Tensor,
|
||||||
|
) -> None:
|
||||||
|
self.shard_size = shard_size
|
||||||
|
self.mailbox = mailbox
|
||||||
|
self.mailbox_c = fused_add_multicast_gemm._as_cute(mailbox)
|
||||||
|
compile_latent = torch.empty(
|
||||||
|
(1, MAX_NUM_TOKENS, LATENT_SIZE),
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=mailbox.device,
|
||||||
|
)
|
||||||
|
compile_weight = torch.empty(
|
||||||
|
(1, shard_size, LATENT_SIZE),
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=mailbox.device,
|
||||||
|
)
|
||||||
|
cluster_size = math.prod(CLUSTER_SHAPE_MN)
|
||||||
|
max_active_clusters = utils.HardwareInfo().get_max_active_clusters(cluster_size)
|
||||||
|
self.compiled = fused_add_multicast_gemm.compile_kernel(
|
||||||
|
(MAX_NUM_TOKENS, shard_size, LATENT_SIZE, 1),
|
||||||
|
fused_add_multicast_gemm._as_cute(
|
||||||
|
compile_latent,
|
||||||
|
dynamic_m=True,
|
||||||
|
),
|
||||||
|
fused_add_multicast_gemm._as_cute(compile_weight),
|
||||||
|
self.mailbox_c,
|
||||||
|
fused_add_multicast_gemm._as_cute(shared_shard),
|
||||||
|
HIDDEN_SIZE,
|
||||||
|
shard_size,
|
||||||
|
MMA_TILER_MN,
|
||||||
|
CLUSTER_SHAPE_MN,
|
||||||
|
max_active_clusters,
|
||||||
|
B_PRIME_STAGES,
|
||||||
|
)
|
||||||
|
|
||||||
|
def launch(
|
||||||
|
self,
|
||||||
|
latent: torch.Tensor,
|
||||||
|
weight: torch.Tensor,
|
||||||
|
shared_shard: torch.Tensor,
|
||||||
|
) -> None:
|
||||||
|
stream = cuda.CUstream(torch.cuda.current_stream().cuda_stream)
|
||||||
|
self.compiled(
|
||||||
|
fused_add_multicast_gemm._as_cute(
|
||||||
|
latent.unsqueeze(0),
|
||||||
|
dynamic_m=True,
|
||||||
|
),
|
||||||
|
fused_add_multicast_gemm._as_cute(weight.unsqueeze(0)),
|
||||||
|
self.mailbox_c,
|
||||||
|
fused_add_multicast_gemm._as_cute(shared_shard),
|
||||||
|
cutlass.Int64(latent.shape[0]),
|
||||||
|
cutlass.Int64(self.mailbox.data_ptr()),
|
||||||
|
stream,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class SkinnyKernel:
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
num_tokens: int,
|
||||||
|
shard_size: int,
|
||||||
|
config: fused_add_multicast_skinny_gemm.SkinnyConfig,
|
||||||
|
) -> None:
|
||||||
|
self.compiled = fused_add_multicast_skinny_gemm.compile_kernel(
|
||||||
|
num_rows=num_tokens,
|
||||||
|
latent_dim=LATENT_SIZE,
|
||||||
|
hidden_dim=HIDDEN_SIZE,
|
||||||
|
shard_dim=shard_size,
|
||||||
|
config=config,
|
||||||
|
)
|
||||||
|
|
||||||
|
def launch(
|
||||||
|
self,
|
||||||
|
latent: torch.Tensor,
|
||||||
|
weight: torch.Tensor,
|
||||||
|
shared_shard: torch.Tensor,
|
||||||
|
mailbox: torch.Tensor,
|
||||||
|
) -> None:
|
||||||
|
self.compiled(
|
||||||
|
fused_add_multicast_skinny_gemm._as_cute(latent),
|
||||||
|
fused_add_multicast_skinny_gemm._as_cute(weight),
|
||||||
|
fused_add_multicast_skinny_gemm._as_cute(shared_shard),
|
||||||
|
cutlass.Int64(mailbox.data_ptr()),
|
||||||
|
cuda.CUstream(torch.cuda.current_stream().cuda_stream),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def check_up_projection_output(
|
||||||
|
actual: torch.Tensor,
|
||||||
|
latent: torch.Tensor,
|
||||||
|
weight: torch.Tensor,
|
||||||
|
shared_shard: torch.Tensor,
|
||||||
|
) -> None:
|
||||||
|
gemm = F.linear(latent.float(), weight.float()).to(torch.bfloat16)
|
||||||
|
expected = (gemm.float() + shared_shard.float()).to(torch.bfloat16)
|
||||||
|
torch.testing.assert_close(actual, expected, atol=8e-2, rtol=3e-2)
|
||||||
|
|
||||||
|
|
||||||
|
def make_up_projection_launches(
|
||||||
|
launch: Callable[[torch.Tensor, torch.Tensor, torch.Tensor], None],
|
||||||
|
latent: torch.Tensor,
|
||||||
|
weights: Sequence[torch.Tensor],
|
||||||
|
shared_shard: torch.Tensor,
|
||||||
|
) -> list[Callable[[], None]]:
|
||||||
|
return [
|
||||||
|
lambda weight=weight: launch(latent, weight, shared_shard) for weight in weights
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def benchmark_up_projection(args: argparse.Namespace) -> None:
|
||||||
|
if args.tp_size <= 0 or HIDDEN_SIZE % args.tp_size:
|
||||||
|
raise ValueError("TP size must be positive and divide the hidden size")
|
||||||
|
if any(not 1 <= num_tokens <= MAX_NUM_TOKENS for num_tokens in args.num_tokens):
|
||||||
|
raise ValueError("--num-tokens values must be in [1, 16]")
|
||||||
|
if args.cache_multiplier <= 0 or args.max_weights <= 0:
|
||||||
|
raise ValueError("cache multiplier and max weights must be positive")
|
||||||
|
if args.warmup_replays < 0 or args.samples <= 0:
|
||||||
|
raise ValueError("warmup replays must be nonnegative and samples positive")
|
||||||
|
|
||||||
|
torch.accelerator.set_device_index(0)
|
||||||
|
device = torch.device("cuda", 0)
|
||||||
|
if torch.cuda.get_device_capability(device)[0] != 10:
|
||||||
|
raise RuntimeError("Kimi K3 latent-MoE tail requires SM100")
|
||||||
|
|
||||||
|
shard_size = HIDDEN_SIZE // args.tp_size
|
||||||
|
weight_count = rotating_weight_count(
|
||||||
|
shard_size,
|
||||||
|
args.cache_multiplier,
|
||||||
|
args.max_weights,
|
||||||
|
)
|
||||||
|
torch.manual_seed(20260726)
|
||||||
|
weights = [
|
||||||
|
torch.randn(
|
||||||
|
(shard_size, LATENT_SIZE),
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=device,
|
||||||
|
)
|
||||||
|
/ LATENT_SIZE**0.5
|
||||||
|
for _ in range(weight_count)
|
||||||
|
]
|
||||||
|
mailbox = torch.empty(
|
||||||
|
(1, MAX_NUM_TOKENS, HIDDEN_SIZE),
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=device,
|
||||||
|
)
|
||||||
|
shared = torch.randn(
|
||||||
|
(MAX_NUM_TOKENS, HIDDEN_SIZE),
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=device,
|
||||||
|
)
|
||||||
|
shared_shard = shared[:, :shard_size]
|
||||||
|
use_dynamic = args.backend in ("dynamic", "both")
|
||||||
|
use_skinny = args.backend in ("skinny", "both")
|
||||||
|
dynamic_kernel = (
|
||||||
|
DynamicKernel(shard_size, mailbox, shared_shard) if use_dynamic else None
|
||||||
|
)
|
||||||
|
|
||||||
|
results = []
|
||||||
|
for num_tokens in args.num_tokens:
|
||||||
|
latent = torch.randn(
|
||||||
|
(num_tokens, LATENT_SIZE),
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=device,
|
||||||
|
)
|
||||||
|
result: dict[str, Any] = {"num_tokens": num_tokens}
|
||||||
|
if dynamic_kernel is not None:
|
||||||
|
launches = make_up_projection_launches(
|
||||||
|
dynamic_kernel.launch,
|
||||||
|
latent,
|
||||||
|
weights,
|
||||||
|
shared_shard,
|
||||||
|
)
|
||||||
|
graph = capture_up_projection_graph(launches)
|
||||||
|
result["dynamic"] = benchmark_up_projection_graph(
|
||||||
|
graph,
|
||||||
|
operations_per_replay=len(launches),
|
||||||
|
warmup_replays=args.warmup_replays,
|
||||||
|
samples=args.samples,
|
||||||
|
)
|
||||||
|
check_up_projection_output(
|
||||||
|
mailbox[0, :num_tokens, :shard_size],
|
||||||
|
latent,
|
||||||
|
weights[-1],
|
||||||
|
shared_shard[:num_tokens],
|
||||||
|
)
|
||||||
|
if use_skinny:
|
||||||
|
configs = args.skinny_config or [
|
||||||
|
fused_add_multicast_skinny_gemm.config_for_m(
|
||||||
|
num_tokens,
|
||||||
|
shard_size,
|
||||||
|
)
|
||||||
|
]
|
||||||
|
skinny_results = []
|
||||||
|
for config in configs:
|
||||||
|
skinny_kernel = SkinnyKernel(num_tokens, shard_size, config)
|
||||||
|
|
||||||
|
def launch_skinny(
|
||||||
|
latent: torch.Tensor,
|
||||||
|
weight: torch.Tensor,
|
||||||
|
shared_shard: torch.Tensor,
|
||||||
|
*,
|
||||||
|
skinny_kernel: SkinnyKernel = skinny_kernel,
|
||||||
|
num_tokens: int = num_tokens,
|
||||||
|
) -> None:
|
||||||
|
skinny_kernel.launch(
|
||||||
|
latent,
|
||||||
|
weight,
|
||||||
|
shared_shard[:num_tokens],
|
||||||
|
mailbox,
|
||||||
|
)
|
||||||
|
|
||||||
|
launches = make_up_projection_launches(
|
||||||
|
launch_skinny,
|
||||||
|
latent,
|
||||||
|
weights,
|
||||||
|
shared_shard,
|
||||||
|
)
|
||||||
|
graph = capture_up_projection_graph(launches)
|
||||||
|
timing = benchmark_up_projection_graph(
|
||||||
|
graph,
|
||||||
|
operations_per_replay=len(launches),
|
||||||
|
warmup_replays=args.warmup_replays,
|
||||||
|
samples=args.samples,
|
||||||
|
)
|
||||||
|
check_up_projection_output(
|
||||||
|
mailbox[0, :num_tokens, :shard_size],
|
||||||
|
latent,
|
||||||
|
weights[-1],
|
||||||
|
shared_shard[:num_tokens],
|
||||||
|
)
|
||||||
|
skinny_results.append(
|
||||||
|
{
|
||||||
|
"config": asdict(config),
|
||||||
|
**timing,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
result["skinny"] = skinny_results
|
||||||
|
results.append(result)
|
||||||
|
|
||||||
|
properties = torch.cuda.get_device_properties(device)
|
||||||
|
report = {
|
||||||
|
"scope": "up-projection",
|
||||||
|
"device": properties.name,
|
||||||
|
"compute_capability": list(torch.cuda.get_device_capability(device)),
|
||||||
|
"tp_size": args.tp_size,
|
||||||
|
"shard_size": shard_size,
|
||||||
|
"weight_count": weight_count,
|
||||||
|
"cache_multiplier": args.cache_multiplier,
|
||||||
|
"warmup_replays": args.warmup_replays,
|
||||||
|
"samples": args.samples,
|
||||||
|
"results": results,
|
||||||
|
}
|
||||||
|
rendered = json.dumps(report, indent=2)
|
||||||
|
print(rendered, flush=True)
|
||||||
|
if args.output is not None:
|
||||||
|
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
args.output.write_text(rendered + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def capture_tail_graph(
|
||||||
|
operation: Callable[[], torch.Tensor],
|
||||||
|
cpu_group: dist.ProcessGroup,
|
||||||
|
) -> tuple[torch.cuda.CUDAGraph, torch.Tensor]:
|
||||||
|
for _ in range(3):
|
||||||
|
dist.barrier(group=cpu_group)
|
||||||
|
output = operation()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
|
||||||
|
dist.barrier(group=cpu_group)
|
||||||
|
graph = torch.cuda.CUDAGraph()
|
||||||
|
with torch.cuda.graph(graph):
|
||||||
|
output = operation()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
return graph, output
|
||||||
|
|
||||||
|
|
||||||
|
def benchmark_tail_graph(
|
||||||
|
graph: torch.cuda.CUDAGraph,
|
||||||
|
*,
|
||||||
|
warmup_replays: int,
|
||||||
|
samples: int,
|
||||||
|
device_group: dist.ProcessGroup,
|
||||||
|
cpu_group: dist.ProcessGroup,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
for _ in range(warmup_replays):
|
||||||
|
graph.replay()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
|
||||||
|
dist.barrier(group=cpu_group)
|
||||||
|
starts = [torch.cuda.Event(enable_timing=True) for _ in range(samples + 1)]
|
||||||
|
ends = [torch.cuda.Event(enable_timing=True) for _ in range(samples + 1)]
|
||||||
|
for start, end in zip(starts, ends):
|
||||||
|
start.record()
|
||||||
|
graph.replay()
|
||||||
|
end.record()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
|
||||||
|
samples_us = torch.tensor(
|
||||||
|
[start.elapsed_time(end) * 1000.0 for start, end in zip(starts, ends)],
|
||||||
|
dtype=torch.float64,
|
||||||
|
device=torch.accelerator.current_device_index(),
|
||||||
|
)
|
||||||
|
dist.all_reduce(samples_us, op=dist.ReduceOp.MAX, group=device_group)
|
||||||
|
return summarize(samples_us[1:].tolist())
|
||||||
|
|
||||||
|
|
||||||
|
def make_inputs(
|
||||||
|
num_tokens: int,
|
||||||
|
rank: int,
|
||||||
|
device: torch.device,
|
||||||
|
) -> tuple[torch.Tensor, torch.Tensor]:
|
||||||
|
torch.manual_seed(20260726 + 100 * num_tokens + rank)
|
||||||
|
routed = torch.randn(
|
||||||
|
(num_tokens, LATENT_SIZE),
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=device,
|
||||||
|
).mul_(0.01)
|
||||||
|
shared = torch.randn(
|
||||||
|
(num_tokens, HIDDEN_SIZE),
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=device,
|
||||||
|
)
|
||||||
|
return routed, shared
|
||||||
|
|
||||||
|
|
||||||
|
def make_reference(
|
||||||
|
routed: torch.Tensor,
|
||||||
|
shared: torch.Tensor,
|
||||||
|
rms_weight: torch.Tensor,
|
||||||
|
up_weight: torch.Tensor,
|
||||||
|
device_group: dist.ProcessGroup,
|
||||||
|
) -> Callable[[], torch.Tensor]:
|
||||||
|
routed_workspace = torch.empty_like(routed)
|
||||||
|
shared_workspace = torch.empty_like(shared)
|
||||||
|
|
||||||
|
def reference() -> torch.Tensor:
|
||||||
|
routed_workspace.copy_(routed)
|
||||||
|
dist.all_reduce(routed_workspace, group=device_group)
|
||||||
|
normalized = F.rms_norm(
|
||||||
|
routed_workspace,
|
||||||
|
(LATENT_SIZE,),
|
||||||
|
rms_weight,
|
||||||
|
RMS_EPS,
|
||||||
|
)
|
||||||
|
projected = F.linear(normalized, up_weight)
|
||||||
|
shared_workspace.copy_(shared)
|
||||||
|
dist.all_reduce(shared_workspace, group=device_group)
|
||||||
|
return projected.add(shared_workspace)
|
||||||
|
|
||||||
|
return reference
|
||||||
|
|
||||||
|
|
||||||
|
def check_fused_output(
|
||||||
|
fused_output: torch.Tensor,
|
||||||
|
reference: Callable[[], torch.Tensor],
|
||||||
|
cpu_group: dist.ProcessGroup,
|
||||||
|
) -> None:
|
||||||
|
dist.barrier(group=cpu_group)
|
||||||
|
expected = reference()
|
||||||
|
torch.testing.assert_close(fused_output, expected, atol=8e-2, rtol=3e-2)
|
||||||
|
|
||||||
|
|
||||||
|
def benchmark_whole_tail(args: argparse.Namespace) -> None:
|
||||||
|
if any(not 1 <= num_tokens <= 16 for num_tokens in args.num_tokens):
|
||||||
|
raise ValueError("--num-tokens values must be in [1, 16]")
|
||||||
|
if args.warmup_replays < 0 or args.samples <= 0:
|
||||||
|
raise ValueError("warmup replays must be nonnegative and samples positive")
|
||||||
|
if args.skinny_max_num_tokens is not None and any(
|
||||||
|
not 0 <= cutoff <= 8 for cutoff in args.skinny_max_num_tokens
|
||||||
|
):
|
||||||
|
raise ValueError("--skinny-max-num-tokens must be in [0, 8]")
|
||||||
|
skinny_configs = dict(args.skinny_config or ())
|
||||||
|
if len(skinny_configs) != len(args.skinny_config or ()):
|
||||||
|
raise ValueError("--skinny-config must not repeat an M value")
|
||||||
|
if any(not 1 <= num_tokens <= 8 for num_tokens in skinny_configs):
|
||||||
|
raise ValueError("--skinny-config M values must be in [1, 8]")
|
||||||
|
if not {"RANK", "WORLD_SIZE", "LOCAL_RANK"} <= os.environ.keys():
|
||||||
|
raise RuntimeError("launch this benchmark with torchrun")
|
||||||
|
|
||||||
|
rank = int(os.environ["RANK"])
|
||||||
|
world_size = int(os.environ["WORLD_SIZE"])
|
||||||
|
local_rank = int(os.environ["LOCAL_RANK"])
|
||||||
|
device = torch.device("cuda", local_rank)
|
||||||
|
torch.accelerator.set_device_index(device)
|
||||||
|
init_distributed_environment()
|
||||||
|
if world_size > 8:
|
||||||
|
set_custom_all_reduce(False)
|
||||||
|
initialize_model_parallel(tensor_model_parallel_size=world_size)
|
||||||
|
device_group = get_tp_group().device_group
|
||||||
|
cpu_group = dist.new_group(backend="gloo")
|
||||||
|
|
||||||
|
if torch.cuda.get_device_capability(device)[0] != 10:
|
||||||
|
raise RuntimeError("Kimi K3 latent-MoE tail requires SM100")
|
||||||
|
|
||||||
|
torch.manual_seed(20260726)
|
||||||
|
rms_weight = 1 + 0.1 * torch.randn(
|
||||||
|
LATENT_SIZE,
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=device,
|
||||||
|
)
|
||||||
|
up_weight = (
|
||||||
|
torch.randn(
|
||||||
|
(HIDDEN_SIZE, LATENT_SIZE),
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=device,
|
||||||
|
)
|
||||||
|
/ LATENT_SIZE**0.5
|
||||||
|
)
|
||||||
|
|
||||||
|
use_reference = args.backend in ("reference", "both")
|
||||||
|
use_fused = args.backend in ("fused", "both")
|
||||||
|
fused_ops = []
|
||||||
|
if use_fused:
|
||||||
|
production_config_for_m = fused_add_multicast_skinny_gemm.config_for_m
|
||||||
|
|
||||||
|
def config_for_m(
|
||||||
|
num_rows: int,
|
||||||
|
shard_dim: int = 896,
|
||||||
|
) -> fused_add_multicast_skinny_gemm.SkinnyConfig:
|
||||||
|
config = skinny_configs.get(num_rows)
|
||||||
|
if config is not None:
|
||||||
|
return config
|
||||||
|
return production_config_for_m(num_rows, shard_dim)
|
||||||
|
|
||||||
|
fused_add_multicast_skinny_gemm.config_for_m = config_for_m
|
||||||
|
cutoffs = args.skinny_max_num_tokens or [latent_moe_tail._SKINNY_MAX_NUM_TOKENS]
|
||||||
|
for cutoff in cutoffs:
|
||||||
|
latent_moe_tail._SKINNY_MAX_NUM_TOKENS = cutoff
|
||||||
|
latent_moe_tail.KimiK3LatentMoETailOp._instances.clear()
|
||||||
|
fused_ops.append(
|
||||||
|
(
|
||||||
|
cutoff,
|
||||||
|
latent_moe_tail.KimiK3LatentMoETailOp.initialize(
|
||||||
|
hidden_size=HIDDEN_SIZE,
|
||||||
|
latent_size=LATENT_SIZE,
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=device,
|
||||||
|
rms_eps=RMS_EPS,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
cutedsl_warmup()
|
||||||
|
|
||||||
|
results = []
|
||||||
|
for num_tokens in args.num_tokens:
|
||||||
|
routed, shared = make_inputs(num_tokens, rank, device)
|
||||||
|
reference = make_reference(
|
||||||
|
routed,
|
||||||
|
shared,
|
||||||
|
rms_weight,
|
||||||
|
up_weight,
|
||||||
|
device_group,
|
||||||
|
)
|
||||||
|
result: dict[str, Any] = {"num_tokens": num_tokens}
|
||||||
|
if use_reference:
|
||||||
|
reference_graph, _ = capture_tail_graph(reference, cpu_group)
|
||||||
|
result["reference"] = benchmark_tail_graph(
|
||||||
|
reference_graph,
|
||||||
|
warmup_replays=args.warmup_replays,
|
||||||
|
samples=args.samples,
|
||||||
|
device_group=device_group,
|
||||||
|
cpu_group=cpu_group,
|
||||||
|
)
|
||||||
|
for cutoff, fused_op in fused_ops:
|
||||||
|
|
||||||
|
def fused(
|
||||||
|
routed: torch.Tensor = routed,
|
||||||
|
shared: torch.Tensor = shared,
|
||||||
|
fused_op: latent_moe_tail.KimiK3LatentMoETailOp = fused_op,
|
||||||
|
) -> torch.Tensor:
|
||||||
|
return fused_op(routed, shared, rms_weight, up_weight)
|
||||||
|
|
||||||
|
fused_graph, fused_output = capture_tail_graph(fused, cpu_group)
|
||||||
|
fused_key = "fused" if len(fused_ops) == 1 else f"fused_skinny_max_{cutoff}"
|
||||||
|
result[fused_key] = benchmark_tail_graph(
|
||||||
|
fused_graph,
|
||||||
|
warmup_replays=args.warmup_replays,
|
||||||
|
samples=args.samples,
|
||||||
|
device_group=device_group,
|
||||||
|
cpu_group=cpu_group,
|
||||||
|
)
|
||||||
|
check_fused_output(fused_output, reference, cpu_group)
|
||||||
|
if "reference" in result:
|
||||||
|
speedup = (
|
||||||
|
result["reference"]["median_us"] / result[fused_key]["median_us"]
|
||||||
|
)
|
||||||
|
if len(fused_ops) == 1:
|
||||||
|
result["speedup"] = speedup
|
||||||
|
else:
|
||||||
|
result[f"{fused_key}_speedup"] = speedup
|
||||||
|
results.append(result)
|
||||||
|
|
||||||
|
properties = torch.cuda.get_device_properties(device)
|
||||||
|
report = {
|
||||||
|
"scope": "whole-tail",
|
||||||
|
"device": properties.name,
|
||||||
|
"compute_capability": list(torch.cuda.get_device_capability(device)),
|
||||||
|
"world_size": world_size,
|
||||||
|
"torch_version": torch.__version__,
|
||||||
|
"cuda_version": torch.version.cuda,
|
||||||
|
"warmup_replays": args.warmup_replays,
|
||||||
|
"samples": args.samples,
|
||||||
|
"skinny_max_num_tokens": [cutoff for cutoff, _ in fused_ops],
|
||||||
|
"skinny_configs": {
|
||||||
|
str(num_tokens): asdict(config)
|
||||||
|
for num_tokens, config in skinny_configs.items()
|
||||||
|
},
|
||||||
|
"timing_scope": {
|
||||||
|
"reference": (
|
||||||
|
"two input copies, two AllReduces, RMSNorm, full replicated "
|
||||||
|
"up-projection GEMM, and final add"
|
||||||
|
),
|
||||||
|
"fused": (
|
||||||
|
"routed AllReduce/RMSNorm plus shared ReduceScatter, sharded "
|
||||||
|
"up-projection/multicast, and Lamport copy"
|
||||||
|
),
|
||||||
|
},
|
||||||
|
"results": results,
|
||||||
|
}
|
||||||
|
if rank == 0:
|
||||||
|
rendered = json.dumps(report, indent=2)
|
||||||
|
print(rendered, flush=True)
|
||||||
|
if args.output is not None:
|
||||||
|
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
args.output.write_text(rendered + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
dist.barrier(group=cpu_group)
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
args = parse_args()
|
||||||
|
if args.scope == "up-projection":
|
||||||
|
benchmark_up_projection(args)
|
||||||
|
return
|
||||||
|
|
||||||
|
from vllm.config import VllmConfig, set_current_vllm_config
|
||||||
|
|
||||||
|
with set_current_vllm_config(VllmConfig()):
|
||||||
|
benchmark_whole_tail(args)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,239 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import statistics
|
||||||
|
from collections.abc import Callable
|
||||||
|
|
||||||
|
import torch
|
||||||
|
import torch.distributed as dist
|
||||||
|
|
||||||
|
import vllm._custom_ops as ops
|
||||||
|
from vllm.distributed.device_communicators.custom_all_reduce import CustomAllreduce
|
||||||
|
|
||||||
|
|
||||||
|
def parse_args() -> argparse.Namespace:
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--tokens", type=int, nargs="+", default=[8, 32, 128, 1024])
|
||||||
|
parser.add_argument("--hidden-size", type=int, default=7168)
|
||||||
|
parser.add_argument("--graph-repeats", type=int, default=20)
|
||||||
|
parser.add_argument("--warmup-replays", type=int, default=5)
|
||||||
|
parser.add_argument("--samples", type=int, default=15)
|
||||||
|
return parser.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def capture_graph(op: Callable[[], None], repeats: int) -> torch.cuda.CUDAGraph:
|
||||||
|
stream = torch.cuda.Stream()
|
||||||
|
stream.wait_stream(torch.cuda.current_stream())
|
||||||
|
with torch.cuda.stream(stream):
|
||||||
|
for _ in range(3):
|
||||||
|
op()
|
||||||
|
stream.synchronize()
|
||||||
|
|
||||||
|
graph = torch.cuda.CUDAGraph()
|
||||||
|
with torch.cuda.graph(graph, stream=stream):
|
||||||
|
for _ in range(repeats):
|
||||||
|
op()
|
||||||
|
torch.cuda.current_stream().wait_stream(stream)
|
||||||
|
return graph
|
||||||
|
|
||||||
|
|
||||||
|
def max_rank_graph_time(
|
||||||
|
graph: torch.cuda.CUDAGraph,
|
||||||
|
repeats: int,
|
||||||
|
warmup_replays: int,
|
||||||
|
samples: int,
|
||||||
|
device_group: dist.ProcessGroup,
|
||||||
|
cpu_group: dist.ProcessGroup,
|
||||||
|
) -> float:
|
||||||
|
for _ in range(warmup_replays):
|
||||||
|
graph.replay()
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
|
||||||
|
timings = []
|
||||||
|
start = torch.cuda.Event(enable_timing=True)
|
||||||
|
end = torch.cuda.Event(enable_timing=True)
|
||||||
|
for _ in range(samples):
|
||||||
|
dist.barrier(group=cpu_group)
|
||||||
|
start.record()
|
||||||
|
graph.replay()
|
||||||
|
end.record()
|
||||||
|
end.synchronize()
|
||||||
|
elapsed = torch.tensor(
|
||||||
|
start.elapsed_time(end) / repeats,
|
||||||
|
dtype=torch.float64,
|
||||||
|
device=torch.accelerator.current_device_index(),
|
||||||
|
)
|
||||||
|
dist.all_reduce(elapsed, op=dist.ReduceOp.MAX, group=device_group)
|
||||||
|
timings.append(elapsed.item())
|
||||||
|
return statistics.median(timings)
|
||||||
|
|
||||||
|
|
||||||
|
def check_outputs(
|
||||||
|
comm: CustomAllreduce,
|
||||||
|
local: torch.Tensor,
|
||||||
|
reduce_input: torch.Tensor,
|
||||||
|
device_group: dist.ProcessGroup,
|
||||||
|
) -> None:
|
||||||
|
expected_gather = torch.empty(
|
||||||
|
(local.shape[0] * dist.get_world_size(), local.shape[1]),
|
||||||
|
dtype=local.dtype,
|
||||||
|
device=local.device,
|
||||||
|
)
|
||||||
|
dist.all_gather_into_tensor(expected_gather, local, group=device_group)
|
||||||
|
gathered = comm.custom_all_gather(local)
|
||||||
|
assert gathered is not None
|
||||||
|
torch.testing.assert_close(gathered, expected_gather)
|
||||||
|
|
||||||
|
expected_scatter = torch.empty_like(local)
|
||||||
|
dist.reduce_scatter_tensor(
|
||||||
|
expected_scatter,
|
||||||
|
reduce_input.clone(),
|
||||||
|
group=device_group,
|
||||||
|
)
|
||||||
|
scattered = comm.custom_reduce_scatter(reduce_input)
|
||||||
|
assert scattered is not None
|
||||||
|
torch.testing.assert_close(scattered, expected_scatter)
|
||||||
|
|
||||||
|
|
||||||
|
def benchmark_shape(
|
||||||
|
comm: CustomAllreduce,
|
||||||
|
global_tokens: int,
|
||||||
|
hidden_size: int,
|
||||||
|
graph_repeats: int,
|
||||||
|
warmup_replays: int,
|
||||||
|
samples: int,
|
||||||
|
device_group: dist.ProcessGroup,
|
||||||
|
cpu_group: dist.ProcessGroup,
|
||||||
|
) -> dict[str, float | int]:
|
||||||
|
world_size = dist.get_world_size()
|
||||||
|
rank = dist.get_rank()
|
||||||
|
padded_tokens = (global_tokens + world_size - 1) // world_size * world_size
|
||||||
|
local_tokens = padded_tokens // world_size
|
||||||
|
local = torch.full(
|
||||||
|
(local_tokens, hidden_size),
|
||||||
|
rank + 1,
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=torch.accelerator.current_device_index(),
|
||||||
|
)
|
||||||
|
reduce_input = torch.full(
|
||||||
|
(padded_tokens, hidden_size),
|
||||||
|
rank + 1,
|
||||||
|
dtype=torch.bfloat16,
|
||||||
|
device=local.device,
|
||||||
|
)
|
||||||
|
check_outputs(comm, local, reduce_input, device_group)
|
||||||
|
|
||||||
|
custom_gather_out = torch.empty(
|
||||||
|
(padded_tokens, hidden_size),
|
||||||
|
dtype=local.dtype,
|
||||||
|
device=local.device,
|
||||||
|
)
|
||||||
|
custom_scatter_out = torch.empty_like(local)
|
||||||
|
nccl_gather_out = torch.empty_like(custom_gather_out)
|
||||||
|
nccl_scatter_out = torch.empty_like(local)
|
||||||
|
|
||||||
|
def custom_ag() -> None:
|
||||||
|
ops.mnnvl_lamport_all_gather(
|
||||||
|
comm._ptr,
|
||||||
|
local,
|
||||||
|
custom_gather_out,
|
||||||
|
comm.mnnvl_lamport_ag_local_ptr,
|
||||||
|
comm.mnnvl_lamport_ag_multicast_ptr,
|
||||||
|
comm.mnnvl_lamport_ag_epoch_ptr,
|
||||||
|
comm.mnnvl_buffer_size,
|
||||||
|
)
|
||||||
|
|
||||||
|
def custom_rs() -> None:
|
||||||
|
ops.mnnvl_lamport_reduce_scatter(
|
||||||
|
comm._ptr,
|
||||||
|
reduce_input,
|
||||||
|
custom_scatter_out,
|
||||||
|
comm.mnnvl_lamport_rs_local_ptr,
|
||||||
|
comm.mnnvl_lamport_rs_epoch_ptr,
|
||||||
|
comm.mnnvl_buffer_size,
|
||||||
|
)
|
||||||
|
|
||||||
|
def nccl_ag() -> None:
|
||||||
|
dist.all_gather_into_tensor(nccl_gather_out, local, group=device_group)
|
||||||
|
|
||||||
|
def nccl_rs() -> None:
|
||||||
|
dist.reduce_scatter_tensor(
|
||||||
|
nccl_scatter_out,
|
||||||
|
reduce_input,
|
||||||
|
group=device_group,
|
||||||
|
)
|
||||||
|
|
||||||
|
graphs = {
|
||||||
|
"custom_ag_us": capture_graph(custom_ag, graph_repeats),
|
||||||
|
"nccl_ag_us": capture_graph(nccl_ag, graph_repeats),
|
||||||
|
"custom_rs_us": capture_graph(custom_rs, graph_repeats),
|
||||||
|
"nccl_rs_us": capture_graph(nccl_rs, graph_repeats),
|
||||||
|
}
|
||||||
|
times = {
|
||||||
|
name: max_rank_graph_time(
|
||||||
|
graph,
|
||||||
|
graph_repeats,
|
||||||
|
warmup_replays,
|
||||||
|
samples,
|
||||||
|
device_group,
|
||||||
|
cpu_group,
|
||||||
|
)
|
||||||
|
* 1000
|
||||||
|
for name, graph in graphs.items()
|
||||||
|
}
|
||||||
|
torch.testing.assert_close(custom_gather_out, nccl_gather_out)
|
||||||
|
torch.testing.assert_close(custom_scatter_out, nccl_scatter_out)
|
||||||
|
return {
|
||||||
|
"global_tokens": global_tokens,
|
||||||
|
"padded_tokens": padded_tokens,
|
||||||
|
"local_bytes": local.nbytes,
|
||||||
|
"full_bytes": reduce_input.nbytes,
|
||||||
|
**times,
|
||||||
|
"ag_speedup": times["nccl_ag_us"] / times["custom_ag_us"],
|
||||||
|
"rs_speedup": times["nccl_rs_us"] / times["custom_rs_us"],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
args = parse_args()
|
||||||
|
local_rank = int(os.environ["LOCAL_RANK"])
|
||||||
|
torch.accelerator.set_device_index(local_rank)
|
||||||
|
dist.init_process_group("nccl")
|
||||||
|
device_group = dist.group.WORLD
|
||||||
|
cpu_group = dist.new_group(backend="gloo")
|
||||||
|
|
||||||
|
comm = CustomAllreduce(
|
||||||
|
group=cpu_group,
|
||||||
|
device=torch.device("cuda", local_rank),
|
||||||
|
)
|
||||||
|
assert not comm.disabled
|
||||||
|
assert comm.world_size == 16
|
||||||
|
assert comm.mnnvl_only
|
||||||
|
assert comm.mnnvl_multicast_ptr
|
||||||
|
|
||||||
|
results = [
|
||||||
|
benchmark_shape(
|
||||||
|
comm,
|
||||||
|
tokens,
|
||||||
|
args.hidden_size,
|
||||||
|
args.graph_repeats,
|
||||||
|
args.warmup_replays,
|
||||||
|
args.samples,
|
||||||
|
device_group,
|
||||||
|
cpu_group,
|
||||||
|
)
|
||||||
|
for tokens in args.tokens
|
||||||
|
]
|
||||||
|
if dist.get_rank() == 0:
|
||||||
|
print(json.dumps(results, indent=2), flush=True)
|
||||||
|
|
||||||
|
comm.close()
|
||||||
|
dist.destroy_process_group(cpu_group)
|
||||||
|
dist.destroy_process_group()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,201 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
"""
|
||||||
|
Benchmark the RDNAHybridW4A16LinearKernel across decode and prefill shapes.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python benchmark_int4_gemm.py
|
||||||
|
python benchmark_int4_gemm.py --models Qwen/Qwen3-4B
|
||||||
|
python benchmark_int4_gemm.py --group-size 128
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import copy
|
||||||
|
import itertools
|
||||||
|
import os
|
||||||
|
|
||||||
|
import torch
|
||||||
|
|
||||||
|
from vllm.triton_utils import triton
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Weight shapes: [K, N], TP_SPLIT_DIM
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
WEIGHT_SHAPES = {
|
||||||
|
"Qwen/Qwen3-4B": [
|
||||||
|
([2560, 3840], 1), # qkv_proj
|
||||||
|
([2560, 2560], 0), # o_proj
|
||||||
|
([2560, 19456], 1), # gate_up_proj
|
||||||
|
([9728, 2560], 0), # down_proj
|
||||||
|
],
|
||||||
|
"Qwen/Qwen2.5-7B-Instruct": [
|
||||||
|
([3584, 4608], 1),
|
||||||
|
([3584, 3584], 0),
|
||||||
|
([3584, 37888], 1),
|
||||||
|
([18944, 3584], 0),
|
||||||
|
],
|
||||||
|
"trymirai/SmolLM2-1.7B-Instruct-AWQ": [
|
||||||
|
([2048, 6144], 1), # qkv_proj
|
||||||
|
([2048, 2048], 0), # o_proj
|
||||||
|
([2048, 16384], 1), # gate_up_proj
|
||||||
|
([8192, 2048], 0), # down_proj
|
||||||
|
],
|
||||||
|
"RedHatAI/Qwen3-8B-quantized.w4a16": [
|
||||||
|
([4096, 6144], 1), # qkv_proj
|
||||||
|
([4096, 4096], 0), # o_proj
|
||||||
|
([4096, 24576], 1), # gate_up_proj
|
||||||
|
([12288, 4096], 0), # down_proj
|
||||||
|
],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Weight packing
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
def prepare_hybrid_weights(K, N, group_size, device="cuda"):
|
||||||
|
"""Create random weights for benchmarking.
|
||||||
|
|
||||||
|
Returns (w_q_skinny, w_s_skinny, w_fp16, w_zp). The triton path derives
|
||||||
|
its int32 view from w_q_skinny, so no separate int32 buffer is returned.
|
||||||
|
"""
|
||||||
|
num_groups = K // group_size
|
||||||
|
|
||||||
|
# Random packed weights — actual values don't matter for throughput
|
||||||
|
w_q_skinny_i32 = torch.randint(
|
||||||
|
0, 2**31, (N, K // 8), dtype=torch.int32, device=device
|
||||||
|
)
|
||||||
|
w_q_skinny = w_q_skinny_i32.view(torch.int8).contiguous()
|
||||||
|
w_s_skinny = torch.randn(N, num_groups, dtype=torch.float16, device=device) * 0.01
|
||||||
|
|
||||||
|
# Raw per-group zero-points for asymmetric benchmarks
|
||||||
|
w_zp = torch.randint(0, 16, (N, num_groups), dtype=torch.int32, device=device).to(
|
||||||
|
torch.float16
|
||||||
|
)
|
||||||
|
|
||||||
|
# FP16 baseline for F.linear
|
||||||
|
w_fp16 = torch.randn(N, K, dtype=torch.float16, device=device) * 0.01
|
||||||
|
|
||||||
|
return w_q_skinny, w_s_skinny, w_fp16, w_zp
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Benchmark
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
PROVIDERS = ["torch-fp16", "hybrid-w4a16", "hybrid-w4a16-zp"]
|
||||||
|
|
||||||
|
|
||||||
|
@triton.testing.perf_report(
|
||||||
|
triton.testing.Benchmark(
|
||||||
|
x_names=["batch_size"],
|
||||||
|
x_vals=[1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 1024, 2048, 4096],
|
||||||
|
x_log=False,
|
||||||
|
line_arg="provider",
|
||||||
|
line_vals=PROVIDERS,
|
||||||
|
line_names=PROVIDERS,
|
||||||
|
ylabel="TFLOP/s (larger is better)",
|
||||||
|
plot_name="FP16 vs Hybrid W4A16",
|
||||||
|
args={},
|
||||||
|
)
|
||||||
|
)
|
||||||
|
def benchmark(batch_size, provider, N, K, group_size, weights):
|
||||||
|
M = batch_size
|
||||||
|
device = "cuda"
|
||||||
|
dtype = torch.float16
|
||||||
|
a = torch.randn((M, K), device=device, dtype=dtype)
|
||||||
|
|
||||||
|
quantiles = [0.5, 0.2, 0.8]
|
||||||
|
|
||||||
|
if provider == "torch-fp16":
|
||||||
|
w_fp16 = weights["w_fp16"]
|
||||||
|
ms, min_ms, max_ms = triton.testing.do_bench_cudagraph(
|
||||||
|
lambda: torch.nn.functional.linear(a, w_fp16),
|
||||||
|
quantiles=quantiles,
|
||||||
|
)
|
||||||
|
elif provider in ("hybrid-w4a16", "hybrid-w4a16-zp"):
|
||||||
|
from vllm.model_executor.kernels.linear.mixed_precision import (
|
||||||
|
rdna_hybrid_w4a16 as _k,
|
||||||
|
)
|
||||||
|
|
||||||
|
_rdna_hybrid_w4a16_apply_impl = _k._rdna_hybrid_w4a16_apply_impl
|
||||||
|
from vllm.utils.platform_utils import num_compute_units
|
||||||
|
|
||||||
|
w = weights
|
||||||
|
cu_count = num_compute_units()
|
||||||
|
use_zp = provider == "hybrid-w4a16-zp"
|
||||||
|
|
||||||
|
def run():
|
||||||
|
return _rdna_hybrid_w4a16_apply_impl(
|
||||||
|
a,
|
||||||
|
w["w_q_skinny"],
|
||||||
|
w["w_s_skinny"],
|
||||||
|
w["w_zp"] if use_zp else None,
|
||||||
|
None, # bias
|
||||||
|
cu_count,
|
||||||
|
group_size,
|
||||||
|
)
|
||||||
|
|
||||||
|
ms, min_ms, max_ms = triton.testing.do_bench_cudagraph(
|
||||||
|
run,
|
||||||
|
quantiles=quantiles,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
return 0.0, 0.0, 0.0
|
||||||
|
|
||||||
|
to_tflops = lambda t_ms: (2 * M * N * K) * 1e-12 / (t_ms * 1e-3)
|
||||||
|
return to_tflops(ms), to_tflops(max_ms), to_tflops(min_ms)
|
||||||
|
|
||||||
|
|
||||||
|
def prepare_shapes(args):
|
||||||
|
KN_model_names = []
|
||||||
|
for model, tp_size in itertools.product(args.models, args.tp_sizes):
|
||||||
|
for KN, tp_dim in copy.deepcopy(WEIGHT_SHAPES[model]):
|
||||||
|
KN[tp_dim] //= tp_size
|
||||||
|
KN.append(model)
|
||||||
|
KN_model_names.append(KN)
|
||||||
|
return KN_model_names
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Benchmark RDNAHybridW4A16LinearKernel"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--models",
|
||||||
|
nargs="+",
|
||||||
|
type=str,
|
||||||
|
default=["Qwen/Qwen3-4B"],
|
||||||
|
choices=list(WEIGHT_SHAPES.keys()),
|
||||||
|
)
|
||||||
|
parser.add_argument("--tp-sizes", nargs="+", type=int, default=[1])
|
||||||
|
parser.add_argument("--group-size", type=int, default=128)
|
||||||
|
parser.add_argument("--save-path", type=str, default=None)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
for K, N, model in prepare_shapes(args):
|
||||||
|
group_size = args.group_size
|
||||||
|
print(f"\n{'=' * 70}")
|
||||||
|
print(f"{model}, N={N} K={K}, group_size={group_size}")
|
||||||
|
print(f"{'=' * 70}")
|
||||||
|
|
||||||
|
w_q_skinny, w_s_skinny, w_fp16, w_zp = prepare_hybrid_weights(K, N, group_size)
|
||||||
|
|
||||||
|
weights = {
|
||||||
|
"w_q_skinny": w_q_skinny,
|
||||||
|
"w_s_skinny": w_s_skinny,
|
||||||
|
"w_fp16": w_fp16,
|
||||||
|
"w_zp": w_zp,
|
||||||
|
}
|
||||||
|
|
||||||
|
save_path = args.save_path or f"bench_int4_res_n{N}_k{K}"
|
||||||
|
os.makedirs(save_path, exist_ok=True)
|
||||||
|
benchmark.run(
|
||||||
|
print_data=True,
|
||||||
|
show_plots=False,
|
||||||
|
save_path=save_path,
|
||||||
|
N=N,
|
||||||
|
K=K,
|
||||||
|
group_size=group_size,
|
||||||
|
weights=weights,
|
||||||
|
)
|
||||||
|
|
||||||
|
print("\nBenchmark finished!")
|
||||||
@@ -154,7 +154,7 @@ def main(
|
|||||||
scale=scale,
|
scale=scale,
|
||||||
causal=True,
|
causal=True,
|
||||||
alibi_slopes=None,
|
alibi_slopes=None,
|
||||||
sliding_window=window_size,
|
sliding_window=window_size if sliding_window is not None else -1,
|
||||||
block_table=block_tables,
|
block_table=block_tables,
|
||||||
softcap=0,
|
softcap=0,
|
||||||
scheduler_metadata=metadata,
|
scheduler_metadata=metadata,
|
||||||
|
|||||||
@@ -0,0 +1,267 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
"""End-to-end autoregressive decode benchmark: ReplaySSM vs the standard SSM kernel.
|
||||||
|
|
||||||
|
Loads a hybrid Mamba2 model, replicates one prompt across the batch, and times a
|
||||||
|
long greedy decode (CUDA graphs on) once with the standard kernel and once with
|
||||||
|
ReplaySSM, then reports the per-step / throughput speedup. The two modes run in
|
||||||
|
separate subprocesses so each gets a clean CUDA context.
|
||||||
|
|
||||||
|
The FlashInfer FP4-MoE autotuner is disabled by default (it is unstable under
|
||||||
|
CUDA-graph capture on the pre-release Blackwell FP4 path); pass
|
||||||
|
--no-disable-flashinfer-autotune for non-FP4 models.
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
python e2e_decode_speedup.py --model-id nvidia/NVIDIA-Nemotron-3-Nano-4B-BF16
|
||||||
|
python e2e_decode_speedup.py --dtype auto --buffer-len 16 \
|
||||||
|
--model-id nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4 # B300 NVFP4
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
|
||||||
|
DEFAULT_PROMPT = "My cat wrote all this CUDA code for a new language model and"
|
||||||
|
|
||||||
|
MODE_LABEL = {"standard": "standard", "replayssm": "ReplaySSM"}
|
||||||
|
|
||||||
|
|
||||||
|
def parse_args():
|
||||||
|
p = argparse.ArgumentParser(
|
||||||
|
description="E2E decode speedup: ReplaySSM vs the standard SSM kernel."
|
||||||
|
)
|
||||||
|
p.add_argument("--model-id", default="nvidia/NVIDIA-Nemotron-3-Nano-4B-BF16")
|
||||||
|
p.add_argument("--prompt", default=DEFAULT_PROMPT)
|
||||||
|
p.add_argument("--batch-size", type=int, default=256)
|
||||||
|
p.add_argument("--num-steps", type=int, default=1000)
|
||||||
|
p.add_argument("--warmup-steps", type=int, default=128)
|
||||||
|
p.add_argument("--repeats", type=int, default=1)
|
||||||
|
p.add_argument(
|
||||||
|
"--buffer-len", type=int, default=16, help="ReplaySSM input-buffer length."
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--dtype",
|
||||||
|
default="bfloat16",
|
||||||
|
choices=["bfloat16", "float16", "float32", "auto"],
|
||||||
|
)
|
||||||
|
p.add_argument("--gpu-memory-utilization", type=float, default=0.9)
|
||||||
|
p.add_argument("--max-model-len", type=int, default=None)
|
||||||
|
p.add_argument(
|
||||||
|
"--disable-flashinfer-autotune",
|
||||||
|
action=argparse.BooleanOptionalAction,
|
||||||
|
default=True,
|
||||||
|
help="Disable the FlashInfer FP4-MoE autotuner (default: on). "
|
||||||
|
"It is unstable under CUDA-graph capture on the "
|
||||||
|
"pre-release Blackwell FP4 path; pass "
|
||||||
|
"--no-disable-flashinfer-autotune for non-FP4 models.",
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--mamba-ssm-cache-dtype",
|
||||||
|
default="auto",
|
||||||
|
choices=["auto", "float32", "float16", "bfloat16"],
|
||||||
|
help="SSM state dtype (both modes). 'auto' = config-driven; "
|
||||||
|
"'float32' = fp32 state, 'bfloat16' = s16 state.",
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--baseline-ssm-config",
|
||||||
|
default="",
|
||||||
|
help="Pin the STANDARD baseline's SSM launch config as "
|
||||||
|
"'bsm,nw' via override_ssm_config (forces the in-process "
|
||||||
|
"engine so the override reaches the kernel). Empty = off.",
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--worker",
|
||||||
|
choices=["standard", "replayssm"],
|
||||||
|
default=None,
|
||||||
|
help=argparse.SUPPRESS,
|
||||||
|
)
|
||||||
|
return p.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_max_model_len(args) -> int:
|
||||||
|
if args.max_model_len is not None:
|
||||||
|
return args.max_model_len
|
||||||
|
return args.num_steps + 256
|
||||||
|
|
||||||
|
|
||||||
|
def run_worker(args):
|
||||||
|
# override_ssm_config is a module global; it only reaches the model if the
|
||||||
|
# engine runs in-process (default V1 spawns a separate EngineCore). Force it.
|
||||||
|
if args.worker == "standard" and args.baseline_ssm_config:
|
||||||
|
os.environ["VLLM_ENABLE_V1_MULTIPROCESSING"] = "0"
|
||||||
|
|
||||||
|
import torch
|
||||||
|
|
||||||
|
from vllm import LLM, SamplingParams
|
||||||
|
|
||||||
|
mode = args.worker
|
||||||
|
max_model_len = resolve_max_model_len(args)
|
||||||
|
|
||||||
|
llm_kwargs = dict(
|
||||||
|
model=args.model_id,
|
||||||
|
tensor_parallel_size=1,
|
||||||
|
dtype=args.dtype,
|
||||||
|
max_model_len=max_model_len,
|
||||||
|
trust_remote_code=True,
|
||||||
|
enable_prefix_caching=False,
|
||||||
|
enable_chunked_prefill=False,
|
||||||
|
max_num_seqs=args.batch_size,
|
||||||
|
max_num_batched_tokens=max(max_model_len, args.batch_size * 64),
|
||||||
|
enforce_eager=False,
|
||||||
|
disable_log_stats=True,
|
||||||
|
gpu_memory_utilization=args.gpu_memory_utilization,
|
||||||
|
# SSM state dtype (applies to both standard and ReplaySSM).
|
||||||
|
mamba_ssm_cache_dtype=args.mamba_ssm_cache_dtype,
|
||||||
|
)
|
||||||
|
if args.disable_flashinfer_autotune:
|
||||||
|
# FP4-MoE autotuner is unstable under CUDA-graph capture on Blackwell;
|
||||||
|
# re-enable (--no-disable-flashinfer-autotune) only for non-FP4 models.
|
||||||
|
llm_kwargs["kernel_config"] = {"enable_flashinfer_autotune": False}
|
||||||
|
if mode == "replayssm":
|
||||||
|
llm_kwargs.update(use_replayssm=True, replayssm_buffer_len=args.buffer_len)
|
||||||
|
|
||||||
|
_ssm_cm = None
|
||||||
|
if mode == "standard" and args.baseline_ssm_config:
|
||||||
|
from vllm.model_executor.layers.mamba.ops.mamba_ssm import override_ssm_config
|
||||||
|
|
||||||
|
_bsm, _nw = (int(x) for x in args.baseline_ssm_config.split(","))
|
||||||
|
_ssm_cm = override_ssm_config((_bsm, _nw))
|
||||||
|
_ssm_cm.__enter__() # active through LLM() graph capture + decode
|
||||||
|
print(
|
||||||
|
f"[{mode}] override_ssm_config -> (BLOCK_SIZE_M={_bsm}, num_warps={_nw})",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
llm = LLM(**llm_kwargs)
|
||||||
|
prompts = [args.prompt] * args.batch_size
|
||||||
|
|
||||||
|
def timed_generate(n_tokens):
|
||||||
|
sp = SamplingParams(
|
||||||
|
n=1,
|
||||||
|
temperature=0.0,
|
||||||
|
ignore_eos=True,
|
||||||
|
min_tokens=n_tokens,
|
||||||
|
max_tokens=n_tokens,
|
||||||
|
)
|
||||||
|
if torch.accelerator.is_available():
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
outs = llm.generate(prompts, sp, use_tqdm=False)
|
||||||
|
if torch.accelerator.is_available():
|
||||||
|
torch.accelerator.synchronize()
|
||||||
|
elapsed = time.perf_counter() - t0
|
||||||
|
produced = min(len(o.outputs[0].token_ids) for o in outs)
|
||||||
|
assert produced == n_tokens, f"expected {n_tokens} tokens, got {produced}"
|
||||||
|
return elapsed
|
||||||
|
|
||||||
|
timed_generate(args.warmup_steps)
|
||||||
|
|
||||||
|
best = None
|
||||||
|
for _ in range(args.repeats):
|
||||||
|
elapsed = timed_generate(args.num_steps)
|
||||||
|
tok_s = args.batch_size * args.num_steps / elapsed
|
||||||
|
per_step_ms = elapsed / args.num_steps * 1e3
|
||||||
|
print(
|
||||||
|
f"[{mode}] {elapsed:.3f}s {tok_s:,.0f} tok/s {per_step_ms:.3f} ms/step",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
if best is None or elapsed < best["elapsed_s"]:
|
||||||
|
best = {
|
||||||
|
"mode": mode,
|
||||||
|
"elapsed_s": elapsed,
|
||||||
|
"tok_s": tok_s,
|
||||||
|
"per_step_ms": per_step_ms,
|
||||||
|
}
|
||||||
|
|
||||||
|
print("RESULT_JSON " + json.dumps(best), flush=True)
|
||||||
|
if _ssm_cm is not None:
|
||||||
|
_ssm_cm.__exit__(None, None, None)
|
||||||
|
|
||||||
|
|
||||||
|
def run_one_mode(args, mode) -> dict:
|
||||||
|
cmd = [
|
||||||
|
sys.executable,
|
||||||
|
__file__,
|
||||||
|
"--worker",
|
||||||
|
mode,
|
||||||
|
"--model-id",
|
||||||
|
args.model_id,
|
||||||
|
"--prompt",
|
||||||
|
args.prompt,
|
||||||
|
"--batch-size",
|
||||||
|
str(args.batch_size),
|
||||||
|
"--num-steps",
|
||||||
|
str(args.num_steps),
|
||||||
|
"--warmup-steps",
|
||||||
|
str(args.warmup_steps),
|
||||||
|
"--repeats",
|
||||||
|
str(args.repeats),
|
||||||
|
"--buffer-len",
|
||||||
|
str(args.buffer_len),
|
||||||
|
"--dtype",
|
||||||
|
args.dtype,
|
||||||
|
"--gpu-memory-utilization",
|
||||||
|
str(args.gpu_memory_utilization),
|
||||||
|
"--mamba-ssm-cache-dtype",
|
||||||
|
args.mamba_ssm_cache_dtype,
|
||||||
|
"--baseline-ssm-config",
|
||||||
|
args.baseline_ssm_config,
|
||||||
|
]
|
||||||
|
cmd.append(
|
||||||
|
"--disable-flashinfer-autotune"
|
||||||
|
if args.disable_flashinfer_autotune
|
||||||
|
else "--no-disable-flashinfer-autotune"
|
||||||
|
)
|
||||||
|
if args.max_model_len is not None:
|
||||||
|
cmd += ["--max-model-len", str(args.max_model_len)]
|
||||||
|
|
||||||
|
result = None
|
||||||
|
proc = subprocess.Popen(
|
||||||
|
cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, bufsize=1
|
||||||
|
)
|
||||||
|
for line in proc.stdout:
|
||||||
|
sys.stdout.write(line)
|
||||||
|
sys.stdout.flush()
|
||||||
|
if line.startswith("RESULT_JSON "):
|
||||||
|
result = json.loads(line[len("RESULT_JSON ") :])
|
||||||
|
proc.wait()
|
||||||
|
if proc.returncode != 0:
|
||||||
|
raise RuntimeError(f"mode '{mode}' worker exited with {proc.returncode}")
|
||||||
|
if result is None:
|
||||||
|
raise RuntimeError(f"mode '{mode}' produced no RESULT_JSON line")
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
args = parse_args()
|
||||||
|
if args.worker is not None:
|
||||||
|
run_worker(args)
|
||||||
|
return
|
||||||
|
|
||||||
|
print(
|
||||||
|
f"model={args.model_id} batch_size={args.batch_size} "
|
||||||
|
f"steps={args.num_steps} buffer_len={args.buffer_len} dtype={args.dtype}"
|
||||||
|
)
|
||||||
|
|
||||||
|
std = run_one_mode(args, "standard")
|
||||||
|
fla = run_one_mode(args, "replayssm")
|
||||||
|
speedup = std["per_step_ms"] / fla["per_step_ms"]
|
||||||
|
|
||||||
|
print()
|
||||||
|
header = f"{'mode':<10}{'ms/step':>12}{'tok/s':>16}{'wall (s)':>12}"
|
||||||
|
print(header)
|
||||||
|
print("-" * len(header))
|
||||||
|
for r in (std, fla):
|
||||||
|
print(
|
||||||
|
f"{MODE_LABEL[r['mode']]:<10}{r['per_step_ms']:>12.3f}"
|
||||||
|
f"{r['tok_s']:>16,.0f}{r['elapsed_s']:>12.3f}"
|
||||||
|
)
|
||||||
|
print("-" * len(header))
|
||||||
|
print(f"speedup (standard / ReplaySSM, per step): {speedup:.3f}x")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -15,6 +15,7 @@ endif()
|
|||||||
#
|
#
|
||||||
set(ENABLE_X86_ISA $ENV{VLLM_CPU_X86})
|
set(ENABLE_X86_ISA $ENV{VLLM_CPU_X86})
|
||||||
set(ENABLE_ARM_BF16 $ENV{VLLM_CPU_ARM_BF16})
|
set(ENABLE_ARM_BF16 $ENV{VLLM_CPU_ARM_BF16})
|
||||||
|
set(ENABLE_ARM_I8MM $ENV{VLLM_CPU_ARM_I8MM})
|
||||||
set(ENABLE_RVV_BF16 $ENV{VLLM_CPU_RVV_BF16})
|
set(ENABLE_RVV_BF16 $ENV{VLLM_CPU_RVV_BF16})
|
||||||
|
|
||||||
include_directories("${CMAKE_SOURCE_DIR}/csrc")
|
include_directories("${CMAKE_SOURCE_DIR}/csrc")
|
||||||
@@ -96,12 +97,14 @@ if (MACOSX_FOUND AND CMAKE_SYSTEM_PROCESSOR STREQUAL "arm64")
|
|||||||
set(ENABLE_NUMA OFF)
|
set(ENABLE_NUMA OFF)
|
||||||
check_sysctl(hw.optional.neon ASIMD_FOUND)
|
check_sysctl(hw.optional.neon ASIMD_FOUND)
|
||||||
check_sysctl(hw.optional.arm.FEAT_BF16 ARM_BF16_FOUND)
|
check_sysctl(hw.optional.arm.FEAT_BF16 ARM_BF16_FOUND)
|
||||||
|
check_sysctl(hw.optional.arm.FEAT_I8MM ARM_I8MM_FOUND)
|
||||||
else()
|
else()
|
||||||
find_isa(${CPUINFO} "Power11" POWER11_FOUND)
|
find_isa(${CPUINFO} "Power11" POWER11_FOUND)
|
||||||
find_isa(${CPUINFO} "POWER10" POWER10_FOUND)
|
find_isa(${CPUINFO} "POWER10" POWER10_FOUND)
|
||||||
find_isa(${CPUINFO} "POWER9" POWER9_FOUND)
|
find_isa(${CPUINFO} "POWER9" POWER9_FOUND)
|
||||||
find_isa(${CPUINFO} "asimd" ASIMD_FOUND) # Check for ARM NEON support
|
find_isa(${CPUINFO} "asimd" ASIMD_FOUND) # Check for ARM NEON support
|
||||||
find_isa(${CPUINFO} "bf16" ARM_BF16_FOUND) # Check for ARM BF16 support
|
find_isa(${CPUINFO} "bf16" ARM_BF16_FOUND) # Check for ARM BF16 support
|
||||||
|
find_isa(${CPUINFO} "i8mm" ARM_I8MM_FOUND) # Check for ARM I8MM support
|
||||||
find_isa(${CPUINFO} "S390" S390_FOUND)
|
find_isa(${CPUINFO} "S390" S390_FOUND)
|
||||||
find_isa(${CPUINFO} "zvfhmin" RVV_FP16_FOUND) # Check for RISC-V Vector FP16 support
|
find_isa(${CPUINFO} "zvfhmin" RVV_FP16_FOUND) # Check for RISC-V Vector FP16 support
|
||||||
find_isa(${CPUINFO} "zvfbfmin" RVV_BF16_FOUND) # Check for RISC-V Vector BF16 support
|
find_isa(${CPUINFO} "zvfbfmin" RVV_BF16_FOUND) # Check for RISC-V Vector BF16 support
|
||||||
@@ -111,6 +114,11 @@ else()
|
|||||||
set(ARM_BF16_FOUND ON)
|
set(ARM_BF16_FOUND ON)
|
||||||
message(STATUS "ARM BF16 support enabled via VLLM_CPU_ARM_BF16 environment variable")
|
message(STATUS "ARM BF16 support enabled via VLLM_CPU_ARM_BF16 environment variable")
|
||||||
endif()
|
endif()
|
||||||
|
if (ENABLE_ARM_I8MM)
|
||||||
|
set(ARM_I8MM_FOUND ON)
|
||||||
|
message(STATUS
|
||||||
|
"ARM I8MM support enabled via VLLM_CPU_ARM_I8MM environment variable")
|
||||||
|
endif()
|
||||||
# Some kernels (e.g. Bianbu on Spacemit X100) do not report zvfbfmin
|
# Some kernels (e.g. Bianbu on Spacemit X100) do not report zvfbfmin
|
||||||
# in /proc/cpuinfo despite hardware support. VLLM_CPU_RVV_BF16=1
|
# in /proc/cpuinfo despite hardware support. VLLM_CPU_RVV_BF16=1
|
||||||
# overrides the detection result.
|
# overrides the detection result.
|
||||||
@@ -166,6 +174,11 @@ elseif (ASIMD_FOUND)
|
|||||||
message(WARNING "BF16 functionality is not available")
|
message(WARNING "BF16 functionality is not available")
|
||||||
set(MARCH_FLAGS "-march=armv8.2-a+dotprod+fp16")
|
set(MARCH_FLAGS "-march=armv8.2-a+dotprod+fp16")
|
||||||
endif()
|
endif()
|
||||||
|
if(ARM_I8MM_FOUND)
|
||||||
|
message(STATUS "I8MM extension detected")
|
||||||
|
string(APPEND MARCH_FLAGS "+i8mm")
|
||||||
|
add_compile_definitions(ARM_I8MM_SUPPORT)
|
||||||
|
endif()
|
||||||
list(APPEND CXX_COMPILE_FLAGS ${MARCH_FLAGS})
|
list(APPEND CXX_COMPILE_FLAGS ${MARCH_FLAGS})
|
||||||
elseif (S390_FOUND)
|
elseif (S390_FOUND)
|
||||||
message(STATUS "S390 detected")
|
message(STATUS "S390 detected")
|
||||||
@@ -430,6 +443,7 @@ set(VLLM_EXT_SRC
|
|||||||
"csrc/cpu/layernorm.cpp"
|
"csrc/cpu/layernorm.cpp"
|
||||||
"csrc/cpu/mla_decode.cpp"
|
"csrc/cpu/mla_decode.cpp"
|
||||||
"csrc/cpu/pos_encoding.cpp"
|
"csrc/cpu/pos_encoding.cpp"
|
||||||
|
"csrc/cpu/mamba_cpu.cpp"
|
||||||
"csrc/moe/dynamic_4bit_int_moe_cpu.cpp"
|
"csrc/moe/dynamic_4bit_int_moe_cpu.cpp"
|
||||||
"csrc/cpu/cpu_attn.cpp"
|
"csrc/cpu/cpu_attn.cpp"
|
||||||
"csrc/cpu/torch_bindings.cpp")
|
"csrc/cpu/torch_bindings.cpp")
|
||||||
@@ -446,8 +460,13 @@ if (ASIMD_FOUND AND NOT APPLE_SILICON_FOUND)
|
|||||||
"csrc/cpu/shm.cpp"
|
"csrc/cpu/shm.cpp"
|
||||||
"csrc/cpu/activation_lut_bf16.cpp"
|
"csrc/cpu/activation_lut_bf16.cpp"
|
||||||
"csrc/cpu/cpu_tanhf_neon.hpp"
|
"csrc/cpu/cpu_tanhf_neon.hpp"
|
||||||
"csrc/cpu/cpu_fused_moe.cpp"
|
|
||||||
${VLLM_EXT_SRC})
|
${VLLM_EXT_SRC})
|
||||||
|
if (ARM_BF16_FOUND)
|
||||||
|
set(VLLM_EXT_SRC "csrc/cpu/cpu_fused_moe.cpp" ${VLLM_EXT_SRC})
|
||||||
|
if (ARM_I8MM_FOUND)
|
||||||
|
set(VLLM_EXT_SRC "csrc/cpu/cpu_fused_moe_int8.cpp" ${VLLM_EXT_SRC})
|
||||||
|
endif()
|
||||||
|
endif()
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
if (POWER9_FOUND OR POWER10_FOUND OR POWER11_FOUND)
|
if (POWER9_FOUND OR POWER10_FOUND OR POWER11_FOUND)
|
||||||
@@ -489,6 +508,7 @@ if (ENABLE_X86_ISA)
|
|||||||
"csrc/cpu/spec_decode_utils.cpp"
|
"csrc/cpu/spec_decode_utils.cpp"
|
||||||
"csrc/cpu/cpu_attn.cpp"
|
"csrc/cpu/cpu_attn.cpp"
|
||||||
"csrc/cpu/dnnl_kernels.cpp"
|
"csrc/cpu/dnnl_kernels.cpp"
|
||||||
|
"csrc/cpu/mamba_cpu.cpp"
|
||||||
"csrc/cpu/torch_bindings.cpp"
|
"csrc/cpu/torch_bindings.cpp"
|
||||||
# TODO: Remove these files
|
# TODO: Remove these files
|
||||||
"csrc/cpu/activation.cpp"
|
"csrc/cpu/activation.cpp"
|
||||||
@@ -502,6 +522,7 @@ if (ENABLE_X86_ISA)
|
|||||||
"csrc/cpu/utils.cpp"
|
"csrc/cpu/utils.cpp"
|
||||||
"csrc/cpu/spec_decode_utils.cpp"
|
"csrc/cpu/spec_decode_utils.cpp"
|
||||||
"csrc/cpu/cpu_attn.cpp"
|
"csrc/cpu/cpu_attn.cpp"
|
||||||
|
"csrc/cpu/mamba_cpu.cpp"
|
||||||
"csrc/cpu/dnnl_kernels.cpp"
|
"csrc/cpu/dnnl_kernels.cpp"
|
||||||
"csrc/cpu/torch_bindings.cpp"
|
"csrc/cpu/torch_bindings.cpp"
|
||||||
# TODO: Remove these files
|
# TODO: Remove these files
|
||||||
|
|||||||
@@ -68,6 +68,9 @@ endif()
|
|||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.8)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.8)
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.9)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.9)
|
||||||
list(APPEND DEEPGEMM_SUPPORT_ARCHS "10.0f")
|
list(APPEND DEEPGEMM_SUPPORT_ARCHS "10.0f")
|
||||||
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.4)
|
||||||
|
list(APPEND DEEPGEMM_SUPPORT_ARCHS "10.7f")
|
||||||
|
endif()
|
||||||
else()
|
else()
|
||||||
list(APPEND DEEPGEMM_SUPPORT_ARCHS "10.0a")
|
list(APPEND DEEPGEMM_SUPPORT_ARCHS "10.0a")
|
||||||
endif()
|
endif()
|
||||||
|
|||||||
@@ -0,0 +1,74 @@
|
|||||||
|
include(FetchContent)
|
||||||
|
|
||||||
|
if(DEFINED ENV{FLASH_KDA_SRC_DIR})
|
||||||
|
set(FLASH_KDA_SRC_DIR $ENV{FLASH_KDA_SRC_DIR})
|
||||||
|
endif()
|
||||||
|
|
||||||
|
if(FLASH_KDA_SRC_DIR)
|
||||||
|
FetchContent_Declare(
|
||||||
|
flashkda
|
||||||
|
SOURCE_DIR ${FLASH_KDA_SRC_DIR}
|
||||||
|
)
|
||||||
|
else()
|
||||||
|
FetchContent_Declare(
|
||||||
|
flashkda
|
||||||
|
GIT_REPOSITORY https://github.com/vllm-project/FlashKDA.git
|
||||||
|
GIT_TAG a3e42bbbece3bb38f7c426b880315294a336e82f
|
||||||
|
GIT_PROGRESS TRUE
|
||||||
|
GIT_SUBMODULES cutlass
|
||||||
|
)
|
||||||
|
endif()
|
||||||
|
|
||||||
|
FetchContent_MakeAvailable(flashkda)
|
||||||
|
message(STATUS "FlashKDA is available at ${flashkda_SOURCE_DIR}")
|
||||||
|
|
||||||
|
set(FLASH_KDA_SUPPORT_ARCHS)
|
||||||
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.0)
|
||||||
|
list(APPEND FLASH_KDA_SUPPORT_ARCHS "9.0a")
|
||||||
|
endif()
|
||||||
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
|
list(APPEND FLASH_KDA_SUPPORT_ARCHS "10.0f" "12.0f")
|
||||||
|
elseif(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.9)
|
||||||
|
list(APPEND FLASH_KDA_SUPPORT_ARCHS "10.0a" "10.3a" "12.0a")
|
||||||
|
endif()
|
||||||
|
|
||||||
|
cuda_archs_loose_intersection(
|
||||||
|
FLASH_KDA_ARCHS "${FLASH_KDA_SUPPORT_ARCHS}" "${CUDA_ARCHS}")
|
||||||
|
|
||||||
|
if(FLASH_KDA_ARCHS)
|
||||||
|
message(STATUS "FlashKDA CUDA architectures: ${FLASH_KDA_ARCHS}")
|
||||||
|
|
||||||
|
set(FLASH_KDA_SOURCES
|
||||||
|
csrc/flashkda_registration.cpp
|
||||||
|
${flashkda_SOURCE_DIR}/csrc/flash_kda.cpp
|
||||||
|
${flashkda_SOURCE_DIR}/csrc/smxx/fwd_launch.cu)
|
||||||
|
set(FLASH_KDA_INCLUDES
|
||||||
|
${flashkda_SOURCE_DIR}/csrc
|
||||||
|
${flashkda_SOURCE_DIR}/cutlass/include
|
||||||
|
${flashkda_SOURCE_DIR}/cutlass/examples/common
|
||||||
|
${flashkda_SOURCE_DIR}/cutlass/tools/util/include)
|
||||||
|
|
||||||
|
set_gencode_flags_for_srcs(
|
||||||
|
SRCS "${FLASH_KDA_SOURCES}"
|
||||||
|
CUDA_ARCHS "${FLASH_KDA_ARCHS}")
|
||||||
|
|
||||||
|
define_extension_target(
|
||||||
|
_flashkda_C
|
||||||
|
DESTINATION vllm
|
||||||
|
LANGUAGE ${VLLM_GPU_LANG}
|
||||||
|
SOURCES ${FLASH_KDA_SOURCES}
|
||||||
|
COMPILE_FLAGS ${VLLM_GPU_FLAGS}
|
||||||
|
ARCHITECTURES ${VLLM_GPU_ARCHES}
|
||||||
|
INCLUDE_DIRECTORIES ${FLASH_KDA_INCLUDES}
|
||||||
|
USE_SABI 3
|
||||||
|
WITH_SOABI)
|
||||||
|
|
||||||
|
target_compile_options(_flashkda_C PRIVATE
|
||||||
|
$<$<COMPILE_LANGUAGE:CUDA>:-UPy_LIMITED_API --expt-relaxed-constexpr --expt-extended-lambda --use_fast_math -O3>
|
||||||
|
$<$<COMPILE_LANGUAGE:CXX>:-UPy_LIMITED_API>)
|
||||||
|
else()
|
||||||
|
message(STATUS
|
||||||
|
"FlashKDA will not compile: CUDA >=12.0 and a supported architecture "
|
||||||
|
"(SM90, SM10x, or SM12x) are required")
|
||||||
|
add_custom_target(_flashkda_C)
|
||||||
|
endif()
|
||||||
@@ -19,7 +19,7 @@ else()
|
|||||||
FetchContent_Declare(
|
FetchContent_Declare(
|
||||||
flashmla
|
flashmla
|
||||||
GIT_REPOSITORY https://github.com/vllm-project/FlashMLA
|
GIT_REPOSITORY https://github.com/vllm-project/FlashMLA
|
||||||
GIT_TAG b70aff3d110a2b1a037e62eac295166b5143643a
|
GIT_TAG a8f794d1251cbfd88a5011445dd5582289c727e4
|
||||||
GIT_PROGRESS TRUE
|
GIT_PROGRESS TRUE
|
||||||
CONFIGURE_COMMAND ""
|
CONFIGURE_COMMAND ""
|
||||||
BUILD_COMMAND ""
|
BUILD_COMMAND ""
|
||||||
@@ -35,7 +35,7 @@ set(FLASHMLA_VENDOR_DIR "${CMAKE_SOURCE_DIR}/vllm/third_party/flashmla")
|
|||||||
file(MAKE_DIRECTORY "${FLASHMLA_VENDOR_DIR}")
|
file(MAKE_DIRECTORY "${FLASHMLA_VENDOR_DIR}")
|
||||||
file(READ "${flashmla_SOURCE_DIR}/flash_mla/flash_mla_interface.py"
|
file(READ "${flashmla_SOURCE_DIR}/flash_mla/flash_mla_interface.py"
|
||||||
FLASHMLA_INTERFACE_CONTENT)
|
FLASHMLA_INTERFACE_CONTENT)
|
||||||
string(REPLACE "import flash_mla.cuda as flash_mla_cuda"
|
string(REPLACE "flash_mla_cuda = torch.ops._flashmla_C"
|
||||||
"import vllm._flashmla_C\nflash_mla_cuda = torch.ops._flashmla_C"
|
"import vllm._flashmla_C\nflash_mla_cuda = torch.ops._flashmla_C"
|
||||||
FLASHMLA_INTERFACE_CONTENT
|
FLASHMLA_INTERFACE_CONTENT
|
||||||
"${FLASHMLA_INTERFACE_CONTENT}")
|
"${FLASHMLA_INTERFACE_CONTENT}")
|
||||||
@@ -60,6 +60,9 @@ if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.9)
|
|||||||
# CUDA 12.9 has introduced "Family-Specific Architecture Features"
|
# CUDA 12.9 has introduced "Family-Specific Architecture Features"
|
||||||
# this supports all compute_10x family
|
# this supports all compute_10x family
|
||||||
list(APPEND SUPPORT_ARCHS "10.0f")
|
list(APPEND SUPPORT_ARCHS "10.0f")
|
||||||
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.4)
|
||||||
|
list(APPEND SUPPORT_ARCHS "10.7f")
|
||||||
|
endif()
|
||||||
elseif(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.8)
|
elseif(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.8)
|
||||||
list(APPEND SUPPORT_ARCHS "10.0a")
|
list(APPEND SUPPORT_ARCHS "10.0a")
|
||||||
endif()
|
endif()
|
||||||
@@ -72,7 +75,7 @@ if(FLASH_MLA_ARCHS)
|
|||||||
list(APPEND VLLM_FLASHMLA_GPU_FLAGS "--expt-relaxed-constexpr" "--expt-extended-lambda" "--use_fast_math")
|
list(APPEND VLLM_FLASHMLA_GPU_FLAGS "--expt-relaxed-constexpr" "--expt-extended-lambda" "--use_fast_math")
|
||||||
|
|
||||||
set(FlashMLA_SOURCES
|
set(FlashMLA_SOURCES
|
||||||
${flashmla_SOURCE_DIR}/csrc/torch_api.cpp
|
${flashmla_SOURCE_DIR}/csrc/api/api.cpp
|
||||||
|
|
||||||
# Misc kernels for decoding
|
# Misc kernels for decoding
|
||||||
${flashmla_SOURCE_DIR}/csrc/smxx/decode/get_decoding_sched_meta/get_decoding_sched_meta.cu
|
${flashmla_SOURCE_DIR}/csrc/smxx/decode/get_decoding_sched_meta/get_decoding_sched_meta.cu
|
||||||
@@ -128,6 +131,7 @@ if(FLASH_MLA_ARCHS)
|
|||||||
|
|
||||||
set(FlashMLA_Extension_INCLUDES
|
set(FlashMLA_Extension_INCLUDES
|
||||||
${flashmla_SOURCE_DIR}/csrc
|
${flashmla_SOURCE_DIR}/csrc
|
||||||
|
${flashmla_SOURCE_DIR}/csrc/kerutils/include
|
||||||
${flashmla_SOURCE_DIR}/csrc/extension/sm90/dense_fp8/
|
${flashmla_SOURCE_DIR}/csrc/extension/sm90/dense_fp8/
|
||||||
${flashmla_SOURCE_DIR}/csrc/cutlass/include
|
${flashmla_SOURCE_DIR}/csrc/cutlass/include
|
||||||
${flashmla_SOURCE_DIR}/csrc/cutlass/tools/util/include
|
${flashmla_SOURCE_DIR}/csrc/cutlass/tools/util/include
|
||||||
@@ -152,15 +156,18 @@ if(FLASH_MLA_ARCHS)
|
|||||||
USE_SABI 3
|
USE_SABI 3
|
||||||
WITH_SOABI)
|
WITH_SOABI)
|
||||||
|
|
||||||
# Keep Stable ABI for the module, but *not* for CUDA/C++ files.
|
# Enable C++20 for the FlashMLA sources (required for std::span, requires, etc.)
|
||||||
# This prevents Py_LIMITED_API from affecting nvcc and C++ compiles.
|
|
||||||
# Also enable C++20 for the FlashMLA sources (required for std::span, requires, etc.)
|
|
||||||
target_compile_options(_flashmla_C PRIVATE
|
target_compile_options(_flashmla_C PRIVATE
|
||||||
$<$<COMPILE_LANGUAGE:CUDA>:-UPy_LIMITED_API>
|
|
||||||
$<$<COMPILE_LANGUAGE:CXX>:-UPy_LIMITED_API>
|
|
||||||
$<$<COMPILE_LANGUAGE:CXX>:-std=c++20>
|
$<$<COMPILE_LANGUAGE:CXX>:-std=c++20>
|
||||||
$<$<COMPILE_LANGUAGE:CUDA>:-std=c++20>)
|
$<$<COMPILE_LANGUAGE:CUDA>:-std=c++20>)
|
||||||
|
|
||||||
|
# _flashmla_C is now ABI-stable torch 2.11+
|
||||||
|
target_compile_definitions(_flashmla_C PRIVATE
|
||||||
|
TORCH_TARGET_VERSION=0x020B000000000000ULL)
|
||||||
|
if(VLLM_GPU_LANG STREQUAL "CUDA")
|
||||||
|
target_compile_definitions(_flashmla_C PRIVATE USE_CUDA)
|
||||||
|
endif()
|
||||||
|
|
||||||
define_extension_target(
|
define_extension_target(
|
||||||
_flashmla_extension_C
|
_flashmla_extension_C
|
||||||
DESTINATION vllm
|
DESTINATION vllm
|
||||||
@@ -172,15 +179,15 @@ if(FLASH_MLA_ARCHS)
|
|||||||
USE_SABI 3
|
USE_SABI 3
|
||||||
WITH_SOABI)
|
WITH_SOABI)
|
||||||
|
|
||||||
# Keep Stable ABI for the module, but *not* for CUDA/C++ files.
|
# _flashmla_extension_C is now ABI-stable w/ torch 2.11+
|
||||||
# This prevents Py_LIMITED_API from affecting nvcc and C++ compiles.
|
target_compile_definitions(_flashmla_extension_C PRIVATE
|
||||||
target_compile_options(_flashmla_extension_C PRIVATE
|
TORCH_TARGET_VERSION=0x020B000000000000ULL)
|
||||||
$<$<COMPILE_LANGUAGE:CUDA>:-UPy_LIMITED_API>
|
if(VLLM_GPU_LANG STREQUAL "CUDA")
|
||||||
$<$<COMPILE_LANGUAGE:CXX>:-UPy_LIMITED_API>)
|
target_compile_definitions(_flashmla_extension_C PRIVATE USE_CUDA)
|
||||||
|
endif()
|
||||||
else()
|
else()
|
||||||
message(STATUS "FlashMLA will not compile: unsupported CUDA architecture ${CUDA_ARCHS}")
|
message(STATUS "FlashMLA will not compile: unsupported CUDA architecture ${CUDA_ARCHS}")
|
||||||
# Create empty targets for setup.py on unsupported systems
|
# Create empty targets for setup.py on unsupported systems
|
||||||
add_custom_target(_flashmla_C)
|
add_custom_target(_flashmla_C)
|
||||||
add_custom_target(_flashmla_extension_C)
|
add_custom_target(_flashmla_extension_C)
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
|
|||||||
@@ -17,7 +17,7 @@ else()
|
|||||||
FetchContent_Declare(
|
FetchContent_Declare(
|
||||||
fmha_sm100
|
fmha_sm100
|
||||||
GIT_REPOSITORY https://github.com/vllm-project/MSA.git
|
GIT_REPOSITORY https://github.com/vllm-project/MSA.git
|
||||||
GIT_TAG 2e63ec37a0fc29bc20f39cd1a52e0f5affc33a73
|
GIT_TAG 890aaa1a37a598ad17ccff0827fea21540d381fa
|
||||||
GIT_PROGRESS TRUE
|
GIT_PROGRESS TRUE
|
||||||
CONFIGURE_COMMAND ""
|
CONFIGURE_COMMAND ""
|
||||||
BUILD_COMMAND ""
|
BUILD_COMMAND ""
|
||||||
|
|||||||
@@ -22,7 +22,7 @@ if(QUTLASS_SRC_DIR)
|
|||||||
set(qutlass_BINARY_DIR "${CMAKE_BINARY_DIR}/qutlass-binary-dir-unused")
|
set(qutlass_BINARY_DIR "${CMAKE_BINARY_DIR}/qutlass-binary-dir-unused")
|
||||||
else()
|
else()
|
||||||
set(_QUTLASS_UPSTREAM_REPO "https://github.com/IST-DASLab/qutlass.git")
|
set(_QUTLASS_UPSTREAM_REPO "https://github.com/IST-DASLab/qutlass.git")
|
||||||
set(_QUTLASS_UPSTREAM_TAG "830d2c4537c7396e14a02a46fbddd18b5d107c65")
|
set(_QUTLASS_UPSTREAM_TAG "e74319e3405ce6d71965732880f5dc1f52371f64")
|
||||||
|
|
||||||
set(_qutlass_fc_root "${FETCHCONTENT_BASE_DIR}")
|
set(_qutlass_fc_root "${FETCHCONTENT_BASE_DIR}")
|
||||||
if(NOT _qutlass_fc_root)
|
if(NOT _qutlass_fc_root)
|
||||||
@@ -55,7 +55,11 @@ message(STATUS "[QUTLASS] QuTLASS is available at ${qutlass_SOURCE_DIR}")
|
|||||||
|
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
cuda_archs_loose_intersection(QUTLASS_SM120_ARCHS "12.0f" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(QUTLASS_SM120_ARCHS "12.0f" "${CUDA_ARCHS}")
|
||||||
cuda_archs_loose_intersection(QUTLASS_SM100_ARCHS "10.0f" "${CUDA_ARCHS}")
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.4)
|
||||||
|
cuda_archs_loose_intersection(QUTLASS_SM100_ARCHS "10.0f;10.7f" "${CUDA_ARCHS}")
|
||||||
|
else()
|
||||||
|
cuda_archs_loose_intersection(QUTLASS_SM100_ARCHS "10.0f" "${CUDA_ARCHS}")
|
||||||
|
endif()
|
||||||
else()
|
else()
|
||||||
cuda_archs_loose_intersection(QUTLASS_SM120_ARCHS "12.0a;12.1a" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(QUTLASS_SM120_ARCHS "12.0a;12.1a" "${CUDA_ARCHS}")
|
||||||
cuda_archs_loose_intersection(QUTLASS_SM100_ARCHS "10.0a;10.3a" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(QUTLASS_SM100_ARCHS "10.0a;10.3a" "${CUDA_ARCHS}")
|
||||||
@@ -125,8 +129,6 @@ if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.8 AND QUTLASS_ARCHS)
|
|||||||
CUDA_ARCHS "${QUTLASS_ARCHS}"
|
CUDA_ARCHS "${QUTLASS_ARCHS}"
|
||||||
)
|
)
|
||||||
|
|
||||||
# QuTLASS uses legacy ATen headers and cannot be built with TORCH_TARGET_VERSION.
|
|
||||||
# Keep it as its own extension (registers torch.ops._qutlass_C).
|
|
||||||
define_extension_target(
|
define_extension_target(
|
||||||
_qutlass_C
|
_qutlass_C
|
||||||
DESTINATION vllm
|
DESTINATION vllm
|
||||||
@@ -139,9 +141,11 @@ if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.8 AND QUTLASS_ARCHS)
|
|||||||
WITH_SOABI)
|
WITH_SOABI)
|
||||||
|
|
||||||
target_compile_definitions(_qutlass_C PRIVATE
|
target_compile_definitions(_qutlass_C PRIVATE
|
||||||
QUTLASS_DISABLE_PYBIND=1
|
QUTLASS_MINIMAL_BUILD=1
|
||||||
TARGET_CUDA_ARCH=${QUTLASS_TARGET_CC}
|
TARGET_CUDA_ARCH=${QUTLASS_TARGET_CC}
|
||||||
CUTLASS_ENABLE_DIRECT_CUDA_DRIVER_CALL=1)
|
CUTLASS_ENABLE_DIRECT_CUDA_DRIVER_CALL=1
|
||||||
|
TORCH_TARGET_VERSION=0x020B000000000000ULL
|
||||||
|
USE_CUDA)
|
||||||
|
|
||||||
set_property(SOURCE ${QUTLASS_SOURCES} APPEND PROPERTY COMPILE_OPTIONS
|
set_property(SOURCE ${QUTLASS_SOURCES} APPEND PROPERTY COMPILE_OPTIONS
|
||||||
$<$<COMPILE_LANGUAGE:CUDA>:--expt-relaxed-constexpr --use_fast_math -O3>
|
$<$<COMPILE_LANGUAGE:CUDA>:--expt-relaxed-constexpr --use_fast_math -O3>
|
||||||
|
|||||||
@@ -0,0 +1,50 @@
|
|||||||
|
include(FetchContent)
|
||||||
|
|
||||||
|
if(DEFINED ENV{TML_FA4_SRC_DIR})
|
||||||
|
set(TML_FA4_SRC_DIR $ENV{TML_FA4_SRC_DIR})
|
||||||
|
endif()
|
||||||
|
|
||||||
|
if(TML_FA4_SRC_DIR)
|
||||||
|
FetchContent_Declare(
|
||||||
|
tml_fa4
|
||||||
|
SOURCE_DIR ${TML_FA4_SRC_DIR}
|
||||||
|
CONFIGURE_COMMAND ""
|
||||||
|
BUILD_COMMAND "")
|
||||||
|
else()
|
||||||
|
FetchContent_Declare(
|
||||||
|
tml_fa4
|
||||||
|
GIT_REPOSITORY https://github.com/vllm-project/tml-fa4.git
|
||||||
|
GIT_TAG b206834606ed5b5f21f8eed6b0683f528ea9cf7d
|
||||||
|
GIT_PROGRESS TRUE
|
||||||
|
CONFIGURE_COMMAND ""
|
||||||
|
BUILD_COMMAND "")
|
||||||
|
endif()
|
||||||
|
|
||||||
|
FetchContent_GetProperties(tml_fa4)
|
||||||
|
if(NOT tml_fa4_POPULATED)
|
||||||
|
FetchContent_Populate(tml_fa4)
|
||||||
|
endif()
|
||||||
|
message(STATUS "tml-fa4 is available at ${tml_fa4_SOURCE_DIR}")
|
||||||
|
|
||||||
|
add_custom_target(tml_fa4)
|
||||||
|
|
||||||
|
# Install into a private namespace so this implementation cannot shadow the
|
||||||
|
# flash_attn package used by vLLM's standard attention backends.
|
||||||
|
install(CODE "
|
||||||
|
file(GLOB_RECURSE TML_FA4_PY_FILES
|
||||||
|
\"${tml_fa4_SOURCE_DIR}/flash_attn/cute/*.py\")
|
||||||
|
foreach(SRC_FILE \${TML_FA4_PY_FILES})
|
||||||
|
file(RELATIVE_PATH REL_PATH
|
||||||
|
\"${tml_fa4_SOURCE_DIR}/flash_attn/cute\" \${SRC_FILE})
|
||||||
|
set(DST_FILE
|
||||||
|
\"\${CMAKE_INSTALL_PREFIX}/vllm/third_party/tml_fa4/\${REL_PATH}\")
|
||||||
|
get_filename_component(DST_DIR \${DST_FILE} DIRECTORY)
|
||||||
|
file(MAKE_DIRECTORY \${DST_DIR})
|
||||||
|
file(READ \${SRC_FILE} FILE_CONTENTS)
|
||||||
|
string(REPLACE
|
||||||
|
\"flash_attn.cute\"
|
||||||
|
\"vllm.third_party.tml_fa4\"
|
||||||
|
FILE_CONTENTS \"\${FILE_CONTENTS}\")
|
||||||
|
file(WRITE \${DST_FILE} \"\${FILE_CONTENTS}\")
|
||||||
|
endforeach()
|
||||||
|
" COMPONENT tml_fa4)
|
||||||
@@ -39,7 +39,7 @@ else()
|
|||||||
FetchContent_Declare(
|
FetchContent_Declare(
|
||||||
vllm-flash-attn
|
vllm-flash-attn
|
||||||
GIT_REPOSITORY https://github.com/vllm-project/flash-attention.git
|
GIT_REPOSITORY https://github.com/vllm-project/flash-attention.git
|
||||||
GIT_TAG bb9a72e7dde0dc614ffc663e052cd6a19ce73a42
|
GIT_TAG ed4b7342bc8f0489dd9b649d5288867e35fc6a32
|
||||||
GIT_PROGRESS TRUE
|
GIT_PROGRESS TRUE
|
||||||
# Don't share the vllm-flash-attn build between build types
|
# Don't share the vllm-flash-attn build between build types
|
||||||
BINARY_DIR ${CMAKE_BINARY_DIR}/vllm-flash-attn
|
BINARY_DIR ${CMAKE_BINARY_DIR}/vllm-flash-attn
|
||||||
|
|||||||
+21
-10
@@ -241,14 +241,15 @@ endmacro()
|
|||||||
# `<major>.<minor>`, dedupes them and then sorts them in ascending order and
|
# `<major>.<minor>`, dedupes them and then sorts them in ascending order and
|
||||||
# stores them in `OUT_ARCHES`.
|
# stores them in `OUT_ARCHES`.
|
||||||
#
|
#
|
||||||
# Example:
|
# Prefer `code=sm_*`; fall back to `arch=compute_*` for PTX-only flags.
|
||||||
# CUDA_ARCH_FLAGS="-gencode arch=compute_75,code=sm_75;...;-gencode arch=compute_90a,code=sm_90a"
|
# This handles mismatches such as `arch=compute_20,code=sm_121`.
|
||||||
# extract_unique_cuda_archs_ascending(OUT_ARCHES CUDA_ARCH_FLAGS)
|
|
||||||
# OUT_ARCHES="7.5;...;9.0"
|
|
||||||
function(extract_unique_cuda_archs_ascending OUT_ARCHES CUDA_ARCH_FLAGS)
|
function(extract_unique_cuda_archs_ascending OUT_ARCHES CUDA_ARCH_FLAGS)
|
||||||
set(_CUDA_ARCHES)
|
set(_CUDA_ARCHES)
|
||||||
foreach(_ARCH ${CUDA_ARCH_FLAGS})
|
foreach(_ARCH ${CUDA_ARCH_FLAGS})
|
||||||
string(REGEX MATCH "arch=compute_\([0-9]+[af]?\)" _COMPUTE ${_ARCH})
|
string(REGEX MATCH "code=sm_\([0-9]+[af]?\)" _COMPUTE ${_ARCH})
|
||||||
|
if (NOT _COMPUTE)
|
||||||
|
string(REGEX MATCH "arch=compute_\([0-9]+[af]?\)" _COMPUTE ${_ARCH})
|
||||||
|
endif()
|
||||||
if (_COMPUTE)
|
if (_COMPUTE)
|
||||||
set(_COMPUTE ${CMAKE_MATCH_1})
|
set(_COMPUTE ${CMAKE_MATCH_1})
|
||||||
endif()
|
endif()
|
||||||
@@ -396,14 +397,24 @@ function(cuda_archs_loose_intersection OUT_CUDA_ARCHS SRC_CUDA_ARCHS TGT_CUDA_AR
|
|||||||
# match — e.g. SRC="12.0f" matches TGT="12.1a" since SM121 is in the SM12x
|
# match — e.g. SRC="12.0f" matches TGT="12.1a" since SM121 is in the SM12x
|
||||||
# family. The output uses TGT's value to preserve the user's compilation flags.
|
# family. The output uses TGT's value to preserve the user's compilation flags.
|
||||||
set(_CUDA_ARCHS)
|
set(_CUDA_ARCHS)
|
||||||
|
# Resolve exact base matches before family fallbacks so a generic entry such
|
||||||
|
# as 10.0f cannot consume a 10.7 target that has a 10.7f source entry.
|
||||||
|
foreach(_arch ${_SRC_CUDA_ARCHS})
|
||||||
|
if(_arch MATCHES "[af]$")
|
||||||
|
string(REGEX REPLACE "[af]$" "" _base "${_arch}")
|
||||||
|
if("${_base}" IN_LIST _TGT_CUDA_ARCHS)
|
||||||
|
list(REMOVE_ITEM _SRC_CUDA_ARCHS "${_arch}")
|
||||||
|
list(REMOVE_ITEM _TGT_CUDA_ARCHS "${_base}")
|
||||||
|
list(APPEND _CUDA_ARCHS "${_arch}")
|
||||||
|
endif()
|
||||||
|
endif()
|
||||||
|
endforeach()
|
||||||
|
|
||||||
foreach(_arch ${_SRC_CUDA_ARCHS})
|
foreach(_arch ${_SRC_CUDA_ARCHS})
|
||||||
if(_arch MATCHES "[af]$")
|
if(_arch MATCHES "[af]$")
|
||||||
list(REMOVE_ITEM _SRC_CUDA_ARCHS "${_arch}")
|
list(REMOVE_ITEM _SRC_CUDA_ARCHS "${_arch}")
|
||||||
string(REGEX REPLACE "[af]$" "" _base "${_arch}")
|
string(REGEX REPLACE "[af]$" "" _base "${_arch}")
|
||||||
if ("${_base}" IN_LIST TGT_CUDA_ARCHS)
|
if("${_base}a" IN_LIST _TGT_CUDA_ARCHS)
|
||||||
list(REMOVE_ITEM _TGT_CUDA_ARCHS "${_base}")
|
|
||||||
list(APPEND _CUDA_ARCHS "${_arch}")
|
|
||||||
elseif("${_base}a" IN_LIST _TGT_CUDA_ARCHS)
|
|
||||||
list(REMOVE_ITEM _TGT_CUDA_ARCHS "${_base}a")
|
list(REMOVE_ITEM _TGT_CUDA_ARCHS "${_base}a")
|
||||||
list(APPEND _CUDA_ARCHS "${_base}a")
|
list(APPEND _CUDA_ARCHS "${_base}a")
|
||||||
elseif("${_base}f" IN_LIST _TGT_CUDA_ARCHS)
|
elseif("${_base}f" IN_LIST _TGT_CUDA_ARCHS)
|
||||||
@@ -487,7 +498,7 @@ endfunction()
|
|||||||
|
|
||||||
function(cuda_archs_sm90plus OUT_CUDA_ARCHS TGT_CUDA_ARCHS)
|
function(cuda_archs_sm90plus OUT_CUDA_ARCHS TGT_CUDA_ARCHS)
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
cuda_archs_loose_intersection(_archs "9.0a;10.0f;11.0f;12.0f" "${TGT_CUDA_ARCHS}")
|
cuda_archs_loose_intersection(_archs "9.0a;10.0f;10.7f;11.0f;12.0f" "${TGT_CUDA_ARCHS}")
|
||||||
else()
|
else()
|
||||||
cuda_archs_loose_intersection(_archs "9.0a;10.0a;10.1a;10.3a;12.0a;12.1a" "${TGT_CUDA_ARCHS}")
|
cuda_archs_loose_intersection(_archs "9.0a;10.0a;10.1a;10.3a;12.0a;12.1a" "${TGT_CUDA_ARCHS}")
|
||||||
endif()
|
endif()
|
||||||
|
|||||||
+1
-2
@@ -67,9 +67,8 @@ void cp_gather_and_upconvert_fp8_kv_cache(
|
|||||||
torch::Tensor const& src_cache, // [NUM_BLOCKS, BLOCK_SIZE, 656]
|
torch::Tensor const& src_cache, // [NUM_BLOCKS, BLOCK_SIZE, 656]
|
||||||
torch::Tensor const& dst, // [TOT_TOKENS, 576]
|
torch::Tensor const& dst, // [TOT_TOKENS, 576]
|
||||||
torch::Tensor const& block_table, // [BATCH, BLOCK_INDICES]
|
torch::Tensor const& block_table, // [BATCH, BLOCK_INDICES]
|
||||||
torch::Tensor const& seq_lens, // [BATCH]
|
|
||||||
torch::Tensor const& workspace_starts, // [BATCH]
|
torch::Tensor const& workspace_starts, // [BATCH]
|
||||||
int64_t batch_size);
|
int64_t batch_size, std::optional<torch::Tensor> seq_starts = std::nullopt);
|
||||||
|
|
||||||
// Indexer K quantization and cache function
|
// Indexer K quantization and cache function
|
||||||
void indexer_k_quant_and_cache(
|
void indexer_k_quant_and_cache(
|
||||||
|
|||||||
@@ -172,4 +172,15 @@
|
|||||||
|
|
||||||
#endif // __riscv_v
|
#endif // __riscv_v
|
||||||
|
|
||||||
|
// Power VSX
|
||||||
|
#ifdef __powerpc__
|
||||||
|
// FP32Vec16::exp() in cpu_types_vsx.hpp delegates to FP32Vec8::exp(), which
|
||||||
|
// implements a vectorised 5-term minimax polynomial using VSX intrinsics.
|
||||||
|
#define DEFINE_FAST_EXP \
|
||||||
|
auto fast_exp = [&](const vec_op::FP32Vec16& vec) \
|
||||||
|
__attribute__((always_inline)) { return vec.exp(); }; \
|
||||||
|
auto fast_exp_f16 = fast_exp;
|
||||||
|
|
||||||
|
#endif // __powerpc__
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -30,7 +30,8 @@ torch::Tensor get_scheduler_metadata(
|
|||||||
const torch::Tensor& query_start_loc, const bool causal,
|
const torch::Tensor& query_start_loc, const bool causal,
|
||||||
const int64_t window_size, const std::string& isa_hint,
|
const int64_t window_size, const std::string& isa_hint,
|
||||||
const bool enable_kv_split,
|
const bool enable_kv_split,
|
||||||
const std::optional<torch::Tensor>& dynamic_causal) {
|
const std::optional<torch::Tensor>& dynamic_causal,
|
||||||
|
const std::string& kv_cache_dtype) {
|
||||||
cpu_attention::ISA isa;
|
cpu_attention::ISA isa;
|
||||||
if (isa_hint == "amx") {
|
if (isa_hint == "amx") {
|
||||||
isa = cpu_attention::ISA::AMX;
|
isa = cpu_attention::ISA::AMX;
|
||||||
@@ -65,9 +66,11 @@ torch::Tensor get_scheduler_metadata(
|
|||||||
input.dynamic_causal =
|
input.dynamic_causal =
|
||||||
dynamic_causal.has_value() ? dynamic_causal->data_ptr<bool>() : nullptr;
|
dynamic_causal.has_value() ? dynamic_causal->data_ptr<bool>() : nullptr;
|
||||||
|
|
||||||
|
const int64_t kv_cache_idx =
|
||||||
|
static_cast<int64_t>(parse_fp8_kv_dtype(kv_cache_dtype));
|
||||||
VLLM_DISPATCH_FLOATING_TYPES(dtype, "get_scheduler_metadata", [&]() {
|
VLLM_DISPATCH_FLOATING_TYPES(dtype, "get_scheduler_metadata", [&]() {
|
||||||
CPU_ATTN_DISPATCH(head_dim, isa, 0, [&]() {
|
CPU_ATTN_DISPATCH(head_dim, isa, kv_cache_idx, [&]() {
|
||||||
input.elem_size = sizeof(scalar_t);
|
input.elem_size = sizeof(attn_impl::kv_cache_t);
|
||||||
input.q_buffer_elem_size = sizeof(attn_impl::q_buffer_t);
|
input.q_buffer_elem_size = sizeof(attn_impl::q_buffer_t);
|
||||||
input.logits_buffer_elem_size = sizeof(attn_impl::logits_buffer_t);
|
input.logits_buffer_elem_size = sizeof(attn_impl::logits_buffer_t);
|
||||||
input.output_buffer_elem_size =
|
input.output_buffer_elem_size =
|
||||||
|
|||||||
@@ -102,7 +102,9 @@ class TileGemm82 {
|
|||||||
kv_cache_t* __restrict__ curr_b = b_tile;
|
kv_cache_t* __restrict__ curr_b = b_tile;
|
||||||
|
|
||||||
for (int32_t k = 0; k < dynamic_k_size; ++k) {
|
for (int32_t k = 0; k < dynamic_k_size; ++k) {
|
||||||
auto [fp32_b_0_reg, fp32_b_1_reg] = load_b_pair_vec(curr_b);
|
auto fp32_b_regs = load_b_pair_vec(curr_b);
|
||||||
|
auto fp32_b_0_reg = fp32_b_regs.first;
|
||||||
|
auto fp32_b_1_reg = fp32_b_regs.second;
|
||||||
|
|
||||||
float* __restrict__ curr_m_a = curr_a;
|
float* __restrict__ curr_m_a = curr_a;
|
||||||
vec_op::unroll_loop<int32_t, M>([&](int32_t i) {
|
vec_op::unroll_loop<int32_t, M>([&](int32_t i) {
|
||||||
|
|||||||
+5
-187
@@ -1,5 +1,6 @@
|
|||||||
#include "cpu/cpu_types.hpp"
|
#include "cpu/cpu_types.hpp"
|
||||||
#include "cpu/utils.hpp"
|
#include "cpu/utils.hpp"
|
||||||
|
#include "cpu/cpu_fused_moe_activations.hpp"
|
||||||
#include "cpu/micro_gemm/cpu_micro_gemm_vec.hpp"
|
#include "cpu/micro_gemm/cpu_micro_gemm_vec.hpp"
|
||||||
#include "cpu/cpu_arch_macros.h"
|
#include "cpu/cpu_arch_macros.h"
|
||||||
|
|
||||||
@@ -43,193 +44,9 @@
|
|||||||
}()
|
}()
|
||||||
|
|
||||||
namespace {
|
namespace {
|
||||||
enum class FusedMOEAct {
|
|
||||||
SiluAndMul,
|
|
||||||
SwigluOAIAndMul,
|
|
||||||
GeluAndMul,
|
|
||||||
GeluTanhAndMul,
|
|
||||||
};
|
|
||||||
|
|
||||||
FusedMOEAct get_act_type(const std::string& act) {
|
using cpu_fused_moe_utils::apply_gated_act;
|
||||||
if (act == "silu") {
|
using cpu_fused_moe_utils::FusedMOEAct;
|
||||||
return FusedMOEAct::SiluAndMul;
|
|
||||||
} else if (act == "swigluoai") {
|
|
||||||
return FusedMOEAct::SwigluOAIAndMul;
|
|
||||||
} else if (act == "gelu") {
|
|
||||||
return FusedMOEAct::GeluAndMul;
|
|
||||||
} else if (act == "gelu_tanh") {
|
|
||||||
return FusedMOEAct::GeluTanhAndMul;
|
|
||||||
} else {
|
|
||||||
TORCH_CHECK(false, "Invalid act type: " + act);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
template <typename scalar_t>
|
|
||||||
void swigluoai_and_mul(float* __restrict__ input, scalar_t* __restrict__ output,
|
|
||||||
const int32_t m_size, const int32_t n_size,
|
|
||||||
const int32_t input_stride,
|
|
||||||
const int32_t output_stride) {
|
|
||||||
using scalar_vec_t = typename cpu_utils::VecTypeTrait<scalar_t>::vec_t;
|
|
||||||
#if !defined(__aarch64__)
|
|
||||||
// For GPT-OSS interleaved gate-up weights
|
|
||||||
alignas(64) static int32_t index[16] = {0, 2, 4, 6, 8, 10, 12, 14,
|
|
||||||
16, 18, 20, 22, 24, 26, 28, 30};
|
|
||||||
vec_op::INT32Vec16 index_vec(index);
|
|
||||||
#endif
|
|
||||||
vec_op::FP32Vec16 gate_up_max_vec(7.0);
|
|
||||||
vec_op::FP32Vec16 up_min_vec(-7.0);
|
|
||||||
vec_op::FP32Vec16 alpha_vec(1.702);
|
|
||||||
vec_op::FP32Vec16 one_vec(1.0);
|
|
||||||
|
|
||||||
DEFINE_FAST_EXP
|
|
||||||
|
|
||||||
for (int32_t m = 0; m < m_size; ++m) {
|
|
||||||
for (int32_t n = 0; n < n_size; n += 32) {
|
|
||||||
// Note: AdvSIMD does not support gather loads
|
|
||||||
#if defined(__aarch64__)
|
|
||||||
vec_op::FP32Vec16 gate_vec(vec_op::uninit);
|
|
||||||
vec_op::FP32Vec16 up_vec(vec_op::uninit);
|
|
||||||
vec_op::FP32Vec16::load_even_odd(input + n, gate_vec, up_vec);
|
|
||||||
#else
|
|
||||||
vec_op::FP32Vec16 gate_vec(input + n, index_vec);
|
|
||||||
vec_op::FP32Vec16 up_vec(input + n + 1, index_vec);
|
|
||||||
#endif
|
|
||||||
gate_vec = gate_vec.min(gate_up_max_vec);
|
|
||||||
up_vec = up_vec.clamp(up_min_vec, gate_up_max_vec);
|
|
||||||
auto sigmoid_vec = one_vec / (one_vec + fast_exp(-gate_vec * alpha_vec));
|
|
||||||
auto glu = gate_vec * sigmoid_vec;
|
|
||||||
auto gated_output_fp32 = (one_vec + up_vec) * glu;
|
|
||||||
scalar_vec_t gated_output = scalar_vec_t(gated_output_fp32);
|
|
||||||
gated_output.save(output + n / 2);
|
|
||||||
}
|
|
||||||
input += input_stride;
|
|
||||||
output += output_stride;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
template <typename scalar_t>
|
|
||||||
void silu_and_mul(float* __restrict__ input, scalar_t* __restrict__ output,
|
|
||||||
const int32_t m_size, const int32_t n_size,
|
|
||||||
const int32_t input_stride, const int32_t output_stride) {
|
|
||||||
using scalar_vec_t = typename cpu_utils::VecTypeTrait<scalar_t>::vec_t;
|
|
||||||
const int32_t dim = n_size / 2;
|
|
||||||
float* __restrict__ gate = input;
|
|
||||||
float* __restrict__ up = input + dim;
|
|
||||||
vec_op::FP32Vec16 one_vec(1.0);
|
|
||||||
|
|
||||||
DEFINE_FAST_EXP
|
|
||||||
|
|
||||||
for (int32_t m = 0; m < m_size; ++m) {
|
|
||||||
for (int32_t n = 0; n < dim; n += 16) {
|
|
||||||
vec_op::FP32Vec16 gate_vec(gate + n);
|
|
||||||
vec_op::FP32Vec16 up_vec(up + n);
|
|
||||||
auto sigmoid_vec = one_vec / (one_vec + fast_exp(-gate_vec));
|
|
||||||
auto silu = gate_vec * sigmoid_vec;
|
|
||||||
auto gated_output_fp32 = up_vec * silu;
|
|
||||||
scalar_vec_t gated_output = scalar_vec_t(gated_output_fp32);
|
|
||||||
gated_output.save(output + n);
|
|
||||||
}
|
|
||||||
gate += input_stride;
|
|
||||||
up += input_stride;
|
|
||||||
output += output_stride;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
template <typename scalar_t>
|
|
||||||
void gelu_and_mul(float* __restrict__ input, scalar_t* __restrict__ output,
|
|
||||||
const int32_t m_size, const int32_t n_size,
|
|
||||||
const int32_t input_stride, const int32_t output_stride) {
|
|
||||||
using scalar_vec_t = typename cpu_utils::VecTypeTrait<scalar_t>::vec_t;
|
|
||||||
const int32_t dim = n_size / 2;
|
|
||||||
float* __restrict__ gate = input;
|
|
||||||
float* __restrict__ up = input + dim;
|
|
||||||
vec_op::FP32Vec16 one_vec(1.0);
|
|
||||||
vec_op::FP32Vec16 w1_vec(M_SQRT1_2);
|
|
||||||
vec_op::FP32Vec16 w2_vec(0.5);
|
|
||||||
alignas(64) float temp[16];
|
|
||||||
|
|
||||||
DEFINE_FAST_EXP
|
|
||||||
|
|
||||||
for (int32_t m = 0; m < m_size; ++m) {
|
|
||||||
for (int32_t n = 0; n < dim; n += 16) {
|
|
||||||
vec_op::FP32Vec16 gate_vec(gate + n);
|
|
||||||
vec_op::FP32Vec16 up_vec(up + n);
|
|
||||||
auto er_input_vec = gate_vec * w1_vec;
|
|
||||||
|
|
||||||
er_input_vec.save(temp);
|
|
||||||
for (int32_t i = 0; i < 16; ++i) {
|
|
||||||
temp[i] = std::erf(temp[i]);
|
|
||||||
}
|
|
||||||
vec_op::FP32Vec16 er_vec(temp);
|
|
||||||
auto gelu = gate_vec * w2_vec * (one_vec + er_vec);
|
|
||||||
auto gated_output_fp32 = up_vec * gelu;
|
|
||||||
scalar_vec_t gated_output = scalar_vec_t(gated_output_fp32);
|
|
||||||
gated_output.save(output + n);
|
|
||||||
}
|
|
||||||
gate += input_stride;
|
|
||||||
up += input_stride;
|
|
||||||
output += output_stride;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
template <typename scalar_t>
|
|
||||||
void gelu_tanh_and_mul(float* __restrict__ input, scalar_t* __restrict__ output,
|
|
||||||
const int32_t m_size, const int32_t n_size,
|
|
||||||
const int32_t input_stride,
|
|
||||||
const int32_t output_stride) {
|
|
||||||
using scalar_vec_t = typename cpu_utils::VecTypeTrait<scalar_t>::vec_t;
|
|
||||||
const int32_t dim = n_size / 2;
|
|
||||||
float* __restrict__ gate = input;
|
|
||||||
float* __restrict__ up = input + dim;
|
|
||||||
vec_op::FP32Vec16 one_vec(1.0);
|
|
||||||
vec_op::FP32Vec16 w1_vec(0.7978845608028654);
|
|
||||||
vec_op::FP32Vec16 w2_vec(0.5);
|
|
||||||
vec_op::FP32Vec16 w3_vec(0.044715);
|
|
||||||
|
|
||||||
for (int32_t m = 0; m < m_size; ++m) {
|
|
||||||
for (int32_t n = 0; n < dim; n += 16) {
|
|
||||||
vec_op::FP32Vec16 gate_vec(gate + n);
|
|
||||||
vec_op::FP32Vec16 up_vec(up + n);
|
|
||||||
auto gate_pow3_vec = gate_vec * gate_vec * gate_vec;
|
|
||||||
auto inner_vec = w1_vec * (gate_vec + w3_vec * gate_pow3_vec);
|
|
||||||
// Note: can't use fast_exp form because diffusiongemma will generate
|
|
||||||
// wrong results
|
|
||||||
auto tanh_vec = inner_vec.tanh();
|
|
||||||
auto gelu_tanh = gate_vec * w2_vec * (one_vec + tanh_vec);
|
|
||||||
auto gated_output_fp32 = up_vec * gelu_tanh;
|
|
||||||
scalar_vec_t gated_output = scalar_vec_t(gated_output_fp32);
|
|
||||||
gated_output.save(output + n);
|
|
||||||
}
|
|
||||||
gate += input_stride;
|
|
||||||
up += input_stride;
|
|
||||||
output += output_stride;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
template <typename scalar_t>
|
|
||||||
FORCE_INLINE void apply_gated_act(const FusedMOEAct act,
|
|
||||||
float* __restrict__ input,
|
|
||||||
scalar_t* __restrict__ output,
|
|
||||||
const int32_t m, const int32_t n,
|
|
||||||
const int32_t input_stride,
|
|
||||||
const int32_t output_stride) {
|
|
||||||
switch (act) {
|
|
||||||
case FusedMOEAct::SwigluOAIAndMul:
|
|
||||||
swigluoai_and_mul(input, output, m, n, input_stride, output_stride);
|
|
||||||
return;
|
|
||||||
case FusedMOEAct::SiluAndMul:
|
|
||||||
silu_and_mul(input, output, m, n, input_stride, output_stride);
|
|
||||||
return;
|
|
||||||
case FusedMOEAct::GeluAndMul:
|
|
||||||
gelu_and_mul(input, output, m, n, input_stride, output_stride);
|
|
||||||
return;
|
|
||||||
case FusedMOEAct::GeluTanhAndMul:
|
|
||||||
gelu_tanh_and_mul(input, output, m, n, input_stride, output_stride);
|
|
||||||
return;
|
|
||||||
default:
|
|
||||||
TORCH_CHECK(false, "Unsupported act type.");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
template <typename scalar_t, typename gemm_t>
|
template <typename scalar_t, typename gemm_t>
|
||||||
void prepack_moe_weight_impl(scalar_t* __restrict__ weight_ptr,
|
void prepack_moe_weight_impl(scalar_t* __restrict__ weight_ptr,
|
||||||
@@ -817,6 +634,7 @@ void fused_moe_impl(scalar_t* __restrict__ output, scalar_t* __restrict__ input,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|
||||||
void prepack_moe_weight(
|
void prepack_moe_weight(
|
||||||
@@ -864,7 +682,7 @@ void cpu_fused_moe(
|
|||||||
const int32_t input_size_2 = w2.size(2);
|
const int32_t input_size_2 = w2.size(2);
|
||||||
const int32_t output_size_2 = w2.size(1);
|
const int32_t output_size_2 = w2.size(1);
|
||||||
const int32_t topk_num = topk_id.size(1);
|
const int32_t topk_num = topk_id.size(1);
|
||||||
const FusedMOEAct act_type = get_act_type(act);
|
const FusedMOEAct act_type = cpu_fused_moe_utils::get_act_type(act);
|
||||||
cpu_utils::ISA isa_type = cpu_utils::get_isa(isa);
|
cpu_utils::ISA isa_type = cpu_utils::get_isa(isa);
|
||||||
TORCH_CHECK(!skip_weighted || topk_num == 1,
|
TORCH_CHECK(!skip_weighted || topk_num == 1,
|
||||||
"skip_weighted is only supported for topk=1 on CPU");
|
"skip_weighted is only supported for topk=1 on CPU");
|
||||||
|
|||||||
@@ -0,0 +1,204 @@
|
|||||||
|
// SPDX-License-Identifier: Apache-2.0
|
||||||
|
// SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
|
||||||
|
#ifndef CPU_FUSED_MOE_ACTIVATIONS_HPP
|
||||||
|
#define CPU_FUSED_MOE_ACTIVATIONS_HPP
|
||||||
|
|
||||||
|
#include <cmath>
|
||||||
|
#include <cstdint>
|
||||||
|
#include <string>
|
||||||
|
|
||||||
|
#include "cpu/cpu_arch_macros.h"
|
||||||
|
#include "cpu/utils.hpp"
|
||||||
|
|
||||||
|
namespace cpu_fused_moe_utils {
|
||||||
|
enum class FusedMOEAct {
|
||||||
|
SiluAndMul,
|
||||||
|
SwigluOAIAndMul,
|
||||||
|
GeluAndMul,
|
||||||
|
GeluTanhAndMul,
|
||||||
|
};
|
||||||
|
|
||||||
|
inline FusedMOEAct get_act_type(const std::string& act) {
|
||||||
|
if (act == "silu") {
|
||||||
|
return FusedMOEAct::SiluAndMul;
|
||||||
|
} else if (act == "swigluoai") {
|
||||||
|
return FusedMOEAct::SwigluOAIAndMul;
|
||||||
|
} else if (act == "gelu") {
|
||||||
|
return FusedMOEAct::GeluAndMul;
|
||||||
|
} else if (act == "gelu_tanh") {
|
||||||
|
return FusedMOEAct::GeluTanhAndMul;
|
||||||
|
} else {
|
||||||
|
TORCH_CHECK(false, "Invalid act type: " + act);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
template <typename scalar_t>
|
||||||
|
void swigluoai_and_mul(float* __restrict__ input, scalar_t* __restrict__ output,
|
||||||
|
const int32_t m_size, const int32_t n_size,
|
||||||
|
const int32_t input_stride,
|
||||||
|
const int32_t output_stride) {
|
||||||
|
using scalar_vec_t = typename cpu_utils::VecTypeTrait<scalar_t>::vec_t;
|
||||||
|
#if !defined(__aarch64__)
|
||||||
|
// For GPT-OSS interleaved gate-up weights
|
||||||
|
alignas(64) static int32_t index[16] = {0, 2, 4, 6, 8, 10, 12, 14,
|
||||||
|
16, 18, 20, 22, 24, 26, 28, 30};
|
||||||
|
vec_op::INT32Vec16 index_vec(index);
|
||||||
|
#endif
|
||||||
|
vec_op::FP32Vec16 gate_up_max_vec(7.0);
|
||||||
|
vec_op::FP32Vec16 up_min_vec(-7.0);
|
||||||
|
vec_op::FP32Vec16 alpha_vec(1.702);
|
||||||
|
vec_op::FP32Vec16 one_vec(1.0);
|
||||||
|
|
||||||
|
DEFINE_FAST_EXP
|
||||||
|
|
||||||
|
for (int32_t m = 0; m < m_size; ++m) {
|
||||||
|
for (int32_t n = 0; n < n_size; n += 32) {
|
||||||
|
// Note: AdvSIMD does not support gather loads
|
||||||
|
#if defined(__aarch64__)
|
||||||
|
vec_op::FP32Vec16 gate_vec(vec_op::uninit);
|
||||||
|
vec_op::FP32Vec16 up_vec(vec_op::uninit);
|
||||||
|
vec_op::FP32Vec16::load_even_odd(input + n, gate_vec, up_vec);
|
||||||
|
#else
|
||||||
|
vec_op::FP32Vec16 gate_vec(input + n, index_vec);
|
||||||
|
vec_op::FP32Vec16 up_vec(input + n + 1, index_vec);
|
||||||
|
#endif
|
||||||
|
gate_vec = gate_vec.min(gate_up_max_vec);
|
||||||
|
up_vec = up_vec.clamp(up_min_vec, gate_up_max_vec);
|
||||||
|
auto sigmoid_vec = one_vec / (one_vec + fast_exp(-gate_vec * alpha_vec));
|
||||||
|
auto glu = gate_vec * sigmoid_vec;
|
||||||
|
auto gated_output_fp32 = (one_vec + up_vec) * glu;
|
||||||
|
scalar_vec_t gated_output = scalar_vec_t(gated_output_fp32);
|
||||||
|
gated_output.save(output + n / 2);
|
||||||
|
}
|
||||||
|
input += input_stride;
|
||||||
|
output += output_stride;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
template <typename scalar_t>
|
||||||
|
void silu_and_mul(float* __restrict__ input, scalar_t* __restrict__ output,
|
||||||
|
const int32_t m_size, const int32_t n_size,
|
||||||
|
const int32_t input_stride, const int32_t output_stride) {
|
||||||
|
using scalar_vec_t = typename cpu_utils::VecTypeTrait<scalar_t>::vec_t;
|
||||||
|
const int32_t dim = n_size / 2;
|
||||||
|
float* __restrict__ gate = input;
|
||||||
|
float* __restrict__ up = input + dim;
|
||||||
|
vec_op::FP32Vec16 one_vec(1.0);
|
||||||
|
|
||||||
|
DEFINE_FAST_EXP
|
||||||
|
|
||||||
|
for (int32_t m = 0; m < m_size; ++m) {
|
||||||
|
for (int32_t n = 0; n < dim; n += 16) {
|
||||||
|
vec_op::FP32Vec16 gate_vec(gate + n);
|
||||||
|
vec_op::FP32Vec16 up_vec(up + n);
|
||||||
|
auto sigmoid_vec = one_vec / (one_vec + fast_exp(-gate_vec));
|
||||||
|
auto silu = gate_vec * sigmoid_vec;
|
||||||
|
auto gated_output_fp32 = up_vec * silu;
|
||||||
|
scalar_vec_t gated_output = scalar_vec_t(gated_output_fp32);
|
||||||
|
gated_output.save(output + n);
|
||||||
|
}
|
||||||
|
gate += input_stride;
|
||||||
|
up += input_stride;
|
||||||
|
output += output_stride;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
template <typename scalar_t>
|
||||||
|
void gelu_and_mul(float* __restrict__ input, scalar_t* __restrict__ output,
|
||||||
|
const int32_t m_size, const int32_t n_size,
|
||||||
|
const int32_t input_stride, const int32_t output_stride) {
|
||||||
|
using scalar_vec_t = typename cpu_utils::VecTypeTrait<scalar_t>::vec_t;
|
||||||
|
const int32_t dim = n_size / 2;
|
||||||
|
float* __restrict__ gate = input;
|
||||||
|
float* __restrict__ up = input + dim;
|
||||||
|
vec_op::FP32Vec16 one_vec(1.0);
|
||||||
|
vec_op::FP32Vec16 w1_vec(M_SQRT1_2);
|
||||||
|
vec_op::FP32Vec16 w2_vec(0.5);
|
||||||
|
alignas(64) float temp[16];
|
||||||
|
|
||||||
|
DEFINE_FAST_EXP
|
||||||
|
|
||||||
|
for (int32_t m = 0; m < m_size; ++m) {
|
||||||
|
for (int32_t n = 0; n < dim; n += 16) {
|
||||||
|
vec_op::FP32Vec16 gate_vec(gate + n);
|
||||||
|
vec_op::FP32Vec16 up_vec(up + n);
|
||||||
|
auto er_input_vec = gate_vec * w1_vec;
|
||||||
|
|
||||||
|
er_input_vec.save(temp);
|
||||||
|
for (int32_t i = 0; i < 16; ++i) {
|
||||||
|
temp[i] = std::erf(temp[i]);
|
||||||
|
}
|
||||||
|
vec_op::FP32Vec16 er_vec(temp);
|
||||||
|
auto gelu = gate_vec * w2_vec * (one_vec + er_vec);
|
||||||
|
auto gated_output_fp32 = up_vec * gelu;
|
||||||
|
scalar_vec_t gated_output = scalar_vec_t(gated_output_fp32);
|
||||||
|
gated_output.save(output + n);
|
||||||
|
}
|
||||||
|
gate += input_stride;
|
||||||
|
up += input_stride;
|
||||||
|
output += output_stride;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
template <typename scalar_t>
|
||||||
|
void gelu_tanh_and_mul(float* __restrict__ input, scalar_t* __restrict__ output,
|
||||||
|
const int32_t m_size, const int32_t n_size,
|
||||||
|
const int32_t input_stride,
|
||||||
|
const int32_t output_stride) {
|
||||||
|
using scalar_vec_t = typename cpu_utils::VecTypeTrait<scalar_t>::vec_t;
|
||||||
|
const int32_t dim = n_size / 2;
|
||||||
|
float* __restrict__ gate = input;
|
||||||
|
float* __restrict__ up = input + dim;
|
||||||
|
vec_op::FP32Vec16 one_vec(1.0);
|
||||||
|
vec_op::FP32Vec16 w1_vec(0.7978845608028654);
|
||||||
|
vec_op::FP32Vec16 w2_vec(0.5);
|
||||||
|
vec_op::FP32Vec16 w3_vec(0.044715);
|
||||||
|
|
||||||
|
for (int32_t m = 0; m < m_size; ++m) {
|
||||||
|
for (int32_t n = 0; n < dim; n += 16) {
|
||||||
|
vec_op::FP32Vec16 gate_vec(gate + n);
|
||||||
|
vec_op::FP32Vec16 up_vec(up + n);
|
||||||
|
auto gate_pow3_vec = gate_vec * gate_vec * gate_vec;
|
||||||
|
auto inner_vec = w1_vec * (gate_vec + w3_vec * gate_pow3_vec);
|
||||||
|
// Note: can't use fast_exp form because diffusiongemma will generate
|
||||||
|
// wrong results
|
||||||
|
auto tanh_vec = inner_vec.tanh();
|
||||||
|
auto gelu_tanh = gate_vec * w2_vec * (one_vec + tanh_vec);
|
||||||
|
auto gated_output_fp32 = up_vec * gelu_tanh;
|
||||||
|
scalar_vec_t gated_output = scalar_vec_t(gated_output_fp32);
|
||||||
|
gated_output.save(output + n);
|
||||||
|
}
|
||||||
|
gate += input_stride;
|
||||||
|
up += input_stride;
|
||||||
|
output += output_stride;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
template <typename scalar_t>
|
||||||
|
FORCE_INLINE void apply_gated_act(const FusedMOEAct act,
|
||||||
|
float* __restrict__ input,
|
||||||
|
scalar_t* __restrict__ output,
|
||||||
|
const int32_t m, const int32_t n,
|
||||||
|
const int32_t input_stride,
|
||||||
|
const int32_t output_stride) {
|
||||||
|
switch (act) {
|
||||||
|
case FusedMOEAct::SwigluOAIAndMul:
|
||||||
|
swigluoai_and_mul(input, output, m, n, input_stride, output_stride);
|
||||||
|
return;
|
||||||
|
case FusedMOEAct::SiluAndMul:
|
||||||
|
silu_and_mul(input, output, m, n, input_stride, output_stride);
|
||||||
|
return;
|
||||||
|
case FusedMOEAct::GeluAndMul:
|
||||||
|
gelu_and_mul(input, output, m, n, input_stride, output_stride);
|
||||||
|
return;
|
||||||
|
case FusedMOEAct::GeluTanhAndMul:
|
||||||
|
gelu_tanh_and_mul(input, output, m, n, input_stride, output_stride);
|
||||||
|
return;
|
||||||
|
default:
|
||||||
|
TORCH_CHECK(false, "Unsupported act type.");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} // namespace cpu_fused_moe_utils
|
||||||
|
|
||||||
|
#endif
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user